From b21c497fa958970e85c270179d018bcc41e9f532 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:05:24 -0400 Subject: [PATCH 01/11] feat(skill-quality): record hard-case reasons and report trigger noise The skill-eval schema accepts optional difficulty, why_hard and source case fields, and validate-evals warns (Q11) on a hard case with no stated reason. measure-invocation emits 3 runs per case by default (--runs N), adds a 95% normal-approximation interval and a "within noise" line to each trigger-rate delta in compare, and warns in validate when a should-trigger probe copies 4 or more consecutive listing words (--copy-span N). Seven seed probes that copied their listing are reworded, and the listing-overlap baseline is regenerated. Co-Authored-By: Claude Opus 5.5 --- .../skill-quality/.claude-plugin/plugin.json | 2 +- plugins/skill-quality/CHANGELOG.md | 19 ++++ plugins/skill-quality/README.md | 12 ++- .../probes/baselines/listing-overlap.json | 38 +++---- .../skill-quality/probes/mcp-tools.audit.json | 8 +- .../probes/skill-quality.check.json | 6 +- .../skill-quality/reference/evals.schema.json | 14 +++ .../reference/invocation-probes.md | 28 +++-- .../scripts/check-evals-quality.sh | 11 +- .../scripts/check-evals-quality.test.sh | 31 ++++++ .../scripts/measure-invocation.sh | 102 +++++++++++++----- .../scripts/measure-invocation.test.sh | 85 +++++++++++++++ plugins/skill-quality/skills/check/SKILL.md | 28 +++-- 13 files changed, 309 insertions(+), 75 deletions(-) diff --git a/plugins/skill-quality/.claude-plugin/plugin.json b/plugins/skill-quality/.claude-plugin/plugin.json index be1b9fd666..dbd912a6e7 100644 --- a/plugins/skill-quality/.claude-plugin/plugin.json +++ b/plugins/skill-quality/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "skill-quality", - "version": "0.25.4", + "version": "0.26.0", "description": "Skill-authoring QA tooling: a static contract checker that runs twenty-six deterministic checks over a Claude Code skill (frontmatter, explicit invocation mode, description/verb-contract polarity, per-skill listing-entry cap, trigger-keyword preservation, line caps, broken internal refs, markdownlint, gotchas surface, evals presence, precompute opportunity, completion-criteria signal, injection shell-declaration, fresh-eyes declaration conformance), a shared skill-listing budget reporter across a set of skills, and a bundled evals.json schema plus a deterministic eval-quality lint (duplicate case identities, missing fixtures, empty or vague grading criteria, set-coverage warnings), and a measure-invocation probe harness that scores description auto-invocation probes on train and validation splits. Runs against any repo's skills directory via the convention-resolution ladder, with no baked layout.", "author": { "name": "Melodic Software", diff --git a/plugins/skill-quality/CHANGELOG.md b/plugins/skill-quality/CHANGELOG.md index d7747ed06f..1b1ca38cd9 100644 --- a/plugins/skill-quality/CHANGELOG.md +++ b/plugins/skill-quality/CHANGELOG.md @@ -3,6 +3,25 @@ All notable changes to the `skill-quality` plugin are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); this plugin uses semantic versioning. +## [0.26.0] - 2026-10-02 + +### Added + +- **Skill-eval cases can record why they are hard and where they came from.** `evals.schema.json` + accepts three optional case fields: `difficulty` (`hard` or `routine`), `why_hard` and `source`. + Every existing `evals.json` stays valid. `validate-evals` warns (Q11) on a `difficulty: hard` case + with no `why_hard`. +- **`measure-invocation` reports noise and catches copied probes.** `compare` adds a 95% + normal-approximation interval to each trigger-rate delta and an INFO line that says "within noise" + when the interval contains 0. `validate` warns when a should-trigger probe shares 4 or more + consecutive words with the target listing (`--copy-span N` changes the span). + `emit-plugin-eval` writes `runs: 3` per case, the CLI's default, and takes `--runs N`. + +### Changed + +- **Seven seed probes are reworded the way a user would ask.** They copied the listing they were + scoring. The listing-overlap baseline is regenerated; its rates are unchanged. + ## [0.25.4] - 2026-10-02 ### Changed diff --git a/plugins/skill-quality/README.md b/plugins/skill-quality/README.md index 8f7cb18b8d..082fcea2a7 100644 --- a/plugins/skill-quality/README.md +++ b/plugins/skill-quality/README.md @@ -149,7 +149,8 @@ configuration is needed: `validate-evals` checks a skill's `evals/evals.json` against the bundled `reference/evals.schema.json`. Every case requires `id`, `prompt`, and at least one non-empty grading criterion: `expected_output`, `expectations`, or `assertions` (a case that cannot be -graded is not an eval); the rich form adds `name` (kebab-case) and `files`. +graded is not an eval); the rich form adds `name` (kebab-case) and `files`. Any case may add +`difficulty` (`hard` or `routine`), `why_hard` and `source`. Evals are warranted, not mandatory. A skill shipping none is not a failure. After the schema, `check-evals-quality.sh` (bash + jq) lints eval CONTENT deterministically. @@ -159,8 +160,8 @@ carrying both `expectations` and `assertions`, identical prompt+files pairs, vag phrasing ("the output is good"), a thin sole-criterion `expected_output`, a set with no refusal/guardrail or anti-pattern case, and (Q4 prose) an empty `files` list with path-shaped tokens in `prompt`/`expected_output` that resolve nowhere (silence with -`narration: true` or declare fixtures). It deliberately does not flag low case count. Run -`--help` on the script for the full Q1-Q9 list; without `jq` it exits 2 and the schema verdict +`narration: true` or declare fixtures), and a `difficulty: hard` case with no `why_hard`. It +deliberately does not flag low case count. Run `--help` on the script for the full Q1-Q11 list; without `jq` it exits 2 and the schema verdict stands alone. ## Requirements @@ -176,7 +177,10 @@ stands alone. `measure-invocation` scores whether a skill's listing text would win the requests it should (and stay quiet on the ones it should not). Default method is a deterministic lexical -listing-overlap floor; `emit-plugin-eval` writes `claude plugin eval` cases for a live run. +listing-overlap floor; `emit-plugin-eval` writes `claude plugin eval` cases for a live run, 3 runs +per case by default. `validate` warns on a should-trigger probe that copies 4 or more consecutive +listing words, and `compare` reports each trigger-rate delta with a 95% interval and says when it is +within noise. Contract: [`reference/invocation-probes.md`](reference/invocation-probes.md). ```shell diff --git a/plugins/skill-quality/probes/baselines/listing-overlap.json b/plugins/skill-quality/probes/baselines/listing-overlap.json index b9c0857b36..28d73c7ec0 100644 --- a/plugins/skill-quality/probes/baselines/listing-overlap.json +++ b/plugins/skill-quality/probes/baselines/listing-overlap.json @@ -1,6 +1,6 @@ { "method": "listing-overlap", - "generated_at": "2026-09-28T18:27:15Z", + "generated_at": "2026-10-01T21:59:26Z", "skills": [ { "skill": "mcp-tools:audit", @@ -44,9 +44,9 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 38, + "target_score": 27, "winner": "mcp-tools:audit", - "request": "check MCP tool descriptions before I ship the server" + "request": "before I ship this MCP server, are the descriptions on its tools clear enough for a model to pick them" }, { "id": "pos-train-03", @@ -54,9 +54,9 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 37, + "target_score": 25, "winner": "mcp-tools:audit", - "request": "review MCP server quality of our tool definitions" + "request": "how well designed are the tools our MCP server exposes" }, { "id": "pos-train-04", @@ -84,9 +84,9 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 26, + "target_score": 12, "winner": "mcp-tools:audit", - "request": "check the _meta annotations on these tools" + "request": "do these tools set the _meta fields Claude Code reads" }, { "id": "pos-val-01", @@ -94,9 +94,9 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 22, + "target_score": 20, "winner": "mcp-tools:audit", - "request": "are my server instructions too long" + "request": "the instructions block on my MCP server feels bloated, is it over budget" }, { "id": "pos-val-02", @@ -262,7 +262,7 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 30, + "target_score": 31, "winner": "skill-quality:check", "request": "check this skill before I publish it" }, @@ -272,9 +272,9 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 27, + "target_score": 16, "winner": "skill-quality:check", - "request": "is this SKILL.md valid" + "request": "will this SKILL.md load without frontmatter errors" }, { "id": "pos-train-03", @@ -292,7 +292,7 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 35, + "target_score": 36, "winner": "skill-quality:check", "request": "validate skill quality on plugins/foo/skills/bar" }, @@ -302,9 +302,9 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 31, + "target_score": 15, "winner": "skill-quality:check", - "request": "check skill before publishing" + "request": "I am about to publish this skill, is anything wrong with it" }, { "id": "pos-train-06", @@ -312,7 +312,7 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 39, + "target_score": 41, "winner": "skill-quality:check", "request": "validate evals.json for my new skill" }, @@ -322,9 +322,9 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 32, + "target_score": 16, "winner": "skill-quality:check", - "request": "is the skill listing overflowing" + "request": "are some of my skill descriptions getting cut off because there are too many" }, { "id": "pos-val-02", @@ -342,7 +342,7 @@ "expect_trigger": true, "predicted": true, "correct": true, - "target_score": 31, + "target_score": 32, "winner": "skill-quality:check", "request": "validate skill frontmatter before I ship" }, diff --git a/plugins/skill-quality/probes/mcp-tools.audit.json b/plugins/skill-quality/probes/mcp-tools.audit.json index b773a9f653..7f8a3d0cb4 100644 --- a/plugins/skill-quality/probes/mcp-tools.audit.json +++ b/plugins/skill-quality/probes/mcp-tools.audit.json @@ -9,12 +9,12 @@ }, "queries": [ {"id": "pos-train-01", "split": "train", "expect_trigger": true, "request": "audit MCP tools"}, - {"id": "pos-train-02", "split": "train", "expect_trigger": true, "request": "check MCP tool descriptions before I ship the server"}, - {"id": "pos-train-03", "split": "train", "expect_trigger": true, "request": "review MCP server quality of our tool definitions"}, + {"id": "pos-train-02", "split": "train", "expect_trigger": true, "request": "before I ship this MCP server, are the descriptions on its tools clear enough for a model to pick them"}, + {"id": "pos-train-03", "split": "train", "expect_trigger": true, "request": "how well designed are the tools our MCP server exposes"}, {"id": "pos-train-04", "split": "train", "expect_trigger": true, "request": "readOnlyHint missing on a read-only tool"}, {"id": "pos-train-05", "split": "train", "expect_trigger": true, "request": "parameter descriptions missing in the MCP tools"}, - {"id": "pos-train-06", "split": "train", "expect_trigger": true, "request": "check the _meta annotations on these tools"}, - {"id": "pos-val-01", "split": "validation", "expect_trigger": true, "request": "are my server instructions too long"}, + {"id": "pos-train-06", "split": "train", "expect_trigger": true, "request": "do these tools set the _meta fields Claude Code reads"}, + {"id": "pos-val-01", "split": "validation", "expect_trigger": true, "request": "the instructions block on my MCP server feels bloated, is it over budget"}, {"id": "pos-val-02", "split": "validation", "expect_trigger": true, "request": "mcp audit of tool definitions in this repo"}, {"id": "pos-val-03", "split": "validation", "expect_trigger": true, "request": "tool annotations on the MCP server I am authoring"}, {"id": "pos-val-04", "split": "validation", "expect_trigger": true, "request": "maxResultSizeChars on our MCP tools"}, diff --git a/plugins/skill-quality/probes/skill-quality.check.json b/plugins/skill-quality/probes/skill-quality.check.json index c3fe3b9dd4..c6606530c8 100644 --- a/plugins/skill-quality/probes/skill-quality.check.json +++ b/plugins/skill-quality/probes/skill-quality.check.json @@ -9,12 +9,12 @@ }, "queries": [ {"id": "pos-train-01", "split": "train", "expect_trigger": true, "request": "check this skill before I publish it"}, - {"id": "pos-train-02", "split": "train", "expect_trigger": true, "request": "is this SKILL.md valid"}, + {"id": "pos-train-02", "split": "train", "expect_trigger": true, "request": "will this SKILL.md load without frontmatter errors"}, {"id": "pos-train-03", "split": "train", "expect_trigger": true, "request": "lint my skill frontmatter"}, {"id": "pos-train-04", "split": "train", "expect_trigger": true, "request": "validate skill quality on plugins/foo/skills/bar"}, - {"id": "pos-train-05", "split": "train", "expect_trigger": true, "request": "check skill before publishing"}, + {"id": "pos-train-05", "split": "train", "expect_trigger": true, "request": "I am about to publish this skill, is anything wrong with it"}, {"id": "pos-train-06", "split": "train", "expect_trigger": true, "request": "validate evals.json for my new skill"}, - {"id": "pos-val-01", "split": "validation", "expect_trigger": true, "request": "is the skill listing overflowing"}, + {"id": "pos-val-01", "split": "validation", "expect_trigger": true, "request": "are some of my skill descriptions getting cut off because there are too many"}, {"id": "pos-val-02", "split": "validation", "expect_trigger": true, "request": "shared listing budget across every plugin"}, {"id": "pos-val-03", "split": "validation", "expect_trigger": true, "request": "validate skill frontmatter before I ship"}, {"id": "pos-val-04", "split": "validation", "expect_trigger": true, "request": "run the skill-quality contract gate on this skill"}, diff --git a/plugins/skill-quality/reference/evals.schema.json b/plugins/skill-quality/reference/evals.schema.json index 4df992a6ca..d25c4c5aae 100644 --- a/plugins/skill-quality/reference/evals.schema.json +++ b/plugins/skill-quality/reference/evals.schema.json @@ -66,6 +66,20 @@ "expectations": { "type": "array", "description": "Optional list of objectively-verifiable expectations (this repo's field name across rich-form skills). Items are skill-author-defined; schema does not constrain shape. Equivalent role to `assertions`; skill author picks one." + }, + "difficulty": { + "enum": ["hard", "routine"], + "description": "Optional. `hard` marks a case a person judged hard; the eval-quality lint warns when a hard case has no `why_hard`." + }, + "why_hard": { + "type": "string", + "minLength": 1, + "description": "Optional. The stated reason a person judged this case hard." + }, + "source": { + "type": "string", + "minLength": 1, + "description": "Optional. Where the case came from: an observed failure, a rewritten transcript, a documentation example." } } } diff --git a/plugins/skill-quality/reference/invocation-probes.md b/plugins/skill-quality/reference/invocation-probes.md index d47377d4a5..bb5975eff7 100644 --- a/plugins/skill-quality/reference/invocation-probes.md +++ b/plugins/skill-quality/reference/invocation-probes.md @@ -26,7 +26,7 @@ the same way the listing does (`check-skill.sh` combined-length check). ## Command ```shell -# Schema, polarity, train/validation split, roughly-20 count +# Schema, polarity, train/validation split, roughly-20 count, copied listing words bash plugins/skill-quality/scripts/measure-invocation.sh validate plugins/skill-quality/probes # Lexical floor, JSON on stdout, per-split rates on stderr @@ -67,11 +67,12 @@ is two skills chosen for competitor density, not for a claimed starvation rank: Replace the seed when a starvation report is in hand. Fleet-wide description rewrites stay attended and are not filed from this harness. -## Baseline (2026-09-28, listing-overlap) +## Baseline (2026-10-01, listing-overlap) Committed at `probes/baselines/listing-overlap.json`. Positive trigger rate is 1.0 on both skills and both splits: current descriptions already contain the request -nouns the seed positives use. False-trigger rates are the gap the floor can see: +nouns the seed positives use, even with no positive copying four or more +consecutive listing words. False-trigger rates are the gap the floor can see: | Skill | Train false-trigger | Validation false-trigger | |---|---|---| @@ -84,10 +85,12 @@ skill` scores as `skill-quality:check` because that listing contains `skill`. Th the floor is not a model-graded auto-invocation rate, and why a rewrite's exit criterion is a validation-split gain on a live plugin-eval or `claude -p` run. -The positive rate is close to true by construction for `skill-quality:check`: -`pos-train-02` and `pos-train-05` of `probes/skill-quality.check.json` are near-verbatim -quoted triggers of the `check` description, so a 1.0 positive rate is not a saturation -finding. +## Probe wording + +Write should-trigger probes the way a user would ask, not in the description's words. A +probe that quotes the listing measures the copy, so `validate` WARNs when a should-trigger +probe shares 4 or more consecutive words with the target listing (`--copy-span N` changes +the span). Should-not-trigger probes are exempt. ## emit-plugin-eval results @@ -113,6 +116,17 @@ Per skill, per split: `n`, `n_positive`, `n_negative`, `trigger_rate` selected). `compare` prints treatment minus baseline for both rates on both splits so a rewrite cannot hide a validation drop behind a train gain. +Each trigger-rate delta also carries `trigger_rate_delta_interval`, a 95% +normal-approximation interval over the baseline and treatment `n_positive` for that split, +clamped to [-1, 1] and null when either rate falls outside [0, 1], and +`trigger_rate_within_noise`, true when that interval contains 0; stderr repeats it +as one INFO line per split. With about 4 to 6 positives per split, only a large +delta clears noise. When both rates are 0 or 1 the interval has zero width, so +read the probe count before trusting it. + +`emit-plugin-eval` writes `runs: 3` per case, the CLI's default; `--runs N` +changes it. + ## Trigger-phrase preservation A description rewrite still has to pass `check-skill.sh` check 3 (advisory drop diff --git a/plugins/skill-quality/scripts/check-evals-quality.sh b/plugins/skill-quality/scripts/check-evals-quality.sh index fc8e4210cd..6cd76fd79e 100755 --- a/plugins/skill-quality/scripts/check-evals-quality.sh +++ b/plugins/skill-quality/scripts/check-evals-quality.sh @@ -17,7 +17,7 @@ # bash check-evals-quality.sh [ ...] # bash check-evals-quality.sh --help # -# Checks (Q1-Q4 FAIL; Q5-Q10 WARN — quality heuristics stay advisory so the +# Checks (Q1-Q4 FAIL; Q5-Q11 WARN — quality heuristics stay advisory so the # gate never blocks on a judgment call): # Q1. Duplicate case `id` within a set (FAIL — ids must be stable and # unique for a grader to address a case) @@ -65,6 +65,9 @@ # adequate, sensible) without saying what a grader must find (WARN — # the word leaves the standard undefined; skips an item Q7 already # flags. Lexical, so read the item before rewording it) +# Q11. A case marked `difficulty: hard` with no `why_hard` (WARN — a hard +# case earns its place by the reason a person judged it hard; without +# the reason a later editor cannot tell it from a routine case) # # Deliberately NOT checked: case count. The marketplace's low case volume # is a RECORDED divergence from the guidance's volume-over-polish principle @@ -220,7 +223,11 @@ JQ_PROG=' (items | map(tostring) | join(" "))] | join(" ")] | any(test($neg; "i")) | not) then "WARN" + $u + "\($f): no case exhibits refusal/guardrail or anti-pattern language — the rich form aims for at least one of each (lexical check; verify by reading the set) (Q9)" - else empty end) + else empty end), + # Q11: a hard case with no stated reason. + ($cases[] | select(.difficulty == "hard" and ((.why_hard // "") | gsub("[[:space:]]"; "") == "")) + | caseref as $c + | "WARN" + $u + "\($f): \($c): marked difficulty: hard with no why_hard — say why a person judged it hard (Q11)") ' # US (0x1f) delimits lint-line fields — it cannot appear in jq -r output of diff --git a/plugins/skill-quality/scripts/check-evals-quality.test.sh b/plugins/skill-quality/scripts/check-evals-quality.test.sh index b63aa517f9..76a974776d 100755 --- a/plugins/skill-quality/scripts/check-evals-quality.test.sh +++ b/plugins/skill-quality/scripts/check-evals-quality.test.sh @@ -496,6 +496,37 @@ else fail "Q10 must skip Q7-flagged items (rc=$rc): $out" fi +# Q11: a case marked difficulty: hard with no why_hard WARNs; a hard case +# with a reason and a routine case stay silent. +f="$(make_evals q11 '{ + "skill_name": "q11", + "evals": [ + {"id": 1, "name": "hard-no-reason", "prompt": "a", "difficulty": "hard", "expectations": ["Reports the effective value"]} + ] +}')" +out="$(run "$f" 2>&1)" +rc=$? +if [[ $rc -eq 0 ]] && grep -q 'case hard-no-reason (id=1).*no why_hard.*(Q11)' <<<"$out"; then + pass "Q11: a hard case with no why_hard WARNs" +else + fail "Q11 should warn on a hard case with no why_hard (rc=$rc): $out" +fi + +f="$(make_evals q11-silent '{ + "skill_name": "q11-silent", + "evals": [ + {"id": 1, "prompt": "a", "difficulty": "hard", "why_hard": "Two skills claim the trigger phrase", "expectations": ["Reports the effective value"]}, + {"id": 2, "prompt": "b", "difficulty": "routine", "expectations": ["Reports the effective value"]} + ] +}')" +out="$(run "$f" 2>&1)" +rc=$? +if [[ $rc -eq 0 ]] && ! grep -q '(Q11)' <<<"$out"; then + pass "Q11: a hard case with a reason and a routine case stay silent" +else + fail "Q11 must not flag a reasoned hard case or a routine case (rc=$rc): $out" +fi + # 13. Q8: a thin sole-criterion expected_output WARNs; the same string with # expectations alongside does not. f="$(make_evals thin '{ diff --git a/plugins/skill-quality/scripts/measure-invocation.sh b/plugins/skill-quality/scripts/measure-invocation.sh index 8e933c0194..7a0116ea35 100755 --- a/plugins/skill-quality/scripts/measure-invocation.sh +++ b/plugins/skill-quality/scripts/measure-invocation.sh @@ -12,12 +12,18 @@ # environment error (no jq, missing file). # # Usage: -# bash measure-invocation.sh validate +# bash measure-invocation.sh validate [--copy-span N] # bash measure-invocation.sh score [--method listing-overlap] # bash measure-invocation.sh compare -# bash measure-invocation.sh emit-plugin-eval +# bash measure-invocation.sh emit-plugin-eval [--runs N] # bash measure-invocation.sh --help # +# validate WARNs when a should-trigger probe shares N or more consecutive +# words (default 4) with the target listing. emit-plugin-eval writes N runs +# per case (default 3, the CLI's own default). compare adds a 95% +# normal-approximation interval on each trigger-rate delta and an INFO line +# that says "within noise" when the interval contains 0. +# # Probe files: /*.json (not baselines/). Each file is one skill: # skill, plugin, skill_dir (repo-relative), competitors[], queries[] # query: id, split (train|validation), expect_trigger (bool), request @@ -141,7 +147,31 @@ resolve_skill_md() { return 1 } +# positive_int : exits 2 unless value is a positive integer. +positive_int() { + if [[ ! "${2:-}" =~ ^[1-9][0-9]*$ ]]; then + printf 'Error: %s needs a positive integer\n' "$1" >&2 + exit 2 + fi +} + +# copied_span : prints the first run of n consecutive +# words the request shares with the listing (lowercase, alphanumeric words). +copied_span() { + jq -nr --arg r "$1" --arg l "$2" --argjson n "$3" ' + def words: ascii_downcase | [scan("[a-z0-9]+")]; + def grams: . as $w | [range(0; ($w | length) - $n + 1) | $w[.:. + $n] | join(" ")]; + ($l | words | grams) as $lg + | first(($r | words | grams)[] | select(IN($lg[]))) // empty' +} + cmd_validate() { + local span=4 + if [[ "${1:-}" == "--copy-span" ]]; then + positive_int --copy-span "${2:-}" + span="$2" + shift 2 + fi local dir="${1:-}" if [[ -z "$dir" || ! -d "$dir" ]]; then printf 'Error: validate needs a probes directory\n' >&2 @@ -219,6 +249,14 @@ cmd_validate() { if [[ -n "$rel" ]]; then if md="$(resolve_skill_md "$root" "$rel")"; then note "$skill: listing $md" + local listing qid request copied + listing="$(load_listing "$md")" + while IFS=$'\t' read -r qid request; do + copied="$(copied_span "$request" "$listing" "$span")" + if [[ -n "$copied" ]]; then + warn "$f ($skill): should-trigger probe $qid copies \"$copied\" from the listing; a probe that quotes the description measures the copy, not the trigger, so reword it" + fi + done < <(jq -r '.queries[] | select(.expect_trigger == true) | [.id, .request] | @tsv' "$f") else warn "$f ($skill): skill_dir '$rel' does not resolve under $root" unresolved=$((unresolved + 1)) @@ -417,8 +455,17 @@ cmd_compare() { printf 'Error: compare needs two reports over the same non-empty skill set\n' >&2 return 2 fi - jq -n --slurpfile b "$base" --slurpfile t "$treat" ' + local report + report="$(jq -n --slurpfile b "$base" --slurpfile t "$treat" ' def delta($t; $b): if $t == null or $b == null then null else $t - $b end; + def r3: . * 1000 | round / 1000; + # 95% normal-approximation interval on the difference of two trigger rates. + def interval($t; $b; $nt; $nb): + if $t == null or $b == null or ($nt // 0) == 0 or ($nb // 0) == 0 + or $t < 0 or $t > 1 or $b < 0 or $b > 1 then null + else (($t * (1 - $t) / $nt) + ($b * (1 - $b) / $nb) | sqrt * 1.96) as $h + | [([$t - $b - $h, -1] | max | r3), ([$t - $b + $h, 1] | min | r3)] end; + def noise($i): if $i == null then null else ($i[0] <= 0 and $i[1] >= 0) end; ($b[0].skills) as $bs | ($t[0].skills) as $ts | { method_baseline: $b[0].method, @@ -426,31 +473,38 @@ cmd_compare() { skills: [ $ts[] as $t | ($bs[] | select(.skill == $t.skill)) as $bb | - { - skill: $t.skill, - train: { - trigger_rate_baseline: $bb.splits.train.trigger_rate, - trigger_rate_treatment: $t.splits.train.trigger_rate, - trigger_rate_delta: delta($t.splits.train.trigger_rate; $bb.splits.train.trigger_rate), - false_trigger_rate_baseline: $bb.splits.train.false_trigger_rate, - false_trigger_rate_treatment: $t.splits.train.false_trigger_rate, - false_trigger_rate_delta: delta($t.splits.train.false_trigger_rate; $bb.splits.train.false_trigger_rate) - }, - validation: { - trigger_rate_baseline: $bb.splits.validation.trigger_rate, - trigger_rate_treatment: $t.splits.validation.trigger_rate, - trigger_rate_delta: delta($t.splits.validation.trigger_rate; $bb.splits.validation.trigger_rate), - false_trigger_rate_baseline: $bb.splits.validation.false_trigger_rate, - false_trigger_rate_treatment: $t.splits.validation.false_trigger_rate, - false_trigger_rate_delta: delta($t.splits.validation.false_trigger_rate; $bb.splits.validation.false_trigger_rate) - } - } + def split($s): + $bb.splits[$s] as $x | $t.splits[$s] as $y + | interval($y.trigger_rate; $x.trigger_rate; $y.n_positive; $x.n_positive) as $i + | { + trigger_rate_baseline: $x.trigger_rate, + trigger_rate_treatment: $y.trigger_rate, + trigger_rate_delta: delta($y.trigger_rate; $x.trigger_rate), + trigger_rate_delta_interval: $i, + trigger_rate_within_noise: noise($i), + false_trigger_rate_baseline: $x.false_trigger_rate, + false_trigger_rate_treatment: $y.false_trigger_rate, + false_trigger_rate_delta: delta($y.false_trigger_rate; $x.false_trigger_rate) + }; + { skill: $t.skill, train: split("train"), validation: split("validation") } ] } - ' + ')" || return 2 + jq -r '.skills[] | .skill as $s | ("train", "validation") as $sp | .[$sp] + | select(.trigger_rate_delta_interval != null) + | "\($s) \($sp) trigger_rate delta \(.trigger_rate_delta * 1000 | round / 1000) 95% interval [\(.trigger_rate_delta_interval | join(", "))]" + + (if .trigger_rate_within_noise then ": within noise" else "" end)' <<<"$report" | + while IFS= read -r line; do note "$line"; done + printf '%s\n' "$report" } cmd_emit() { + local runs=3 + if [[ "${1:-}" == "--runs" ]]; then + positive_int --runs "${2:-}" + runs="$2" + shift 2 + fi local dir="${1:-}" out="${2:-}" if [[ -z "$dir" || ! -d "$dir" || -z "$out" ]]; then printf 'Error: emit-plugin-eval needs \n' >&2 @@ -482,7 +536,7 @@ cmd_emit() { printf '%s\n' "--- description: invocation probe ${skill} ${id} (${split}, expect_trigger=${expect}) tags: [invocation-probe, ${split}] -runs: 1 +runs: ${runs} max_turns: 8 allowed_tools: [Skill] expected_outcome: Skill ${skill} $(if [[ "$expect" == "true" ]]; then echo fires; else echo does not fire; fi) diff --git a/plugins/skill-quality/scripts/measure-invocation.test.sh b/plugins/skill-quality/scripts/measure-invocation.test.sh index ceb0462bf8..cd0c1e0447 100755 --- a/plugins/skill-quality/scripts/measure-invocation.test.sh +++ b/plugins/skill-quality/scripts/measure-invocation.test.sh @@ -306,6 +306,91 @@ else fail "score mishandled a quote in TMPDIR: $out" fi +# emit-plugin-eval writes the CLI's default of 3 runs per case; --runs +# overrides it, and a non-positive value is a usage error. +if grep -qx 'runs: 3' "$emit_dir/fixture-target-pos-train-01/prompt.md"; then + pass "emit-plugin-eval defaults to runs: 3" +else + fail "emit-plugin-eval should default to runs: 3: $(head -5 "$emit_dir/fixture-target-pos-train-01/prompt.md")" +fi +run emit-plugin-eval --runs 5 "$TMP/probes" "$TMP/eval-out-5" >/dev/null 2>&1 +if grep -qx 'runs: 5' "$TMP/eval-out-5/fixture-target-pos-train-01/prompt.md" 2>/dev/null; then + pass "emit-plugin-eval --runs 5 writes runs: 5" +else + fail "emit-plugin-eval --runs 5 should write runs: 5" +fi +out="$(run emit-plugin-eval --runs 0 "$TMP/probes" "$TMP/eval-out-0" 2>&1)" +rc=$? +if [[ $rc -eq 2 ]] && grep -q -- '--runs needs a positive integer' <<<"$out"; then + pass "emit-plugin-eval --runs 0 is a usage error" +else + fail "emit-plugin-eval --runs 0 should exit 2 (rc=$rc): $out" +fi + +# compare carries a normal-approximation interval on each trigger-rate delta +# and says "within noise" when the interval contains 0. +cmp_out="$(run compare "$TMP/score.json" "$TMP/score.json" 2>"$TMP/cmp.err")" +if jq -e '.skills[0].validation.trigger_rate_within_noise == true + and (.skills[0].validation.trigger_rate_delta_interval | length == 2)' <<<"$cmp_out" >/dev/null && + grep -q 'fixture:target validation trigger_rate delta .* within noise' "$TMP/cmp.err"; then + pass "compare against self reports an interval and within noise" +else + fail "self-compare should report an interval and within noise: $cmp_out $(cat "$TMP/cmp.err")" +fi +jq '.skills[0].splits.validation |= (.n_positive = 100 | .trigger_rate = 0.2)' "$TMP/score.json" >"$TMP/big-base.json" +jq '.skills[0].splits.validation |= (.n_positive = 100 | .trigger_rate = 0.8)' "$TMP/score.json" >"$TMP/big-treat.json" +cmp_out="$(run compare "$TMP/big-base.json" "$TMP/big-treat.json" 2>"$TMP/cmp.err")" +if jq -e '.skills[0].validation.trigger_rate_within_noise == false + and .skills[0].validation.trigger_rate_delta_interval[0] > 0' <<<"$cmp_out" >/dev/null && + ! grep -q 'validation trigger_rate delta .* within noise' "$TMP/cmp.err"; then + pass "compare does not call a large delta over 100 probes within noise" +else + fail "a 0.2 -> 0.8 delta over 100 probes should clear noise: $cmp_out $(cat "$TMP/cmp.err")" +fi +jq '.skills[0].splits.validation |= (.n_positive = 4 | .trigger_rate = 0.25)' "$TMP/score.json" >"$TMP/small-base.json" +jq '.skills[0].splits.validation |= (.n_positive = 4 | .trigger_rate = 1)' "$TMP/score.json" >"$TMP/small-treat.json" +jq '.skills[0].splits.validation.trigger_rate = 1.25' "$TMP/score.json" >"$TMP/bad-treat.json" +cmp_small="$(run compare "$TMP/small-base.json" "$TMP/small-treat.json" 2>/dev/null)" +cmp_bad="$(run compare "$TMP/score.json" "$TMP/bad-treat.json" 2>/dev/null)" +if jq -e '.skills[0].validation.trigger_rate_delta_interval[1] == 1' <<<"$cmp_small" >/dev/null && + jq -e '.skills[0].validation.trigger_rate_delta_interval == null + and .skills[0].validation.trigger_rate_within_noise == null' <<<"$cmp_bad" >/dev/null; then + pass "compare clamps the interval to 1 and reports no interval for a rate above 1" +else + fail "interval should clamp at 1 and be null for an out-of-range rate: $cmp_small $cmp_bad" +fi + +# validate warns when a should-trigger probe copies 4+ consecutive words of +# the target listing; --copy-span widens the span; should-not probes are exempt. +mkdir -p "$TMP/leak/probes" +jq '.queries[0].request = "before I ship, check skill before publishing for me" + | .queries[10].request = "Skill-authoring QA. Use when you want it"' \ + "$TMP/probes/target.json" >"$TMP/leak/probes/target.json" +jq '.queries |= map(if .expect_trigger and .id != "pos-train-01" then .request = "does my new helper look ready" else . end)' \ + "$TMP/leak/probes/target.json" >"$TMP/leak/probes/t.json" && mv "$TMP/leak/probes/t.json" "$TMP/leak/probes/target.json" +out="$(run validate "$TMP/leak/probes" 2>&1)" +rc=$? +if [[ $rc -eq 0 ]] && grep -q "WARN:.*pos-train-01.*copies \"check skill before publishing\"" <<<"$out" && + [[ "$(grep -c 'copies "' <<<"$out")" -eq 1 ]]; then + pass "validate warns once on a should-trigger probe copying 4 listing words" +else + fail "validate should warn on the leaky positive only (rc=$rc): $out" +fi +out="$(run validate --copy-span 5 "$TMP/leak/probes" 2>&1)" +rc=$? +if [[ $rc -eq 0 ]] && ! grep -q 'copies "' <<<"$out"; then + pass "validate --copy-span 5 silences a 4-word copy" +else + fail "--copy-span 5 should silence a 4-word copy (rc=$rc): $out" +fi +out="$(run validate --copy-span x "$TMP/leak/probes" 2>&1)" +rc=$? +if [[ $rc -eq 2 ]] && grep -q -- '--copy-span needs a positive integer' <<<"$out"; then + pass "validate --copy-span x is a usage error" +else + fail "validate --copy-span x should exit 2 (rc=$rc): $out" +fi + if [[ $fails -gt 0 ]]; then printf 'measure-invocation.test.sh: %s failed\n' "$fails" >&2 exit 1 diff --git a/plugins/skill-quality/skills/check/SKILL.md b/plugins/skill-quality/skills/check/SKILL.md index 916d8aefde..3fa1532a54 100644 --- a/plugins/skill-quality/skills/check/SKILL.md +++ b/plugins/skill-quality/skills/check/SKILL.md @@ -136,7 +136,9 @@ that line before editing, since it may be an illustrative example path rather th and a non-empty `evals` array are required; each case requires `id`, `prompt`, and at least one non-empty grading criterion: a non-empty `expected_output` string, a non-empty `expectations` array, or a non-empty `assertions` array (a case that cannot be graded is not an eval); a - rich-form case may add `name` (kebab-case) and `files`. + rich-form case may add `name` (kebab-case) and `files`, and any case may add `difficulty` + (`hard` or `routine`), `why_hard` (the reason a person judged it hard) and `source` (where + the case came from). 4. Report each violation with its JSON path, or confirm the file conforms. 5. Run the deterministic eval-quality lint over every located file, all at once. The script accepts multiple paths: @@ -149,8 +151,9 @@ that line before editing, since it may be an illustrative example path rather th an unresolvable `files` fixture, an empty criterion item), then its `WARN:` lines grouped after (advisory quality heuristics: vague criterion phrasing, thin sole-criterion `expected_output`, identical prompt+files pairs, a set with no refusal/anti-pattern case, a criterion that leaves an - evaluative word such as "appropriate" or "correctly" undefined). - The script exits 0 when only warnings remain; run `--help` for the full Q1-Q10 check list. + evaluative word such as "appropriate" or "correctly" undefined, a `difficulty: hard` case + with no `why_hard`). + The script exits 0 when only warnings remain; run `--help` for the full Q1-Q11 check list. If `jq` is absent the script exits 2. Report that the quality lint was skipped for that reason; the schema verdict from steps 3-4 still stands. @@ -217,8 +220,12 @@ Model-graded `claude plugin eval` cases are emitted on demand. Contract: bash "${CLAUDE_PLUGIN_ROOT}/scripts/measure-invocation.sh" score ``` - `compare ` prints per-skill train and validation deltas. `emit-plugin-eval - ` writes `claude plugin eval` cases into an empty or new ``. + `validate` WARNs when a should-trigger probe shares 4 or more consecutive words with the target + listing (`--copy-span N` changes the span); reword such a probe the way a user would ask. + `compare ` prints per-skill train and validation deltas, each trigger-rate + delta with a 95% interval and a "within noise" line when the interval contains 0. + `emit-plugin-eval [--runs N] ` writes `claude plugin eval` cases into an + empty or new ``, 3 runs per case unless `--runs` says otherwise. Claim: each case is a directory holding `prompt.md` (frontmatter fields, body is the prompt) and `graders/`, and a `tool_used` grader with `tool: Skill` and an `input_match` on the skill name checks that the skill fired. Basis: , the @@ -226,8 +233,9 @@ Model-graded `claude plugin eval` cases are emitted on demand. Contract: case layout, the `prompt.md` fields, or the `tool_used` grader fields. 3. Report per skill, per split (`train` and `validation`): `trigger_rate` and `false_trigger_rate`, sample size, and the method name. A rewrite is compared with `compare` - against `probes/baselines/listing-overlap.json` (or a later model-graded snapshot). The action - is complete when both splits are named; a single blended rate is not the done-condition. + against `probes/baselines/listing-overlap.json` (or a later model-graded snapshot); a delta + reported "within noise" is not a gain. The action is complete when both splits are named; a + single blended rate is not the done-condition. 4. `listing-overlap` is a floor, not a model-graded auto-invocation rate. Say so when reporting. Do not treat a 1.0 positive rate on the floor as proof the description saturates live auto-invocation. @@ -257,9 +265,7 @@ tool. This gate does not automate that reachability check; author and review aga - `measure-invocation`'s default `listing-overlap` method is a lexical floor. A 1.0 positive trigger rate means the description already contains the request's nouns, not that live auto-invocation saturates. Report both splits and name the method. -- The shipped probes are this marketplace's skills, and two positives of - `probes/skill-quality.check.json` quote the `check` description nearly verbatim, so a 1.0 - positive rate there is close to true by construction. Outside the marketplace checkout `score` +- The shipped probes are this marketplace's skills. Outside the marketplace checkout `score` needs your own probes directory; no script turns `plugin-eval` or `claude -p` results into a report for `compare`. - A git repository is optional. Git-backed checks (trigger-keyword preservation, vendor @@ -361,7 +367,7 @@ tool. This gate does not automate that reachability check; author and review aga phrases, "read-only by default", the noun "remediation", and a negated "or rewrites" list do not advertise mutation. - `check-evals-quality.sh` requires `jq` (exit 2 without it, and the schema validation of - `validate-evals` steps 3-4 is unaffected). Its WARN-tier checks (Q5-Q10) are lexical heuristics: + `validate-evals` steps 3-4 is unaffected). Its WARN-tier checks (Q5-Q11) are lexical heuristics: Q9 (set-coverage) detects refusal/anti-pattern cases by wording, so a set whose guardrail case phrases the prohibition unusually can WARN despite covering it. Read the set before adding a case. It deliberately does not flag low case count: the marketplace's low eval volume is a recorded From 454bac41bf5fb3c888a8e12f19756bd7835dc8b9 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:54:15 -0400 Subject: [PATCH 02/11] feat(evals): report eval noise, route by repository kind, and add a hard pilot case plugin-eval gains scripts/noise-report.py: per-arm mean intervals, a paired interval on the with-versus-without difference, "within noise", "n too small to call", a near-ceiling warning, a pass-count view and judge-vote agreement. Its preflight warns when the tested model and the judge are the same model. design routes by repository kind, approves every candidate input through scripts/render-review.py (Markdown by default, escaped HTML as the option), keeps raw transcripts out of cases in public repositories, and checks graders before their scores are trusted. A new rule file states the transcript rule for eval files. methodology points at the bundled hillclimb and build-eval guides through reference/hillclimb.md and records this plugin's defaults and source conflicts in reference/local-decisions.md. Six settings carry the defaults. The pilot suite gains the hard case noise-before-gain and tags measurable-criterion as a regression guard. Co-Authored-By: Claude Opus 5.5 --- .claude/rules/cost-claims.md | 13 +- .claude/rules/eval-case-transcripts.md | 17 + AGENTS.md | 3 +- docs/native-surfaces.md | 10 +- docs/native-surfaces/records.json | 12 +- docs/upstream/claudedevs-cost-performance.md | 39 +- plugins/evals/.claude-plugin/plugin.json | 41 +- plugins/evals/CHANGELOG.md | 102 +++ plugins/evals/README.md | 25 +- .../evals/ci-pin-models/graders/pin-both.md | 8 + .../evals/ci-pin-models/graders/read-json.md | 8 + .../ci-pin-models/graders/skill-fired.md | 5 + .../ci-pin-models/graders/trust-plugin.md | 5 + plugins/evals/evals/ci-pin-models/prompt.md | 10 + .../evals/ci-pin-models/samples/pin-both.json | 38 + .../ci-pin-models/samples/read-json.json | 38 + .../ci-pin-models/samples/skill-fired.json | 72 ++ .../ci-pin-models/samples/trust-plugin.json | 26 + .../graders/names-geval.md | 5 + .../graders/no-evals-skill.md | 8 + .../evals/control-deepeval-geval/prompt.md | 10 + .../samples/names-geval.json | 26 + .../samples/no-evals-skill.json | 87 ++ .../graders/no-skill-fired.md | 1 + .../evals/evals/control-no-trigger/prompt.md | 4 +- .../samples/names-conftest.json | 22 + .../samples/no-skill-fired.json | 87 ++ .../graders/names-is-json.md | 5 + .../graders/no-evals-skill.md | 8 + .../evals/control-promptfoo-json/prompt.md | 10 + .../samples/names-is-json.json | 26 + .../samples/no-evals-skill.json | 87 ++ .../graders/expectations-array.md | 8 + .../graders/no-evals-skill.md | 8 + .../prompt.md | 16 + .../samples/expectations-array.json | 38 + .../samples/no-evals-skill.json | 87 ++ .../graders/judge-default.md | 8 + .../graders/skill-fired.md | 5 + .../graders/warns-not-refuses.md | 11 + .../evals/default-judge-same-model/prompt.md | 10 + .../samples/judge-default.json | 38 + .../samples/skill-fired.json | 72 ++ .../samples/warns-not-refuses.json | 38 + .../graders/criteria-first.md | 8 + .../graders/skill-fired.md | 5 + .../evals/design-criteria-first/prompt.md | 10 + .../samples/criteria-first.json | 38 + .../samples/skill-fired.json | 72 ++ .../graders/before-handoff.md | 11 + .../graders/hands-off.md | 8 + .../graders/skill-fired.md | 5 + .../evals/design-plugin-handoff/prompt.md | 10 + .../samples/before-handoff.json | 38 + .../samples/hands-off.json | 38 + .../samples/skill-fired.json | 72 ++ .../graders/contains-not-a-type.md | 11 + .../graders/names-validator.md | 8 + .../free-load-check/graders/skill-fired.md | 5 + .../free-load-check/graders/timeout-key.md | 5 + .../graders/token-validator.md | 5 + plugins/evals/evals/free-load-check/prompt.md | 10 + .../samples/contains-not-a-type.json | 38 + .../samples/names-validator.json | 38 + .../free-load-check/samples/skill-fired.json | 72 ++ .../free-load-check/samples/timeout-key.json | 26 + .../samples/token-validator.json | 18 + .../graders/grading-choice.md | 12 + .../graders/methodology-wording.md | 4 +- .../graders/skill-fired.md | 2 +- .../evals/grading-method-choice/prompt.md | 6 +- .../samples/grading-choice.json | 38 + .../samples/methodology-wording.json | 74 ++ .../samples/skill-fired.json | 72 ++ .../graders/expected-behaviour.md | 8 + .../graders/skill-fired.md | 5 + .../evals/interval-setting-scope/prompt.md | 10 + .../samples/expected-behaviour.json | 38 + .../samples/skill-fired.json | 72 ++ .../graders/no-number-from-invalid.md | 11 + .../graders/skill-fired.md | 5 + .../graders/token-keep-temp.md | 5 + .../graders/why-invalid.md | 8 + .../evals/invalid-gate-no-number/prompt.md | 10 + .../samples/no-number-from-invalid.json | 38 + .../samples/skill-fired.json | 72 ++ .../samples/token-keep-temp.json | 14 + .../samples/why-invalid.json | 38 + .../graders/confirm-at-three.md | 8 + .../graders/confirm-noise-report.md | 8 + .../graders/iteration-flags.md | 8 + .../graders/skill-fired.md | 5 + .../evals/iterate-then-confirm/prompt.md | 10 + .../samples/confirm-at-three.json | 38 + .../samples/confirm-noise-report.json | 38 + .../samples/iteration-flags.json | 38 + .../samples/skill-fired.json | 72 ++ .../graders/four-properties.md | 4 +- .../graders/skill-fired.md | 2 +- .../evals/measurable-criterion/prompt.md | 4 +- .../samples/four-properties.json | 38 + .../samples/skill-fired.json | 72 ++ .../graders/does-not-apply.md | 8 + .../evals/no-model-in-the-loop/prompt.md | 10 + .../samples/does-not-apply.json | 38 + .../noise-before-gain/graders/ceiling.md | 11 + .../graders/not-established.md | 12 + .../noise-before-gain/graders/skill-fired.md | 5 + .../evals/evals/noise-before-gain/prompt.md | 10 + .../noise-before-gain/samples/ceiling.json | 38 + .../samples/not-established.json | 38 + .../samples/skill-fired.json | 72 ++ .../graders/gate-before-numbers.md | 8 + .../graders/noise-verdict-rule.md | 11 + .../graders/run-flags.md | 8 + .../graders/skill-fired.md | 5 + .../graders/token-noise-report.md | 5 + .../graders/token-run-validity.md | 5 + .../noise-report-before-posting/prompt.md | 10 + .../samples/gate-before-numbers.json | 38 + .../samples/noise-verdict-rule.json | 38 + .../samples/run-flags.json | 38 + .../samples/skill-fired.json | 72 ++ .../samples/token-noise-report.json | 14 + .../samples/token-run-validity.json | 14 + .../graders/missing-delta-not-zero.md | 11 + .../graders/skill-fired.md | 5 + .../graders/skipped-judge-is-failure.md | 11 + .../omitted-delta-skipped-judge/prompt.md | 10 + .../samples/missing-delta-not-zero.json | 38 + .../samples/skill-fired.json | 72 ++ .../samples/skipped-judge-is-failure.json | 38 + .../price-before-running/graders/estimate.md | 13 + .../graders/skill-fired.md | 5 + .../graders/under-ceiling.md | 11 + .../evals/price-before-running/prompt.md | 10 + .../samples/estimate.json | 38 + .../samples/skill-fired.json | 72 ++ .../samples/under-ceiling.json | 38 + .../reference-read-denied/graders/cause.md | 11 + .../reference-read-denied/graders/fix.md | 8 + .../graders/skill-fired.md | 5 + .../evals/reference-read-denied/prompt.md | 10 + .../reference-read-denied/samples/cause.json | 38 + .../reference-read-denied/samples/fix.json | 38 + .../samples/skill-fired.json | 72 ++ .../graders/did-not-start-build-eval.md | 8 + .../graders/routes-to-build-eval.md | 8 + .../graders/skill-fired.md | 5 + .../graders/token-build-eval.md | 6 + .../evals/route-claude-api-app/prompt.md | 10 + .../samples/did-not-start-build-eval.json | 86 ++ .../samples/routes-to-build-eval.json | 38 + .../samples/skill-fired.json | 83 ++ .../samples/token-build-eval.json | 18 + .../graders/did-not-start-hillclimb.md | 8 + .../graders/routes-to-hillclimb.md | 8 + .../graders/token-hillclimb.md | 6 + .../route-hillclimb-existing-suite/prompt.md | 10 + .../samples/did-not-start-hillclimb.json | 86 ++ .../samples/routes-to-hillclimb.json | 38 + .../samples/token-hillclimb.json | 18 + .../route-mixed-repo/graders/plugin-part.md | 8 + .../route-mixed-repo/graders/service-part.md | 8 + .../route-mixed-repo/graders/skill-fired.md | 5 + .../evals/evals/route-mixed-repo/prompt.md | 10 + .../route-mixed-repo/samples/plugin-part.json | 38 + .../samples/service-part.json | 38 + .../route-mixed-repo/samples/skill-fired.json | 83 ++ .../graders/cannot-measure.md | 11 + .../graders/preferred-route.md | 10 + .../rules-not-a-target/graders/skill-fired.md | 5 + .../evals/evals/rules-not-a-target/prompt.md | 10 + .../samples/cannot-measure.json | 38 + .../samples/preferred-route.json | 38 + .../samples/skill-fired.json | 72 ++ .../graders/refuse-before-spend.md | 11 + .../graders/route-elsewhere.md | 8 + .../graders/skill-fired.md | 5 + .../evals/sandbox-refusal-windows/prompt.md | 10 + .../samples/refuse-before-spend.json | 38 + .../samples/route-elsewhere.json | 38 + .../samples/skill-fired.json | 72 ++ .../target-before-json/graders/skill-fired.md | 5 + .../graders/target-first.md | 8 + .../evals/evals/target-before-json/prompt.md | 10 + .../samples/skill-fired.json | 72 ++ .../samples/target-first.json | 38 + .../graders/early-access-means-update.md | 11 + .../version-floor/graders/skill-fired.md | 5 + .../graders/unavailable-is-server-side.md | 8 + .../version-floor/graders/version-number.md | 5 + plugins/evals/evals/version-floor/prompt.md | 10 + .../samples/early-access-means-update.json | 38 + .../version-floor/samples/skill-fired.json | 72 ++ .../samples/unavailable-is-server-side.json | 42 + .../version-floor/samples/version-number.json | 30 + .../graders/no-way-around.md | 8 + .../graders/paste-outside.md | 8 + .../graders/skill-fired.md | 5 + .../evals/evals/worktree-guard-stop/prompt.md | 10 + .../samples/no-way-around.json | 38 + .../samples/paste-outside.json | 38 + .../samples/skill-fired.json | 72 ++ .../graders/fired-indicator.md | 8 + .../wrap-bare-skill/graders/must-wrap.md | 8 + .../wrap-bare-skill/graders/skill-fired.md | 5 + plugins/evals/evals/wrap-bare-skill/prompt.md | 10 + .../samples/fired-indicator.json | 38 + .../wrap-bare-skill/samples/must-wrap.json | 38 + .../wrap-bare-skill/samples/skill-fired.json | 83 ++ .../graders/check-and-rerun.md | 8 + .../graders/not-marked-partial.md | 8 + .../graders/skill-fired.md | 5 + .../evals/zeros-after-usage-limit/prompt.md | 10 + .../samples/check-and-rerun.json | 42 + .../samples/not-marked-partial.json | 38 + .../samples/skill-fired.json | 72 ++ plugins/evals/skills/design/SKILL.md | 157 +++- plugins/evals/skills/design/evals/evals.json | 35 +- .../skills/design/scripts/render-review.py | 140 ++++ .../design/scripts/render-review.test.sh | 19 + .../design/scripts/test_render_review.py | 184 +++++ plugins/evals/skills/methodology/SKILL.md | 92 ++- .../methodology/reference/eval-design.md | 29 +- .../skills/methodology/reference/grading.md | 2 + .../skills/methodology/reference/hillclimb.md | 98 +++ .../methodology/reference/local-decisions.md | 95 +++ .../skills/methodology/reference/recipes.md | 10 +- .../methodology/reference/success-criteria.md | 2 + plugins/evals/skills/plugin-eval/SKILL.md | 248 +++++- .../evals/skills/plugin-eval/evals/evals.json | 28 +- .../plugin-eval/reference/case-authoring.md | 4 +- .../evals/skills/plugin-eval/reference/ci.md | 19 +- .../plugin-eval/reference/reading-results.md | 58 +- .../plugin-eval/scripts/calibrate-judge.py | 726 +++++++++++++++++ .../scripts/calibrate-judge.test.sh | 41 + .../fixtures/calibrate-judge/result.json | 91 +++ .../suite/capital-city/graders/names-paris.md | 6 + .../suite/capital-city/graders/no-tool.md | 6 + .../suite/capital-city/prompt.md | 8 + .../capital-city/samples/names-paris.json | 26 + .../suite/capital-city/samples/no-tool.json | 4 + .../traces/answered-itself.jsonl | 2 + .../calibrate-judge/traces/reproduced.jsonl | 3 + .../fixtures/run-validity/r2-clean-runs.json | 56 ++ .../fixtures/run-validity/r2-invalid.json | 80 ++ .../fixtures/run-validity/traces/LyFFkR.jsonl | 4 + .../fixtures/run-validity/traces/axyiOf.jsonl | 4 + .../fixtures/run-validity/traces/clean.jsonl | 2 + .../fixtures/run-validity/traces/d6Z5IW.jsonl | 4 + .../plugin-eval/scripts/noise-report.py | 469 +++++++++++ .../plugin-eval/scripts/noise-report.test.sh | 41 + .../plugin-eval/scripts/run-validity.py | 771 ++++++++++++++++++ .../plugin-eval/scripts/run-validity.test.sh | 41 + .../scripts/test_calibrate_judge.py | 537 ++++++++++++ .../plugin-eval/scripts/test_noise_report.py | 460 +++++++++++ .../plugin-eval/scripts/test_run_validity.py | 613 ++++++++++++++ plugins/evals/skills/validate/SKILL.md | 80 +- .../validate/scripts/test_validate_cases.py | 345 +++++++- .../skills/validate/scripts/validate-cases.py | 359 +++++++- scripts/affected-tests-no-suite.txt | 7 + 262 files changed, 11434 insertions(+), 167 deletions(-) create mode 100644 .claude/rules/eval-case-transcripts.md create mode 100644 plugins/evals/evals/ci-pin-models/graders/pin-both.md create mode 100644 plugins/evals/evals/ci-pin-models/graders/read-json.md create mode 100644 plugins/evals/evals/ci-pin-models/graders/skill-fired.md create mode 100644 plugins/evals/evals/ci-pin-models/graders/trust-plugin.md create mode 100644 plugins/evals/evals/ci-pin-models/prompt.md create mode 100644 plugins/evals/evals/ci-pin-models/samples/pin-both.json create mode 100644 plugins/evals/evals/ci-pin-models/samples/read-json.json create mode 100644 plugins/evals/evals/ci-pin-models/samples/skill-fired.json create mode 100644 plugins/evals/evals/ci-pin-models/samples/trust-plugin.json create mode 100644 plugins/evals/evals/control-deepeval-geval/graders/names-geval.md create mode 100644 plugins/evals/evals/control-deepeval-geval/graders/no-evals-skill.md create mode 100644 plugins/evals/evals/control-deepeval-geval/prompt.md create mode 100644 plugins/evals/evals/control-deepeval-geval/samples/names-geval.json create mode 100644 plugins/evals/evals/control-deepeval-geval/samples/no-evals-skill.json create mode 100644 plugins/evals/evals/control-no-trigger/samples/names-conftest.json create mode 100644 plugins/evals/evals/control-no-trigger/samples/no-skill-fired.json create mode 100644 plugins/evals/evals/control-promptfoo-json/graders/names-is-json.md create mode 100644 plugins/evals/evals/control-promptfoo-json/graders/no-evals-skill.md create mode 100644 plugins/evals/evals/control-promptfoo-json/prompt.md create mode 100644 plugins/evals/evals/control-promptfoo-json/samples/names-is-json.json create mode 100644 plugins/evals/evals/control-promptfoo-json/samples/no-evals-skill.json create mode 100644 plugins/evals/evals/control-skill-creator-evals-json/graders/expectations-array.md create mode 100644 plugins/evals/evals/control-skill-creator-evals-json/graders/no-evals-skill.md create mode 100644 plugins/evals/evals/control-skill-creator-evals-json/prompt.md create mode 100644 plugins/evals/evals/control-skill-creator-evals-json/samples/expectations-array.json create mode 100644 plugins/evals/evals/control-skill-creator-evals-json/samples/no-evals-skill.json create mode 100644 plugins/evals/evals/default-judge-same-model/graders/judge-default.md create mode 100644 plugins/evals/evals/default-judge-same-model/graders/skill-fired.md create mode 100644 plugins/evals/evals/default-judge-same-model/graders/warns-not-refuses.md create mode 100644 plugins/evals/evals/default-judge-same-model/prompt.md create mode 100644 plugins/evals/evals/default-judge-same-model/samples/judge-default.json create mode 100644 plugins/evals/evals/default-judge-same-model/samples/skill-fired.json create mode 100644 plugins/evals/evals/default-judge-same-model/samples/warns-not-refuses.json create mode 100644 plugins/evals/evals/design-criteria-first/graders/criteria-first.md create mode 100644 plugins/evals/evals/design-criteria-first/graders/skill-fired.md create mode 100644 plugins/evals/evals/design-criteria-first/prompt.md create mode 100644 plugins/evals/evals/design-criteria-first/samples/criteria-first.json create mode 100644 plugins/evals/evals/design-criteria-first/samples/skill-fired.json create mode 100644 plugins/evals/evals/design-plugin-handoff/graders/before-handoff.md create mode 100644 plugins/evals/evals/design-plugin-handoff/graders/hands-off.md create mode 100644 plugins/evals/evals/design-plugin-handoff/graders/skill-fired.md create mode 100644 plugins/evals/evals/design-plugin-handoff/prompt.md create mode 100644 plugins/evals/evals/design-plugin-handoff/samples/before-handoff.json create mode 100644 plugins/evals/evals/design-plugin-handoff/samples/hands-off.json create mode 100644 plugins/evals/evals/design-plugin-handoff/samples/skill-fired.json create mode 100644 plugins/evals/evals/free-load-check/graders/contains-not-a-type.md create mode 100644 plugins/evals/evals/free-load-check/graders/names-validator.md create mode 100644 plugins/evals/evals/free-load-check/graders/skill-fired.md create mode 100644 plugins/evals/evals/free-load-check/graders/timeout-key.md create mode 100644 plugins/evals/evals/free-load-check/graders/token-validator.md create mode 100644 plugins/evals/evals/free-load-check/prompt.md create mode 100644 plugins/evals/evals/free-load-check/samples/contains-not-a-type.json create mode 100644 plugins/evals/evals/free-load-check/samples/names-validator.json create mode 100644 plugins/evals/evals/free-load-check/samples/skill-fired.json create mode 100644 plugins/evals/evals/free-load-check/samples/timeout-key.json create mode 100644 plugins/evals/evals/free-load-check/samples/token-validator.json create mode 100644 plugins/evals/evals/grading-method-choice/graders/grading-choice.md create mode 100644 plugins/evals/evals/grading-method-choice/samples/grading-choice.json create mode 100644 plugins/evals/evals/grading-method-choice/samples/methodology-wording.json create mode 100644 plugins/evals/evals/grading-method-choice/samples/skill-fired.json create mode 100644 plugins/evals/evals/interval-setting-scope/graders/expected-behaviour.md create mode 100644 plugins/evals/evals/interval-setting-scope/graders/skill-fired.md create mode 100644 plugins/evals/evals/interval-setting-scope/prompt.md create mode 100644 plugins/evals/evals/interval-setting-scope/samples/expected-behaviour.json create mode 100644 plugins/evals/evals/interval-setting-scope/samples/skill-fired.json create mode 100644 plugins/evals/evals/invalid-gate-no-number/graders/no-number-from-invalid.md create mode 100644 plugins/evals/evals/invalid-gate-no-number/graders/skill-fired.md create mode 100644 plugins/evals/evals/invalid-gate-no-number/graders/token-keep-temp.md create mode 100644 plugins/evals/evals/invalid-gate-no-number/graders/why-invalid.md create mode 100644 plugins/evals/evals/invalid-gate-no-number/prompt.md create mode 100644 plugins/evals/evals/invalid-gate-no-number/samples/no-number-from-invalid.json create mode 100644 plugins/evals/evals/invalid-gate-no-number/samples/skill-fired.json create mode 100644 plugins/evals/evals/invalid-gate-no-number/samples/token-keep-temp.json create mode 100644 plugins/evals/evals/invalid-gate-no-number/samples/why-invalid.json create mode 100644 plugins/evals/evals/iterate-then-confirm/graders/confirm-at-three.md create mode 100644 plugins/evals/evals/iterate-then-confirm/graders/confirm-noise-report.md create mode 100644 plugins/evals/evals/iterate-then-confirm/graders/iteration-flags.md create mode 100644 plugins/evals/evals/iterate-then-confirm/graders/skill-fired.md create mode 100644 plugins/evals/evals/iterate-then-confirm/prompt.md create mode 100644 plugins/evals/evals/iterate-then-confirm/samples/confirm-at-three.json create mode 100644 plugins/evals/evals/iterate-then-confirm/samples/confirm-noise-report.json create mode 100644 plugins/evals/evals/iterate-then-confirm/samples/iteration-flags.json create mode 100644 plugins/evals/evals/iterate-then-confirm/samples/skill-fired.json create mode 100644 plugins/evals/evals/measurable-criterion/samples/four-properties.json create mode 100644 plugins/evals/evals/measurable-criterion/samples/skill-fired.json create mode 100644 plugins/evals/evals/no-model-in-the-loop/graders/does-not-apply.md create mode 100644 plugins/evals/evals/no-model-in-the-loop/prompt.md create mode 100644 plugins/evals/evals/no-model-in-the-loop/samples/does-not-apply.json create mode 100644 plugins/evals/evals/noise-before-gain/graders/ceiling.md create mode 100644 plugins/evals/evals/noise-before-gain/graders/not-established.md create mode 100644 plugins/evals/evals/noise-before-gain/graders/skill-fired.md create mode 100644 plugins/evals/evals/noise-before-gain/prompt.md create mode 100644 plugins/evals/evals/noise-before-gain/samples/ceiling.json create mode 100644 plugins/evals/evals/noise-before-gain/samples/not-established.json create mode 100644 plugins/evals/evals/noise-before-gain/samples/skill-fired.json create mode 100644 plugins/evals/evals/noise-report-before-posting/graders/gate-before-numbers.md create mode 100644 plugins/evals/evals/noise-report-before-posting/graders/noise-verdict-rule.md create mode 100644 plugins/evals/evals/noise-report-before-posting/graders/run-flags.md create mode 100644 plugins/evals/evals/noise-report-before-posting/graders/skill-fired.md create mode 100644 plugins/evals/evals/noise-report-before-posting/graders/token-noise-report.md create mode 100644 plugins/evals/evals/noise-report-before-posting/graders/token-run-validity.md create mode 100644 plugins/evals/evals/noise-report-before-posting/prompt.md create mode 100644 plugins/evals/evals/noise-report-before-posting/samples/gate-before-numbers.json create mode 100644 plugins/evals/evals/noise-report-before-posting/samples/noise-verdict-rule.json create mode 100644 plugins/evals/evals/noise-report-before-posting/samples/run-flags.json create mode 100644 plugins/evals/evals/noise-report-before-posting/samples/skill-fired.json create mode 100644 plugins/evals/evals/noise-report-before-posting/samples/token-noise-report.json create mode 100644 plugins/evals/evals/noise-report-before-posting/samples/token-run-validity.json create mode 100644 plugins/evals/evals/omitted-delta-skipped-judge/graders/missing-delta-not-zero.md create mode 100644 plugins/evals/evals/omitted-delta-skipped-judge/graders/skill-fired.md create mode 100644 plugins/evals/evals/omitted-delta-skipped-judge/graders/skipped-judge-is-failure.md create mode 100644 plugins/evals/evals/omitted-delta-skipped-judge/prompt.md create mode 100644 plugins/evals/evals/omitted-delta-skipped-judge/samples/missing-delta-not-zero.json create mode 100644 plugins/evals/evals/omitted-delta-skipped-judge/samples/skill-fired.json create mode 100644 plugins/evals/evals/omitted-delta-skipped-judge/samples/skipped-judge-is-failure.json create mode 100644 plugins/evals/evals/price-before-running/graders/estimate.md create mode 100644 plugins/evals/evals/price-before-running/graders/skill-fired.md create mode 100644 plugins/evals/evals/price-before-running/graders/under-ceiling.md create mode 100644 plugins/evals/evals/price-before-running/prompt.md create mode 100644 plugins/evals/evals/price-before-running/samples/estimate.json create mode 100644 plugins/evals/evals/price-before-running/samples/skill-fired.json create mode 100644 plugins/evals/evals/price-before-running/samples/under-ceiling.json create mode 100644 plugins/evals/evals/reference-read-denied/graders/cause.md create mode 100644 plugins/evals/evals/reference-read-denied/graders/fix.md create mode 100644 plugins/evals/evals/reference-read-denied/graders/skill-fired.md create mode 100644 plugins/evals/evals/reference-read-denied/prompt.md create mode 100644 plugins/evals/evals/reference-read-denied/samples/cause.json create mode 100644 plugins/evals/evals/reference-read-denied/samples/fix.json create mode 100644 plugins/evals/evals/reference-read-denied/samples/skill-fired.json create mode 100644 plugins/evals/evals/route-claude-api-app/graders/did-not-start-build-eval.md create mode 100644 plugins/evals/evals/route-claude-api-app/graders/routes-to-build-eval.md create mode 100644 plugins/evals/evals/route-claude-api-app/graders/skill-fired.md create mode 100644 plugins/evals/evals/route-claude-api-app/graders/token-build-eval.md create mode 100644 plugins/evals/evals/route-claude-api-app/prompt.md create mode 100644 plugins/evals/evals/route-claude-api-app/samples/did-not-start-build-eval.json create mode 100644 plugins/evals/evals/route-claude-api-app/samples/routes-to-build-eval.json create mode 100644 plugins/evals/evals/route-claude-api-app/samples/skill-fired.json create mode 100644 plugins/evals/evals/route-claude-api-app/samples/token-build-eval.json create mode 100644 plugins/evals/evals/route-hillclimb-existing-suite/graders/did-not-start-hillclimb.md create mode 100644 plugins/evals/evals/route-hillclimb-existing-suite/graders/routes-to-hillclimb.md create mode 100644 plugins/evals/evals/route-hillclimb-existing-suite/graders/token-hillclimb.md create mode 100644 plugins/evals/evals/route-hillclimb-existing-suite/prompt.md create mode 100644 plugins/evals/evals/route-hillclimb-existing-suite/samples/did-not-start-hillclimb.json create mode 100644 plugins/evals/evals/route-hillclimb-existing-suite/samples/routes-to-hillclimb.json create mode 100644 plugins/evals/evals/route-hillclimb-existing-suite/samples/token-hillclimb.json create mode 100644 plugins/evals/evals/route-mixed-repo/graders/plugin-part.md create mode 100644 plugins/evals/evals/route-mixed-repo/graders/service-part.md create mode 100644 plugins/evals/evals/route-mixed-repo/graders/skill-fired.md create mode 100644 plugins/evals/evals/route-mixed-repo/prompt.md create mode 100644 plugins/evals/evals/route-mixed-repo/samples/plugin-part.json create mode 100644 plugins/evals/evals/route-mixed-repo/samples/service-part.json create mode 100644 plugins/evals/evals/route-mixed-repo/samples/skill-fired.json create mode 100644 plugins/evals/evals/rules-not-a-target/graders/cannot-measure.md create mode 100644 plugins/evals/evals/rules-not-a-target/graders/preferred-route.md create mode 100644 plugins/evals/evals/rules-not-a-target/graders/skill-fired.md create mode 100644 plugins/evals/evals/rules-not-a-target/prompt.md create mode 100644 plugins/evals/evals/rules-not-a-target/samples/cannot-measure.json create mode 100644 plugins/evals/evals/rules-not-a-target/samples/preferred-route.json create mode 100644 plugins/evals/evals/rules-not-a-target/samples/skill-fired.json create mode 100644 plugins/evals/evals/sandbox-refusal-windows/graders/refuse-before-spend.md create mode 100644 plugins/evals/evals/sandbox-refusal-windows/graders/route-elsewhere.md create mode 100644 plugins/evals/evals/sandbox-refusal-windows/graders/skill-fired.md create mode 100644 plugins/evals/evals/sandbox-refusal-windows/prompt.md create mode 100644 plugins/evals/evals/sandbox-refusal-windows/samples/refuse-before-spend.json create mode 100644 plugins/evals/evals/sandbox-refusal-windows/samples/route-elsewhere.json create mode 100644 plugins/evals/evals/sandbox-refusal-windows/samples/skill-fired.json create mode 100644 plugins/evals/evals/target-before-json/graders/skill-fired.md create mode 100644 plugins/evals/evals/target-before-json/graders/target-first.md create mode 100644 plugins/evals/evals/target-before-json/prompt.md create mode 100644 plugins/evals/evals/target-before-json/samples/skill-fired.json create mode 100644 plugins/evals/evals/target-before-json/samples/target-first.json create mode 100644 plugins/evals/evals/version-floor/graders/early-access-means-update.md create mode 100644 plugins/evals/evals/version-floor/graders/skill-fired.md create mode 100644 plugins/evals/evals/version-floor/graders/unavailable-is-server-side.md create mode 100644 plugins/evals/evals/version-floor/graders/version-number.md create mode 100644 plugins/evals/evals/version-floor/prompt.md create mode 100644 plugins/evals/evals/version-floor/samples/early-access-means-update.json create mode 100644 plugins/evals/evals/version-floor/samples/skill-fired.json create mode 100644 plugins/evals/evals/version-floor/samples/unavailable-is-server-side.json create mode 100644 plugins/evals/evals/version-floor/samples/version-number.json create mode 100644 plugins/evals/evals/worktree-guard-stop/graders/no-way-around.md create mode 100644 plugins/evals/evals/worktree-guard-stop/graders/paste-outside.md create mode 100644 plugins/evals/evals/worktree-guard-stop/graders/skill-fired.md create mode 100644 plugins/evals/evals/worktree-guard-stop/prompt.md create mode 100644 plugins/evals/evals/worktree-guard-stop/samples/no-way-around.json create mode 100644 plugins/evals/evals/worktree-guard-stop/samples/paste-outside.json create mode 100644 plugins/evals/evals/worktree-guard-stop/samples/skill-fired.json create mode 100644 plugins/evals/evals/wrap-bare-skill/graders/fired-indicator.md create mode 100644 plugins/evals/evals/wrap-bare-skill/graders/must-wrap.md create mode 100644 plugins/evals/evals/wrap-bare-skill/graders/skill-fired.md create mode 100644 plugins/evals/evals/wrap-bare-skill/prompt.md create mode 100644 plugins/evals/evals/wrap-bare-skill/samples/fired-indicator.json create mode 100644 plugins/evals/evals/wrap-bare-skill/samples/must-wrap.json create mode 100644 plugins/evals/evals/wrap-bare-skill/samples/skill-fired.json create mode 100644 plugins/evals/evals/zeros-after-usage-limit/graders/check-and-rerun.md create mode 100644 plugins/evals/evals/zeros-after-usage-limit/graders/not-marked-partial.md create mode 100644 plugins/evals/evals/zeros-after-usage-limit/graders/skill-fired.md create mode 100644 plugins/evals/evals/zeros-after-usage-limit/prompt.md create mode 100644 plugins/evals/evals/zeros-after-usage-limit/samples/check-and-rerun.json create mode 100644 plugins/evals/evals/zeros-after-usage-limit/samples/not-marked-partial.json create mode 100644 plugins/evals/evals/zeros-after-usage-limit/samples/skill-fired.json create mode 100755 plugins/evals/skills/design/scripts/render-review.py create mode 100755 plugins/evals/skills/design/scripts/render-review.test.sh create mode 100755 plugins/evals/skills/design/scripts/test_render_review.py create mode 100644 plugins/evals/skills/methodology/reference/hillclimb.md create mode 100644 plugins/evals/skills/methodology/reference/local-decisions.md create mode 100755 plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py create mode 100755 plugins/evals/skills/plugin-eval/scripts/calibrate-judge.test.sh create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/result.json create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/graders/names-paris.md create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/graders/no-tool.md create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/prompt.md create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/samples/names-paris.json create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/samples/no-tool.json create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/traces/answered-itself.jsonl create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/traces/reproduced.jsonl create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/r2-clean-runs.json create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/r2-invalid.json create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/LyFFkR.jsonl create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/axyiOf.jsonl create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/clean.jsonl create mode 100644 plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/d6Z5IW.jsonl create mode 100755 plugins/evals/skills/plugin-eval/scripts/noise-report.py create mode 100755 plugins/evals/skills/plugin-eval/scripts/noise-report.test.sh create mode 100755 plugins/evals/skills/plugin-eval/scripts/run-validity.py create mode 100755 plugins/evals/skills/plugin-eval/scripts/run-validity.test.sh create mode 100755 plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py create mode 100755 plugins/evals/skills/plugin-eval/scripts/test_noise_report.py create mode 100755 plugins/evals/skills/plugin-eval/scripts/test_run_validity.py diff --git a/.claude/rules/cost-claims.md b/.claude/rules/cost-claims.md index f188403342..943f067ec3 100644 --- a/.claude/rules/cost-claims.md +++ b/.claude/rules/cost-claims.md @@ -1,5 +1,5 @@ --- -description: "Cost claims link the costs and pricing docs and state no prices or per-task figures; `docs/upstream/` records may list vendor figures labeled vendor-reported" +description: "Cost claims link the costs and pricing docs and state no prices or per-task figures; `docs/upstream/` records may list vendor figures labeled vendor-reported, and a skill that prices its own runs may state its dated, measured run costs" paths: - "plugins/*/skills/**" - "plugins/*/agents/**" @@ -24,9 +24,14 @@ conversation, is fine; the number is the docs' to state. On a subscription the `/usage` session cost is a list-price estimate of the work, not a bill, as the costs page says. Do not present it as spend. -One exception: a record under `docs/upstream/` may list a post's figures, each labeled -vendor-reported, because recording what the post said is its job. Guidance elsewhere links the -record and copies none of its figures. +Two exceptions: + +- A record under `docs/upstream/` may list a post's figures, each labeled vendor-reported, because + recording what the post said is its job. Guidance elsewhere links the record and copies none of + its figures. +- A skill that prices its own runs before spending may state the run costs it measured on its own + suite, each with the measurement date, the Claude Code version, the setup measured, and a + recheck trigger, labeled a list-price estimate. It states no rate. Basis: the two pages above, checked 2026-10-01. Recheck when either page moves the section linked here. diff --git a/.claude/rules/eval-case-transcripts.md b/.claude/rules/eval-case-transcripts.md new file mode 100644 index 0000000000..ca7067dff0 --- /dev/null +++ b/.claude/rules/eval-case-transcripts.md @@ -0,0 +1,17 @@ +--- +description: "Eval cases in this public repository never hold a raw session or product transcript; a one-to-one rewrite with every identifying detail changed is allowed after the identifying-details review" +paths: + - "plugins/*/evals/**" + - "plugins/*/skills/*/evals/**" +--- + +# Eval case transcripts + +This repository is public. No eval case here holds a raw Claude Code session transcript or a raw +product transcript (a chat log, support ticket, or log line from a real user), in a prompt, an +expected output, a grader, or a fixture. + +A one-to-one rewrite of a transcript is allowed: one case per original, with every sensitive or +identifying detail changed. Before it lands, a person approves it through the input approval in +`/evals:design`, which carries the identifying-details checklist. That approval is the review; there +is no second pass. diff --git a/AGENTS.md b/AGENTS.md index f9e9dd9654..10da9d8966 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -44,7 +44,8 @@ and its content is not already in context, read the file directly. | Surface | Covers | Topic | |---|---|---| -| `.claude/rules/cost-claims.md` | `plugins/*/skills/**, plugins/*/agents/**, plugins/*/reference/**, docs/**/*.md, prompts/**` | Cost claims link the costs and pricing docs and state no prices or per-task figures; `docs/upstream/` records may list vendor figures labeled vendor-reported | +| `.claude/rules/cost-claims.md` | `plugins/*/skills/**, plugins/*/agents/**, plugins/*/reference/**, docs/**/*.md, prompts/**` | Cost claims link the costs and pricing docs and state no prices or per-task figures; `docs/upstream/` records may list vendor figures labeled vendor-reported, and a skill that prices its own runs may state its dated, measured run costs | +| `.claude/rules/eval-case-transcripts.md` | `plugins/*/evals/**, plugins/*/skills/*/evals/**` | Eval cases in this public repository never hold a raw session or product transcript; a one-to-one rewrite with every identifying detail changed is allowed after the identifying-details review | | `.claude/rules/mod-authoring.md` | `plugins/*/hooks/**, plugins/*/types/**` | Mods stay deferred under ADR 0035: no plugin gains a `modules` key until its five go criteria pass; when they do, load the built-in `plugin-authoring` skill and the upstream mods docs first | | `.claude/rules/ruff-pin.md` | `**/*.py` | Python linting runs through the pinned ruff wrapper, never a bare ruff on PATH | | `.claude/rules/skill-bodies-state-current-rules.md` | `plugins/*/skills/**, plugins/*/agents/**` | Skill and agent bodies point at the live upstream source for any volatile specific instead of restating it, recorded as pointer, as-of date and recheck trigger, and name their successor in a `## Next` section; read before editing any skill body | diff --git a/docs/native-surfaces.md b/docs/native-surfaces.md index 719f2dd7f8..bbe587a129 100644 --- a/docs/native-surfaces.md +++ b/docs/native-surfaces.md @@ -526,17 +526,17 @@ and when. See [`docs/conventions/native-references/`](conventions/native-referen ### `claude-api` → `evals:methodology` -- **Verdict:** `complementary`: Different jobs on the same object. The bundled skill's hillclimb subcommand consumes an eval suite and searches model and effort for the cheapest configuration that holds the target (train/test split, one change per round, held-out scoring), and build-eval scaffolds the suite it needs; both run evals and change configuration. evals:methodology is knowledge about designing the suite (criteria, anatomy, grading, effort as an axis) and runs nothing. The two chain: design the suite here, hand it to the search. Recorded when the effort-axis note citing hillclimb landed in the methodology reference. +- **Verdict:** `complementary`: Different jobs on the same object. The bundled skill's hillclimb and build-eval subcommands act on an eval suite, as their published guides describe; evals:methodology is knowledge about designing the suite (criteria, anatomy, grading, effort as an axis) and runs nothing. The two chain: design the suite here, then the user types the subcommand. Routing by repository kind and the hillclimb step map live in the methodology skill and its reference/hillclimb.md. - **Integration:** `route` - **Native surface:** `claude-api` (bundled skill; markers: gated) - **Our component:** `evals:methodology` (skill) - **Evidence:** - - binary extraction 2026-09-09 (claude.exe 2.1.263): subcommand array includes build-eval and hillclimb; bundled shared/evals/eval-hillclimb.md read end to end (train/test split, one proposal per round, held-out scoring) - - hillclimb and build-eval absent from anthropics/skills HEAD 41bbe19 (2026-09-03) and from the platform claude-api-skill docs page + - binary extraction 2026-09-09 (claude.exe 2.1.263): subcommand array includes build-eval and hillclimb + - subcommand pointer: https://code.claude.com/docs/en/skills#work-on-claude-api-projects; published guides: https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md and build-eval.md at the same commit (as of 2026-10-01) - our description: 'Knowledge (WHY/WHAT of eval design), not a runner; ... for running and scoring a plugin's suite against a no-plugin baseline use /evals:plugin-eval' - - reference/eval-design.md 'Effort as an eval axis' cites the subcommand behind the presence gate + - reference/hillclimb.md names each hillclimb step by a link to its section and states only this repository's facts; reference/eval-design.md 'Effort as an eval axis' points to it behind the presence gate - **Observation:** extraction: extracted from binary 2.1.263 at node_modules/@anthropic-ai/claude-code/bin/claude.exe (subcommand array; bundled shared/evals/eval-hillclimb.md extracted and read); bulk registrar enumeration was broken at this build, so this row's evidence is the targeted extraction, not the inventory JSON (2026-09-09) -- **Recheck trigger:** a Claude Code release changes the bundled claude-api skill's subcommand set, or the public anthropics/skills repo or the docs page gains hillclimb/build-eval (verified 2026-09-11) +- **Recheck trigger:** a Claude Code release changes the bundled claude-api skill's subcommand set, the platform claude-api-skill docs page lists build-eval and hillclimb, or a commit to anthropics/skills changes skills/claude-api/shared/evals/ (verified 2026-10-01) - **Baked:** description phrase yes · Boundary section yes · Native step no · suggest sentence no - **Budget caveat:** the baked phrase may be dropped from the skill listing under budget pressure. It is the best available routing surface, not a guaranteed one diff --git a/docs/native-surfaces/records.json b/docs/native-surfaces/records.json index 6742758d39..57143faf66 100644 --- a/docs/native-surfaces/records.json +++ b/docs/native-surfaces/records.json @@ -690,12 +690,12 @@ }, "verdict": "complementary", "integration": "route", - "reason": "Different jobs on the same object. The bundled skill's hillclimb subcommand consumes an eval suite and searches model and effort for the cheapest configuration that holds the target (train/test split, one change per round, held-out scoring), and build-eval scaffolds the suite it needs; both run evals and change configuration. evals:methodology is knowledge about designing the suite (criteria, anatomy, grading, effort as an axis) and runs nothing. The two chain: design the suite here, hand it to the search. Recorded when the effort-axis note citing hillclimb landed in the methodology reference.", + "reason": "Different jobs on the same object. The bundled skill's hillclimb and build-eval subcommands act on an eval suite, as their published guides describe; evals:methodology is knowledge about designing the suite (criteria, anatomy, grading, effort as an axis) and runs nothing. The two chain: design the suite here, then the user types the subcommand. Routing by repository kind and the hillclimb step map live in the methodology skill and its reference/hillclimb.md.", "evidence": [ - "binary extraction 2026-09-09 (claude.exe 2.1.263): subcommand array includes build-eval and hillclimb; bundled shared/evals/eval-hillclimb.md read end to end (train/test split, one proposal per round, held-out scoring)", - "hillclimb and build-eval absent from anthropics/skills HEAD 41bbe19 (2026-09-03) and from the platform claude-api-skill docs page", + "binary extraction 2026-09-09 (claude.exe 2.1.263): subcommand array includes build-eval and hillclimb", + "subcommand pointer: https://code.claude.com/docs/en/skills#work-on-claude-api-projects; published guides: https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md and build-eval.md at the same commit (as of 2026-10-01)", "our description: 'Knowledge (WHY/WHAT of eval design), not a runner; ... for running and scoring a plugin's suite against a no-plugin baseline use /evals:plugin-eval'", - "reference/eval-design.md 'Effort as an eval axis' cites the subcommand behind the presence gate" + "reference/hillclimb.md names each hillclimb step by a link to its section and states only this repository's facts; reference/eval-design.md 'Effort as an eval axis' points to it behind the presence gate" ], "observation": { "class": "extraction", @@ -703,8 +703,8 @@ "date": "2026-09-09" }, "recheck": { - "trigger": "a Claude Code release changes the bundled claude-api skill's subcommand set, or the public anthropics/skills repo or the docs page gains hillclimb/build-eval", - "verified": "2026-09-11" + "trigger": "a Claude Code release changes the bundled claude-api skill's subcommand set, the platform claude-api-skill docs page lists build-eval and hillclimb, or a commit to anthropics/skills changes skills/claude-api/shared/evals/", + "verified": "2026-10-01" }, "baked": { "description_phrase": true, diff --git a/docs/upstream/claudedevs-cost-performance.md b/docs/upstream/claudedevs-cost-performance.md index 1b94603ae2..34e7e10db8 100644 --- a/docs/upstream/claudedevs-cost-performance.md +++ b/docs/upstream/claudedevs-cost-performance.md @@ -29,8 +29,8 @@ work list). The three native-overlap verdicts this effort produced (bundled `cla against `harness-config:audit-instructions`, `evals:methodology`, and `playbooks:fable-5`) are baked as `## Boundary` sections in those skill bodies with detail in a same-skill reference file, per the amended native-references convention (1.1.0: a non-`defer` extraction-evidence -row lands together with its Boundary section). Open TRACK triggers: the anthropics/skills repo -or the claude-api docs page gaining hillclimb/build-eval; a Console-side check confirming the +row lands together with its Boundary section). Open TRACK triggers: the platform claude-api +skill docs page listing build-eval and hillclimb; a Console-side check confirming the cache-diagnostics UI; a second real need for API-cost tooling in this marketplace. ## Source and verification @@ -51,13 +51,19 @@ cache-diagnostics UI; a second real need for API-cost tooling in this marketplac anthropics/skills clone (HEAD `41bbe19`, 2026-09-03) plus the skill bundled inside Claude Code 2.1.263. - Five verification findings qualify adoption everywhere below: - 1. **hillclimb repo lag.** Our extraction found `/claude-api hillclimb` (and `build-eval`) in - the bundled skill inside the Claude Code binary, and an exhaustive grep found them absent - from the public anthropics/skills repo the article links (HEAD 2026-09-03) and from the - skill's platform-docs page. A reader following the article's GitHub link will not find - them. Pointer: our binary extraction and clone grep, recorded in the claude-api row of - [`docs/native-surfaces/records.json`](../native-surfaces/records.json), and [In Claude Code (bundled)](https://platform.claude.com/docs/en/agents-and-tools/agent-skills/claude-api-skill#in-claude-code-bundled). - As of: 2026-09-09. Recheck trigger: the repo or docs page gains the subcommands. + 1. **Where hillclimb and build-eval are published.** Records in this repository point at the + published sources. + - **Pointer**: for the subcommands, see + [Work on Claude API projects](https://code.claude.com/docs/en/skills#work-on-claude-api-projects); + for their guides, see + [`eval-hillclimb.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md) + and + [`build-eval.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md) + at the pinned commit. + - **As of**: 2026-10-01 + - **Recheck trigger**: the + [Claude API skill](https://platform.claude.com/docs/en/agents-and-tools/agent-skills/claude-api-skill) + docs page lists build-eval and hillclimb; then repoint there. 2. **Claude Console diagnostics UI unverified.** The API half of the cache-diagnostics topic is verified against [Cache miss reason types](https://platform.claude.com/docs/en/build-with-claude/cache-diagnostics#cache-miss-reason-types); @@ -90,9 +96,8 @@ cache-diagnostics UI; a second real need for API-cost tooling in this marketplac single evidence pool because no independent second pool exists publicly: automatic-caching breakpoint movement (docs plus a restatement page; substance re-confirmed live), cost-optimize behavior (skill source only; the skill's docs page does - not document the command), hillclimb-bundled and hillclimb-absent (binary extraction and - an exhaustive clone grep, both direct observations). Rows built on these carry the - qualification rather than a second citation. + not document the command). Rows built on these carry the qualification rather than a second + citation. The hillclimb distribution rows use the published-source pointers in finding 1. ## Row schema @@ -160,8 +165,8 @@ routing restriction. The verdict is baked where the model reads it: a `## Bounda bundled claude-api skill` section in the `audit-instructions` body (routing, mutation gate, availability rule) with its records in `plugins/harness-config/skills/audit-instructions/reference/bundled-claude-api.md`. Recheck -fires with the store row's trigger (subcommand set changes, or the public repo / docs page -gains hillclimb). +fires with the store row's trigger (subcommand set changes, or the platform claude-api skill +docs page lists hillclimb and build-eval). | Topic | Ours | Verdict | Pointer | As of | |---|---|---|---|---| @@ -182,7 +187,7 @@ tooling. |---|---|---|---|---| | Effort miscalibration in both directions | PLUGIN-PHILOSOPHY Effort tiers; opus-5 chapter overthinking guidance; fable-5-1 low-effort recall caveat | COVERED, plus a sharpening ADOPT (decided 2026-09-10). Explore evidence re-verified 2026-09-09. Work item: fold the article's two sharpest phrasings on miscalibration, in our words, into the existing surfaces | [How effort works](https://platform.claude.com/docs/en/build-with-claude/effort#how-effort-works) | 2026-09-09 | | A stronger model at lower effort | Nowhere; adaptation chapters deliberately carry no pricing | ADOPT (decided 2026-09-10). Land as a pricing-free section in the fable-5-1 model-adaptation chapter plus a one-line pointer in PLUGIN-PHILOSOPHY Effort tiers; numbers cited vendor-reported; pricing stays pointer-resolved through the claude-api skill | [Compare models on cost per task](https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task) and [Model pricing](https://platform.claude.com/docs/en/about-claude/pricing#model-pricing) | 2026-09-09 | -| Effort sweeps on a non-saturated eval | `evals` plugin has zero effort content | ADOPT (decided 2026-09-10). Land as an effort-axis note in the evals plugin citing the bundled hillclimb per the Lane M posture (bundled-only, public-repo lag noted) | [Tune effort](https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort) | 2026-09-09 | +| Effort sweeps on a non-saturated eval | `evals` plugin has zero effort content | ADOPT (decided 2026-09-10). Land as an effort-axis note in the evals plugin citing the bundled hillclimb per the Lane M posture, pointing at its published guide (finding 1) | [Tune effort](https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort) | 2026-09-09 | | Effort changes mid-conversation and the cache | PLUGIN-PHILOSOPHY cache caveat + criteria I17-b carry the session-side version | COVERED session-side (decided 2026-09-10). The API-side model list is read at the pointer only inside whatever T1/T3 adoptions get written, per the Lane M beta posture; no separate surface | [Per-message effort (beta)](https://platform.claude.com/docs/en/build-with-claude/effort#change-effort-mid-conversation-beta) | 2026-09-09 | ## Lane T4: API cost optimization and profiling @@ -195,7 +200,7 @@ cost). Batch API, output bounding as a cost lever, and the usage/cost Admin API | Topic | Ours | Verdict | Pointer | As of | |---|---|---|---|---| | Spend profiling with the bundled cost-optimize | No incumbent for API-application profiling | TRACK on the bundled cost-optimize, plus one mention in the new playbooks chapter as the automation for its levers (decided 2026-09-10). Our source-as-spec read found it proposes rather than silently applies. New-plugin question deferred to a second real need. Recheck: a docs page starts covering the command | Our read of the bundled skill source (`shared/cost-optimization.md`); no docs page covers the command | 2026-09-09 | -| Model and effort search with the bundled hillclimb | No incumbent; `evals` owns eval design without a cost axis | Cited per the Lane M posture: bundled-only, public-repo lag noted (decided 2026-09-10); the evals effort-axis note carries the citation. Recheck: the repo or docs page gains the subcommand | Our extraction of the bundled skill source from the binary (finding 1); [In Claude Code (bundled)](https://platform.claude.com/docs/en/agents-and-tools/agent-skills/claude-api-skill#in-claude-code-bundled) | 2026-09-09 | +| Model and effort search with the bundled hillclimb | No incumbent; `evals` owns eval design without a cost axis | Cited per the Lane M posture (decided 2026-09-10); the evals methodology skill's `reference/hillclimb.md` carries the citation. Recheck: the platform claude-api skill docs page lists the subcommand | [Work on Claude API projects](https://code.claude.com/docs/en/skills#work-on-claude-api-projects) and [`eval-hillclimb.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md) at the pinned commit (finding 1) | 2026-10-01 | | Batching unattended work | Absent (sole mention is a routines.md disclaimer) | ADOPT (chapter row; decided 2026-09-10) | [Batch processing pricing](https://platform.claude.com/docs/en/about-claude/pricing#batch-processing) | 2026-09-09 | | Output bounding as a cost lever | In tension with prompt-audit Group 1f, which removes numeric output ceilings from skill bodies | Recorded scope-disjoint (decided 2026-09-10): output bounding is an API-request cost lever, never a skill-body instruction pattern; one sentence in the chapter says so. Tension identified by explore, 2026-09-09 | [Set budgets and output caps](https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps) | 2026-09-09 | | Org spend profiling through the Admin API | Absent | ADOPT (chapter row; decided 2026-09-10) | [Usage and Cost API: Cost API](https://platform.claude.com/docs/en/manage-claude/usage-cost-api#cost-api) | 2026-09-09 | @@ -209,7 +214,7 @@ Decided at interview, 2026-09-10: | Record shape | DECIDED: keep this file's shape. Only the row-schema FORMAT is borrowed from aihero-course.md (per-row records, verdict vocabulary); this source is unrelated to AI Hero and this record stands alone | Owner interview, 2026-09-10 | | Native-overlap gate before any new skill: the article's guidance IS the bundled claude-api skill | DECIDED: run `/harness-ops:audit-native-overlap` against the four topics first and record its verdicts as gate rows; adoption scope is NOT pre-restricted on paper. The owner receives full information per topic and decides at each lane interview. Amended 2026-09-11: a registry row alone is not the deliverable; each non-`defer` verdict lands as a `## Boundary` section in the skill body with detail in a same-skill reference file, and the native-references convention (1.1.0) now requires the pair | Owner interview, 2026-09-10 and 2026-09-11; PLUGIN-PHILOSOPHY Native-first section; ADR-0028 precedent | | Vendor-internal numbers and beta features | DECIDED: adopt mechanisms only; cite figures as vendor-reported and unreproduced; every adopted line touching a beta feature carries its beta qualifier and GA/model-list boundary | Owner interview, 2026-09-10 | -| Citing hillclimb while the public repo lags | DECIDED: cite it as a bundled Claude Code command with an upstream-drift record noting the public-repo lag; recheck trigger fires when the anthropics/skills repo or the skill's docs page gains the subcommand | Owner interview, 2026-09-10 | +| Citing hillclimb | DECIDED: cite it as a bundled Claude Code command with an upstream-drift record that points at the published guides (finding 1); recheck trigger fires when the platform claude-api skill docs page lists the subcommand | Owner interview, 2026-09-10; eval-design and hillclimbing interview, 2026-10-01 | ## Interview queue diff --git a/plugins/evals/.claude-plugin/plugin.json b/plugins/evals/.claude-plugin/plugin.json index 7be65caa7d..002af33aeb 100644 --- a/plugins/evals/.claude-plugin/plugin.json +++ b/plugins/evals/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "evals", - "version": "0.3.13", + "version": "0.4.0", "description": "LLM evaluation methodology, eval-suite design, and guided practice around the Claude Code eval runner, distilled from Anthropic's official evaluation guidance: a knowledge router over success criteria, eval design, and grading methods (/evals:methodology); an action skill that interviews for measurable success criteria and scaffolds a graded eval suite for an LLM app or a Claude Code skill (/evals:design); a guided runner that preflights the CLI and the target, prices a suite before it spends, and reads the with-versus-without delta (/evals:plugin-eval); and a static case-file check that spends nothing and makes no model call (/evals:validate).", "author": { "name": "Melodic Software", @@ -21,6 +21,45 @@ "title": "No cost ceiling", "description": "Drop --max-cost-usd from the invocation so a suite always runs to completion. The estimate is still printed before the run; only the ceiling goes away.", "default": false + }, + "split_policy": { + "type": "string", + "title": "Train/test split policy", + "description": "How eval cases split when a skill or prompt is tuned against them. train-test (the default) splits train and test as the bundled hillclimb guide does; reporting-only also holds back a split that no keep-or-revert decision or final pick reads, used only to report the result.", + "default": "train-test", + "options": ["train-test", "reporting-only"] + }, + "interval_method": { + "type": "string", + "title": "Interval method for pass counts", + "description": "Interval the noise report puts on the count of cases that pass the threshold. normal (the default) uses the normal approximation; wilson uses the Wilson score interval; jeffreys uses the Jeffreys interval. Score intervals and the with-versus-without difference always use normal.", + "default": "normal", + "options": ["normal", "wilson", "jeffreys"] + }, + "review_format": { + "type": "string", + "title": "Case review output format", + "description": "Format /evals:design renders candidate eval cases in for your approval. markdown (the default) is a table plus one fenced block per case; html is one page with every field escaped.", + "default": "markdown", + "options": ["markdown", "html"] + }, + "grader_run_twice": { + "type": "boolean", + "title": "Judge-vote agreement report", + "description": "When on, /evals:plugin-eval's noise report shows how often the judge votes for each llm grader agreed, read from votes the run already took, at no extra spend.", + "default": true + }, + "same_model_warning": { + "type": "boolean", + "title": "Same-model judge warning", + "description": "When on, /evals:plugin-eval's preflight warns when the tested model and the judge model resolve to the same model, and /evals:design repeats the reminder beside the build-eval route.", + "default": true + }, + "labelled_grader_check": { + "type": "boolean", + "title": "Labelled-set grader check", + "description": "When on, /evals:design checks a grader against a set of cases you label, not only the handful-of-cases agreement check.", + "default": false } }, "keywords": ["evals", "evaluation", "success-criteria", "llm-judge", "grading", "rubric", "testing", "knowledge", "skill"] diff --git a/plugins/evals/CHANGELOG.md b/plugins/evals/CHANGELOG.md index 2e55bee4f9..6f227c087e 100644 --- a/plugins/evals/CHANGELOG.md +++ b/plugins/evals/CHANGELOG.md @@ -1,5 +1,107 @@ # Changelog: evals +## [0.4.0] - 2026-10-02 + +### Added + +- **`plugin-eval` reports noise before a gain counts.** A new `scripts/noise-report.py` reads + `aggregate-result.json` and prints an interval on each arm's mean, a paired interval on the + with-versus-without difference, "within noise" when that interval contains 0, "n too small to + call" below 3 cases or when every per-case delta is equal, a warning when the without-arm mean + is 0.95 or higher, a pass-count view at the run's threshold, and judge-vote agreement per `llm` + grader. `## Reading the delta` tells the reader to run it. +- **The preflight warns when the tested model and the judge are the same model.** It is advice + only and never blocks a run. +- **`design` routes by repository kind and has every input approved.** A Claude API app is told + to type `/claude-api build-eval`; a skill or plugin repository continues here and runs through + `plugin-eval`. Candidate cases render for approval through a new `scripts/render-review.py`, + Markdown by default and escaped HTML as the option. Raw transcripts stay out of cases in a public + repository or one of unknown visibility. Graders are checked against a handful of cases before + their scores are trusted. +- **`methodology` points at the bundled hillclimb guides.** A new `reference/hillclimb.md` links + each step at a pinned commit and states only this plugin's facts; a new + `reference/local-decisions.md` holds this plugin's defaults and source conflicts. The skill + routes by repository kind. +- **Six settings:** `split_policy`, `interval_method`, `review_format`, `grader_run_twice`, + `same_model_warning` and `labelled_grader_check`. +- **The suite grows to 30 cases.** 21 are hard cases, each saying why it is hard; 4 are routine + guards where the base model already answers well; 4 are near-miss controls that must not invoke + an evals skill; and 1 is a knowledge case. `noise-before-gain` is the one a person judged hard: a + model tends to take a small gain over a near-ceiling baseline at face value. +- **Every `llm` grader has labelled samples.** All 47 hold must-pass and must-fail answers in + `samples/.json`, so each rubric can be calibrated against them. +- **`plugin-eval` calibrates a judge.** A new `scripts/calibrate-judge.py` builds one case per + labelled sample, whose agent replies with the sample word for word, and scores the judge's + verdicts against the labels: agreement, false positives and negatives, split votes, runs whose + reply was not the sample (whitespace and bold markers aside), and samples never judged. A grader under 90% agreement prints a + `FAIL grader` line and the script exits 1. `## Calibrating a judge` gives the commands. +- **`plugin-eval` checks a run is valid before its score counts.** A new + `scripts/run-validity.py` reads `aggregate-result.json` and the kept traces and prints VALID or + INVALID: an incomplete or empty run, skipped paid graders, an errored run, a row count that + differs from `--runs`, any permission denial in a trace, or a should-trigger case whose skill did + not fire makes it INVALID; split judge votes, an unrecorded judge model and ceiling cases are + warnings. The run command now keeps traces (`--keep-temp`). +- **`validate` tests free graders offline against sample answers.** Each case can hold + `samples/.json` with answers that must pass and answers that must fail; a `regex`, + `tool_used`, `tool_order` or `file_exists` grader that rejects a must-pass answer or accepts a + must-fail one is a FAIL, and one with no sample file is a WARN. `llm` graders with samples get a + calibration WARN, since only a paid judge run can check them. +- **`validate` warns on the two undocumented `prompt.md` keys.** `artifact_publish` and + `growthbook_overrides` load but are not on the reference page, so they can change without notice. + +### Changed + +- **The pilot suite's graders are fixed.** `methodology-wording` stays a `regex` but is an unscored + with-arm indicator (`arm: with-only`), and a new `llm` grader, `grading-choice`, scores answer + quality; `noise-verdict` is replaced by the `not-established` and `ceiling` graders, so one judge + error costs half a run; `four-properties` requires the target number itself to be justified. Each + case's description says what it measures, and the three `llm`-graded tracked cases name `--judge-model sonnet`. +- **`run-validity` exempts a no-trigger control.** A case is a control when its `prompt.md` tags or + description say so, or when it has a max-0 `Skill` grader aimed at this plugin's own skills and + no grader requiring a call. A guard on another skill does not make a control. +- **The four controls score their must-not-invoke guard in both arms** (`arm: both`), since + staying quiet is what a control measures. The routing cases' guards stay unscored indicators. +- **`methodology` states the target-number rule.** A target number is realistic only when a + measured baseline, a prior result, a benchmark or expert review justifies it; a grader checked + against labels, or a miss called severe, does not. +- **The `plugin-eval` hub records the reference-read denial.** A path-scoped `Read` grant does not + lift the refusal of a spoke file, so anything a case depends on goes in the hub. +- **The cost anchor for a fresh suite is 0.1 USD per run in either arm**, judge calls included, + called headroom. The old 0.8 USD per without-run applies only when a kept trace shows the + without-arm loading a large skill. +- **CI pins a full model ID for the agent and the judge**, never an alias such as `sonnet`. +- **`reading-results.md` says to count rows per arm**, never `runsPerCase`. +- **`plugin-eval`'s description names comparing two runs and asking whether a gain is real**, so + those questions route to it. +- **Each "volume" line says volume means cheaper grading, never easier cases.** +- **The design skill's tone test case expects a checkable pass/fail rubric**, not a 5-point + scale. +- **The "not yet public" distribution records are pointers to the published guides.** +- **The hubs answer without opening a spoke or a script**, since an eval run can read only the + hub. `validate` lists the `prompt.md` keys (`timeout_seconds`, not `timeout`), the grader types + and their options, and the bounds; a pattern check is `type: regex` with `pattern`. + `methodology` answers from its quick guide when it covers the question, and a rewritten + criterion grounds its target the way the success-criteria page's Achievable property allows; + with no data given it states the target relative to the current baseline, never a made-up + baseline figure, expert agreement or X/Y/Z placeholders. `plugin-eval` answers a CI question from its CI section, + names the pass-count line as the only one the interval setting changes, says an estimate under + the ceiling starts with `--max-cost-usd` and no prompt, says to rerun cases a usage limit + zeroed once it resets, and says the unavailable message leaves no access to request. +- **Descriptions route more asks to the skill that answers them.** `plugin-eval` names a + `claude plugin eval` refused by Bash or in a worktree, and whether it can measure CLAUDE.md or + rules; `methodology` and `design` name evals for a service that calls the Messages API and the + route for each part of a repository. +- **`run-validity` prints the path a denied call aimed at**, and a denial is a warning when that + run scored the same as every denial-free run of its case in the same arm. A with-arm denial + aimed at, under or above the plugin's directory or at no path, or any denial with a different + score or nothing to compare, stays a FAIL. On an INVALID line those warned cases are marked + "warnings only, not counted". +- **The hubs lead with the right first step.** `methodology` and `design` present the reference + files as background for a human reader, and both give `/claude-api build-eval` as a Claude API + app's first step, ahead of any hand-written criteria or cases. `plugin-eval` answers for the + platform the user states, never the session's own host, and with a null `error` runs the + validity gate and the noise report before saying anything about the cases. + ## [0.3.13] - 2026-10-02 ### Changed diff --git a/plugins/evals/README.md b/plugins/evals/README.md index 85aa7064fb..fa8e7edbc1 100644 --- a/plugins/evals/README.md +++ b/plugins/evals/README.md @@ -65,7 +65,7 @@ knowledge router with no decision contract, per the migration playbook's warrant ## Configuration -Two `userConfig` keys, both read by `/evals:plugin-eval`: +Eight `userConfig` keys. Two set the run's cost ceiling, read by `/evals:plugin-eval`: - **`max_cost_usd`** (number, default `5`): the ceiling passed to the CLI as `--max-cost-usd`. It bounds one invocation rather than a session's total, and a suite that reaches it stops partway @@ -73,6 +73,23 @@ Two `userConfig` keys, both read by `/evals:plugin-eval`: - **`unlimited_cost`** (boolean, default `false`): removes the ceiling and nothing else. The estimate still prints before the run, so an expensive suite is still visible before it starts. +Six set how evals are designed and read. Each default follows Anthropic's published eval guides; +where this repository departs from a source, the record is in the methodology skill's +`reference/local-decisions.md`. + +- **`split_policy`** (string, default `train-test`): `reporting-only` also holds back a split that + no keep-or-revert decision or final pick reads, used only to report the result. +- **`interval_method`** (string, default `normal`): `wilson` or `jeffreys` changes the interval on + the count of passing cases. Score intervals stay normal. +- **`review_format`** (string, default `markdown`): `html` renders the case review as one escaped + page instead. +- **`grader_run_twice`** (boolean, default `true`): reports judge-vote agreement per `llm` grader + from votes the run already took. +- **`same_model_warning`** (boolean, default `true`): warns when the tested model and the judge are + the same model. +- **`labelled_grader_check`** (boolean, default `false`): adds a check of each grader against cases + you label. + No hooks and no MCP servers. The methodology, design, and validate surfaces make no network calls and no model calls; a `claude plugin eval` run does, on your own account, which is what the estimate and the ceiling exist for. The one other outbound surface is the maintainer-only @@ -91,6 +108,12 @@ reads it from. | --- | --- | --- | --- | --- | | `max_cost_usd` | number
*min 0* | `5` | `CLAUDE_PLUGIN_OPTION_MAX_COST_USD` | Ceiling /evals:plugin-eval passes to the CLI as --max-cost-usd. It bounds one invocation, not a session's total, and a run that reaches it stops mid-suite with partial results. Raise it for a suite whose estimate exceeds it. | | `unlimited_cost` | boolean | `false` | `CLAUDE_PLUGIN_OPTION_UNLIMITED_COST` | Drop --max-cost-usd from the invocation so a suite always runs to completion. The estimate is still printed before the run; only the ceiling goes away. | +| `split_policy` | string | `"train-test"` | `CLAUDE_PLUGIN_OPTION_SPLIT_POLICY` | How eval cases split when a skill or prompt is tuned against them. train-test (the default) splits train and test as the bundled hillclimb guide does; reporting-only also holds back a split that no keep-or-revert decision or final pick reads, used only to report the result. | +| `interval_method` | string | `"normal"` | `CLAUDE_PLUGIN_OPTION_INTERVAL_METHOD` | Interval the noise report puts on the count of cases that pass the threshold. normal (the default) uses the normal approximation; wilson uses the Wilson score interval; jeffreys uses the Jeffreys interval. Score intervals and the with-versus-without difference always use normal. | +| `review_format` | string | `"markdown"` | `CLAUDE_PLUGIN_OPTION_REVIEW_FORMAT` | Format /evals:design renders candidate eval cases in for your approval. markdown (the default) is a table plus one fenced block per case; html is one page with every field escaped. | +| `grader_run_twice` | boolean | `true` | `CLAUDE_PLUGIN_OPTION_GRADER_RUN_TWICE` | When on, /evals:plugin-eval's noise report shows how often the judge votes for each llm grader agreed, read from votes the run already took, at no extra spend. | +| `same_model_warning` | boolean | `true` | `CLAUDE_PLUGIN_OPTION_SAME_MODEL_WARNING` | When on, /evals:plugin-eval's preflight warns when the tested model and the judge model resolve to the same model, and /evals:design repeats the reminder beside the build-eval route. | +| `labelled_grader_check` | boolean | `false` | `CLAUDE_PLUGIN_OPTION_LABELLED_GRADER_CHECK` | When on, /evals:design checks a grader against a set of cases you label, not only the handful-of-cases agreement check. | ### How to set these diff --git a/plugins/evals/evals/ci-pin-models/graders/pin-both.md b/plugins/evals/evals/ci-pin-models/graders/pin-both.md new file mode 100644 index 0000000000..cc10e63954 --- /dev/null +++ b/plugins/evals/evals/ci-pin-models/graders/pin-both.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the CI command pins both the model under test (`--model`) and the judge (`--judge-model`) with full model IDs (for example `claude-sonnet-5`), or the answer tells the user to pin both by full model ID, so a model rollout is not read as a plugin regression. Full IDs held in workflow variables count. + +FAIL if only one of the two models is pinned; either pin is an alias such as `sonnet`, `haiku` or `opus`, or the answer says an alias is enough to pin; the answer recommends tracking the latest model; or it later contradicts or retracts this. diff --git a/plugins/evals/evals/ci-pin-models/graders/read-json.md b/plugins/evals/evals/ci-pin-models/graders/read-json.md new file mode 100644 index 0000000000..9faaa1882b --- /dev/null +++ b/plugins/evals/evals/ci-pin-models/graders/read-json.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says the job should read the result JSON as well as the exit code, because the exit code alone does not say what happened (for example, exit 1 has several causes and exit 2 means a partial run). + +FAIL if it says the exit code alone is enough, reads the JSON only to post a summary or a report, or later contradicts or retracts this. diff --git a/plugins/evals/evals/ci-pin-models/graders/skill-fired.md b/plugins/evals/evals/ci-pin-models/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/ci-pin-models/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/ci-pin-models/graders/trust-plugin.md b/plugins/evals/evals/ci-pin-models/graders/trust-plugin.md new file mode 100644 index 0000000000..663481f22f --- /dev/null +++ b/plugins/evals/evals/ci-pin-models/graders/trust-plugin.md @@ -0,0 +1,5 @@ +--- +type: regex +pattern: "--trust-plugin" +arm: both +--- diff --git a/plugins/evals/evals/ci-pin-models/prompt.md b/plugins/evals/evals/ci-pin-models/prompt.md new file mode 100644 index 0000000000..6d2b0b590d --- /dev/null +++ b/plugins/evals/evals/ci-pin-models/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The base model pins the agent model at most, often by alias; the non-TTY trust refusal, pinning the judge as well, by full model ID, and the overloaded exit codes are this plugin's CI rules." +tags: [ci, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The command passes `--trust-plugin` and pins both `--model` and `--judge-model` by full model ID, not an alias; the answer says pinning both keeps a model rollout from reading as a plugin regression, and that the job reads the JSON as well as the exit code" +--- + +I want a GitHub Actions job that runs `claude plugin eval` on my plugin for every PR and fails the PR when the suite regresses. Give me the command the job should run and anything that's easy to get wrong. Answer in under 150 words. diff --git a/plugins/evals/evals/ci-pin-models/samples/pin-both.json b/plugins/evals/evals/ci-pin-models/samples/pin-both.json new file mode 100644 index 0000000000..2a072f5e78 --- /dev/null +++ b/plugins/evals/evals/ci-pin-models/samples/pin-both.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "why": "P-canonical: the command pins `--model claude-sonnet-5` and `--judge-model claude-haiku-4-5` (full IDs, not aliases) and says pinning both keeps a rollout from reading as a plugin regression.", + "answer": "Run this step, with your API key supplied from a repository secret:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin \\\n --model claude-sonnet-5 --judge-model claude-haiku-4-5 \\\n --json results.json --threshold 0.8\n```\n\nEasy to get wrong:\n\n1. Pin both models by full model ID, not an alias like `sonnet`, which can move to a newer model. If either the agent model or the judge changes under you, the scores move and a model rollout reads as a plugin regression.\n2. Pass `--trust-plugin`. A CI runner has no terminal, and an untrusted plugin directory is refused without it.\n3. Read `results.json`, not just the exit code. Exit 1 has several causes and exit 2 means a partial run, so the JSON tells you which happened." + }, + { + "why": "P-alt: pins both through workflow variables holding full model IDs and explains the rollout risk in different words (an unpinned agent or judge shifts every score and the job blames the plugin).", + "answer": "Keep both model choices fixed in the workflow, so the pull request is the only thing that changes between runs:\n\n```yaml\nenv:\n EVAL_MODEL: claude-sonnet-4-5\n JUDGE_MODEL: claude-haiku-4-5\n# step:\nrun: claude plugin eval ./plugins/notes --trust-plugin --model \"$EVAL_MODEL\" --judge-model \"$JUDGE_MODEL\" --json results.json\n```\n\nThe usual mistake is leaving the judge on its default. When a new model rolls out, an unpinned agent or judge shifts every score and the job blames your plugin for it. Bump the two IDs on purpose, in their own PR. Also parse `results.json` in a later step instead of trusting the exit code alone: exit 1 covers several different failures and exit 2 is a partial run." + }, + { + "why": "P-caveat: pins both by full model ID with the rollout reason; the caveat about bumping both pins together in a dedicated PR and the note on run-to-run noise keep the advice and do not undercut it.", + "answer": "The job step:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --model claude-sonnet-5 --judge-model claude-haiku-4-5 --json results.json\n```\n\nWhat trips people up:\n\n- Unpinned models. Pin the model under test and the judge, each by full ID; otherwise a model update moves the scores and looks like your plugin regressed. When you do want a newer model, change both pins in a dedicated PR and treat its numbers as the new baseline.\n- Trust. `--trust-plugin` is required, since the runner can't answer the trust prompt.\n- The exit code. It fails the job below threshold, but exit 1 has several causes and exit 2 means partial, so read the JSON too.\n- Noise. One PR's dip may be run-to-run variation; check before blaming the change.\n- Forks. Pull requests from forks don't get repository secrets, so decide whether those runs skip or fail." + }, + { + "why": "P-plain: pins both models by full ID with the rollout reason; it omits reading the JSON and omits `--trust-plugin`, which are other graders' points.", + "answer": "Have the workflow run the eval against your plugin directory with both models fixed:\n\n```bash\nclaude plugin eval ./plugins/notes --model claude-opus-5-5 --judge-model claude-haiku-4-5 --threshold 0.8\n```\n\nThe step exits non-zero when a case drops below the threshold, which fails the PR check.\n\nThe thing most people get wrong is model drift. Pin the model under test and the judge model, both of them. If either one floats, a model rollout shifts the scores and the job reports it as a plugin regression even though your code didn't change. Use full model IDs rather than aliases, and keep the two IDs in workflow variables so that updating them is a deliberate, reviewable change rather than something that happens to you." + } + ], + "fail": [ + { + "why": "F-wrong: recommends tracking the latest model with `--model latest` and leaves the judge unpinned.", + "answer": "Use this as the job step:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --model latest --json results.json\n```\n\nEasy to get wrong:\n\n- Testing against a stale model. Point `--model` at `latest` so every PR is checked against the newest model your users actually hit; a pinned model hides regressions until you upgrade.\n- The judge flag. Leave `--judge-model` at its default and let the runner keep it current.\n- `--trust-plugin`, which a headless runner needs or the run is refused.\n- The exit code. Read `results.json` as well, because exit 1 has several causes and exit 2 means a partial run." + }, + { + "why": "F-partial: pins only the model under test (`--model claude-sonnet-4-5`); the judge is never pinned. Everything else (trust flag, reading the JSON) is right.", + "answer": "Job command:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --model claude-sonnet-4-5 --json results.json --threshold 0.8\n```\n\nThings that go wrong:\n\n1. Model drift. Pin the model under test to a full ID so a model rollout isn't read as a plugin regression.\n2. Trust. Without `--trust-plugin`, a non-interactive runner refuses an untrusted plugin directory.\n3. Exit codes. Exit 1 is overloaded and exit 2 means a partial run, so parse `results.json` for each case's result before deciding the PR regressed.\n4. Secrets. Pull requests from forks don't receive repository secrets, so the API key is missing there; decide whether those runs skip or fail.\n5. Cost. Add `--max-cost-usd` so a suite that misbehaves can't run up a bill on every push." + }, + { + "why": "F-retract: the command pins both, then the answer calls the judge pin overkill and tells the user to drop `--judge-model`.", + "answer": "Pin both models in the command:\n\n```bash\nclaude plugin eval ./plugins/notes --model sonnet --judge-model haiku --json results.json\n```\n\nThat keeps a rollout from looking like a regression. On reflection, though, the judge pin is overkill: the judge only checks rubric wording, so a judge update barely moves the scores. Drop `--judge-model` and let it default, and you have one less thing to maintain.\n\nThe other common mistake is trusting the exit code by itself. Exit 1 has several causes and exit 2 means a partial run, so have a later step parse `results.json` and report which cases failed and whether the run completed." + }, + { + "why": "F-near: pins both models and gives the rollout reason, but with the aliases `sonnet` and `haiku`, and says an alias is a stable enough pin; trust flag and JSON reading are right.", + "answer": "Run:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --model sonnet --judge-model haiku --json results.json --threshold 0.8\n```\n\nWhat's easy to get wrong:\n\n- Pin both models, the one under test and the judge, so a model rollout can't pose as a plugin regression. The short names `sonnet` and `haiku` are enough for that; there's no need to hunt down dated model IDs.\n- `--trust-plugin` is mandatory on a runner, which can't answer the trust prompt.\n- Gate on the JSON as well as the exit code. Exit 1 has several causes and exit 2 means a partial run, so the JSON says which happened.\n- Forks don't get repository secrets, so decide whether those runs skip or fail." + } + ] +} diff --git a/plugins/evals/evals/ci-pin-models/samples/read-json.json b/plugins/evals/evals/ci-pin-models/samples/read-json.json new file mode 100644 index 0000000000..9d0efee50e --- /dev/null +++ b/plugins/evals/evals/ci-pin-models/samples/read-json.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "why": "P-canonical: says to add a step that reads `results.json` because exit 1 has several causes and exit 2 means a partial run.", + "answer": "The job step:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --model sonnet --judge-model haiku --json results.json\n```\n\nThen add a step that reads `results.json`. The exit code alone doesn't tell you what happened: exit 1 can mean a case fell below the threshold or one of several other failures, and exit 2 means the run was partial. Fail the PR on a real regression and report the other states for what they are.\n\nAlso easy to miss: pin both models so a model rollout isn't mistaken for a plugin regression, and keep `--trust-plugin`, because a runner with no terminal is refused without it." + }, + { + "why": "P-alt: makes the point reasoning-first in different words: a non-zero exit is not a diagnosis (six causes for 1, partial for 2), so inspect the written JSON before marking the PR red.", + "answer": "A non-zero exit is a signal, not a diagnosis. `claude plugin eval` returns 1 for six different causes and 2 when only part of the suite ran, so a job that gates on `$?` alone can't tell a regressed case from a broken run. Write the results to a file and inspect it in the next step:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin \\\n --model claude-sonnet-4-5 --judge-model claude-haiku-4-5 \\\n --json results.json\n```\n\nThe follow-up step should check whether the run was complete and which cases failed before it marks the PR red. Keep both model flags fixed too, or a model update shows up as a regression. A short `jq` script is enough for the check, and printing which cases failed makes a red check easy to act on." + }, + { + "why": "P-caveat: reads the JSON to decide whether the suite regressed (exit 1 overloaded, exit 2 partial); posting a summary is offered as an extra on top of gating, not as the only use, so the 'summary only' fail clause does not apply.", + "answer": "Command:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --model sonnet --judge-model haiku --json results.json --threshold 0.8\n```\n\nPitfalls:\n\n1. Reading only the exit code. Exit 1 is overloaded and exit 2 marks a partial run, so parse `results.json` too and let it say whether the suite actually regressed. Posting a short summary from it as a PR comment is a nice extra once the gate itself reads the file.\n2. Floating models. Pin `--model` and `--judge-model`, or a rollout looks like a regression.\n3. Trust. A headless runner needs `--trust-plugin`.\n4. Cost. Add `--max-cost-usd` so a runaway suite can't burn budget on every push." + }, + { + "why": "P-plain: says to parse the JSON as well as the exit code and why (exit 1 has several reasons, exit 2 partial); it pins no models and omits `--trust-plugin`, which are other graders' points.", + "answer": "Have the workflow run the suite and write machine-readable results:\n\n```bash\nclaude plugin eval ./plugins/notes --json results.json --threshold 0.8\n```\n\nThen add a second step that parses `results.json` before deciding the outcome. The most common mistake is treating the exit code as the whole story. Exit 1 is returned for several different reasons, only one of which is a case falling below the threshold, and exit 2 means the run was partial. The JSON tells you which of those happened and whether the comparison is even valid, so the job can fail on a genuine regression and flag a broken run separately." + } + ], + "fail": [ + { + "why": "F-wrong: says the exit code alone is enough and tells the user to skip `--json` and any parsing step; model pins and `--trust-plugin` are right.", + "answer": "The job step:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --model sonnet --judge-model haiku --threshold 0.8\n```\n\nThat's all the gate needs. The command exits non-zero whenever the suite falls below the threshold, and GitHub Actions fails the step on any non-zero exit, so the PR goes red on a regression without extra scripting. Skip `--json` and a parsing step; they add moving parts for no benefit.\n\nEasy to get wrong: pin both the model and the judge so a model rollout isn't mistaken for a regression, and keep `--trust-plugin`, since the runner can't answer the trust prompt. If you want a record, upload the job log as an artifact; the exit code already carries everything the gate needs to know." + }, + { + "why": "F-partial: writes `--json results.json` but never says the job should read it or why the exit code is not enough; every other point (pins, trust flag, cost, forks) is right.", + "answer": "Use this as the job step:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin \\\n --model claude-sonnet-4-5 --judge-model claude-haiku-4-5 \\\n --json results.json --threshold 0.8\n```\n\nEasy to get wrong:\n\n1. Pin both models. An unpinned agent or judge lets a model rollout show up as a plugin regression.\n2. `--trust-plugin` is required, because a non-interactive runner refuses an untrusted plugin directory.\n3. Keep the target before `--json`, or the flag swallows the path.\n4. Set `--max-cost-usd` so the suite can't overspend on a busy day.\n5. Pull requests from forks don't receive secrets, so decide how those runs behave." + }, + { + "why": "F-retract: states the read-the-JSON advice with its reason, then says to skip that step because the exit code is the signal you want.", + "answer": "Command for the job:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --model sonnet --judge-model haiku --json results.json\n```\n\nCommon advice is to parse `results.json` after the run, since exit 1 has several causes and exit 2 means partial. In practice you can skip that step: any non-zero exit fails the job, and that's exactly the signal you want for blocking a PR. Keep the JSON for when you need to debug locally.\n\nThe mistakes that really bite are unpinned models, where a rollout looks like a regression, and a missing `--trust-plugin`, which gets the run refused on a headless runner." + }, + { + "why": "F-near: lets the exit code alone decide pass or fail and reads the JSON only to post a summary comment, which the rubric fails.", + "answer": "Job step:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --model sonnet --judge-model haiku --json results.json --threshold 0.8\n```\n\nLet the exit code decide pass or fail; any non-zero status turns the check red. Then add a step that reads `results.json` and posts a summary table to the PR as a comment, so reviewers see each case's score without opening the logs.\n\nEasy to get wrong: pinning only one model (pin both, or a rollout looks like a regression), forgetting `--trust-plugin` on a headless runner, and not giving the comment step write permission on pull requests." + } + ] +} diff --git a/plugins/evals/evals/ci-pin-models/samples/skill-fired.json b/plugins/evals/evals/ci-pin-models/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/ci-pin-models/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/ci-pin-models/samples/trust-plugin.json b/plugins/evals/evals/ci-pin-models/samples/trust-plugin.json new file mode 100644 index 0000000000..9b1253381e --- /dev/null +++ b/plugins/evals/evals/ci-pin-models/samples/trust-plugin.json @@ -0,0 +1,26 @@ +{ + "pass": [ + { + "answer": "claude plugin eval ./plugins/notes --trust-plugin --model claude-sonnet-4-5 --judge-model claude-haiku-4-5 --json results.json", + "why": "in the command" + }, + { + "answer": "Easy to miss: a CI runner has no TTY, so without `--trust-plugin` the run is refused at the trust prompt.", + "why": "named as a gotcha" + } + ], + "fail": [ + { + "answer": "claude plugin eval ./plugins/notes --model sonnet --json results.json --threshold 0.8", + "why": "confident wrong: no trust flag" + }, + { + "answer": "Set CLAUDE_TRUST_ALL=1 in the job env so the trust prompt is skipped.", + "why": "an invented variable instead" + }, + { + "answer": "Pass --yes so the trust prompt auto-accepts in CI.", + "why": "an invented flag instead" + } + ] +} diff --git a/plugins/evals/evals/control-deepeval-geval/graders/names-geval.md b/plugins/evals/evals/control-deepeval-geval/graders/names-geval.md new file mode 100644 index 0000000000..48a37fab8b --- /dev/null +++ b/plugins/evals/evals/control-deepeval-geval/graders/names-geval.md @@ -0,0 +1,5 @@ +--- +type: regex +pattern: "GEval\\s*\\(" +arm: both +--- diff --git a/plugins/evals/evals/control-deepeval-geval/graders/no-evals-skill.md b/plugins/evals/evals/control-deepeval-geval/graders/no-evals-skill.md new file mode 100644 index 0000000000..a704030826 --- /dev/null +++ b/plugins/evals/evals/control-deepeval-geval/graders/no-evals-skill.md @@ -0,0 +1,8 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:methodology|plugin-eval|validate)|evals:design)\"" +min: 0 +max: 0 +arm: both +--- diff --git a/plugins/evals/evals/control-deepeval-geval/prompt.md b/plugins/evals/evals/control-deepeval-geval/prompt.md new file mode 100644 index 0000000000..9757dbba71 --- /dev/null +++ b/plugins/evals/evals/control-deepeval-geval/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Near-miss negative control: Near miss on the 'LLM judge' and 'grading rubric' triggers: a DeepEval GEval API question that belongs to DeepEval's docs. No evals skill should fire." +tags: [control] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer constructs a `GEval(...)` metric, and no evals skill is invoked" +--- + +In DeepEval, how do I set up a GEval metric with my own grading criteria so an LLM judge scores whether the answer is correct? Show the Python. Answer in under 120 words. diff --git a/plugins/evals/evals/control-deepeval-geval/samples/names-geval.json b/plugins/evals/evals/control-deepeval-geval/samples/names-geval.json new file mode 100644 index 0000000000..448c99ed28 --- /dev/null +++ b/plugins/evals/evals/control-deepeval-geval/samples/names-geval.json @@ -0,0 +1,26 @@ +{ + "pass": [ + { + "answer": "```python\nfrom deepeval.metrics import GEval\nfrom deepeval.test_case import LLMTestCaseParams\n\ncorrectness = GEval(\n name=\"Correctness\",\n criteria=\"Is the actual output factually correct given the expected output?\",\n evaluation_params=[LLMTestCaseParams.ACTUAL_OUTPUT, LLMTestCaseParams.EXPECTED_OUTPUT],\n)\n```", + "why": "full construction" + }, + { + "answer": "metric = GEval (name=\"Correct\", evaluation_steps=[\"Compare the answer with the expected output\"], evaluation_params=[...])", + "why": "space before the parenthesis" + } + ], + "fail": [ + { + "answer": "```python\nfrom deepeval.metrics import AnswerRelevancyMetric\nmetric = AnswerRelevancyMetric(threshold=0.7)\n```", + "why": "confident wrong: another metric class" + }, + { + "answer": "Subclass BaseMetric and call your judge model in measure(); that gives you full control of the criteria.", + "why": "a custom metric instead" + }, + { + "answer": "Use the G-Eval approach: write your criteria as a rubric and ask the judge for a 1-10 score.", + "why": "names G-Eval but builds nothing" + } + ] +} diff --git a/plugins/evals/evals/control-deepeval-geval/samples/no-evals-skill.json b/plugins/evals/evals/control-deepeval-geval/samples/no-evals-skill.json new file mode 100644 index 0000000000..a364cb9770 --- /dev/null +++ b/plugins/evals/evals/control-deepeval-geval/samples/no-evals-skill.json @@ -0,0 +1,87 @@ +{ + "pass": [ + { + "answer": [], + "why": "answered with no tool call" + }, + { + "answer": [ + { + "tool": "Grep", + "input": { + "pattern": "assert" + } + } + ], + "why": "searched the workspace, no skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a skill from another plugin" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "design" + } + } + ], + "why": "the bundled design skill, not evals:design" + } + ], + "fail": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "methodology fired on a near miss" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "plugin-eval fired" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "validate" + } + } + ], + "why": "bare evals skill name" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:design" + } + } + ], + "why": "evals:design fired" + } + ] +} diff --git a/plugins/evals/evals/control-no-trigger/graders/no-skill-fired.md b/plugins/evals/evals/control-no-trigger/graders/no-skill-fired.md index a01eeac9f7..a704030826 100644 --- a/plugins/evals/evals/control-no-trigger/graders/no-skill-fired.md +++ b/plugins/evals/evals/control-no-trigger/graders/no-skill-fired.md @@ -1,6 +1,7 @@ --- type: tool_used tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:methodology|plugin-eval|validate)|evals:design)\"" min: 0 max: 0 arm: both diff --git a/plugins/evals/evals/control-no-trigger/prompt.md b/plugins/evals/evals/control-no-trigger/prompt.md index 609d1f208b..fc010b51bd 100644 --- a/plugins/evals/evals/control-no-trigger/prompt.md +++ b/plugins/evals/evals/control-no-trigger/prompt.md @@ -1,10 +1,10 @@ --- -description: A pytest question that must be answered without invoking any skill in this plugin +description: "A pytest question that must be answered without invoking any skill in this plugin" tags: [control] runs: 3 max_turns: 10 allowed_tools: [Read, Glob, Grep, Skill] -expected_outcome: The answer names conftest.py and no Skill call is made +expected_outcome: "The answer names conftest.py and no Skill call to an evals skill is made" --- In pytest, how do I share one fixture across several test files without importing it in each file? Answer in under 100 words. diff --git a/plugins/evals/evals/control-no-trigger/samples/names-conftest.json b/plugins/evals/evals/control-no-trigger/samples/names-conftest.json new file mode 100644 index 0000000000..f9de30c46a --- /dev/null +++ b/plugins/evals/evals/control-no-trigger/samples/names-conftest.json @@ -0,0 +1,22 @@ +{ + "pass": [ + { + "answer": "Define the fixture in `tests/conftest.py`. pytest collects it on its own, so every test file under that directory can take the fixture as a parameter with no import. Add `scope=\"session\"` to build it once per run.", + "why": "file named in backticks with a path" + }, + { + "answer": "Move the fixture into a conftest.py file in the tests folder or any parent of it, and pytest makes it available to every test below that folder.", + "why": "file named in plain prose" + } + ], + "fail": [ + { + "answer": "Put the fixture in a conftest file next to your tests and pytest will find it.", + "why": "near miss: names a conftest file without the file name" + }, + { + "answer": "Put the fixtures in a module and list it in the pytest_plugins variable of your root test package.", + "why": "a different mechanism, never names the file" + } + ] +} diff --git a/plugins/evals/evals/control-no-trigger/samples/no-skill-fired.json b/plugins/evals/evals/control-no-trigger/samples/no-skill-fired.json new file mode 100644 index 0000000000..a9bc555321 --- /dev/null +++ b/plugins/evals/evals/control-no-trigger/samples/no-skill-fired.json @@ -0,0 +1,87 @@ +{ + "pass": [ + { + "answer": [], + "why": "answered with no tool call" + }, + { + "answer": [ + { + "tool": "Grep", + "input": { + "pattern": "conftest" + } + } + ], + "why": "searched the workspace but invoked no skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "another plugin's skill is outside this control" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "pytest-helper" + } + } + ], + "why": "a non-evals skill is outside this control" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "design" + } + } + ], + "why": "the bundled design skill, not evals:design" + } + ], + "fail": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "this plugin's skill fired on an off-topic prompt" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:design" + } + } + ], + "why": "evals:design fired" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare evals skill name" + } + ] +} diff --git a/plugins/evals/evals/control-promptfoo-json/graders/names-is-json.md b/plugins/evals/evals/control-promptfoo-json/graders/names-is-json.md new file mode 100644 index 0000000000..b3c7311d3e --- /dev/null +++ b/plugins/evals/evals/control-promptfoo-json/graders/names-is-json.md @@ -0,0 +1,5 @@ +--- +type: regex +pattern: "(?`, `--runs 1` and `--ablation none` together for iterating on one case. Explanations such as "no baseline arm" or "so you don't pay for the baseline arm" are fine. A separate, later step that confirms a change at 3 runs with the default ablation is not a contradiction of the iteration flags. + +FAIL if any of the three flags is missing or has another value; another filter (`--tag`, `--only`) replaces `--case`; or the answer later contradicts or retracts one of them. diff --git a/plugins/evals/evals/iterate-then-confirm/graders/skill-fired.md b/plugins/evals/evals/iterate-then-confirm/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/iterate-then-confirm/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/iterate-then-confirm/prompt.md b/plugins/evals/evals/iterate-then-confirm/prompt.md new file mode 100644 index 0000000000..0f4951a9f9 --- /dev/null +++ b/plugins/evals/evals/iterate-then-confirm/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The flag combination is this plugin's guidance on a CLI the base model does not know; the blind base answers had no case flag and asked for 3 to 5 or 5+ runs to confirm. Self-reference: confirm-noise-report checks this plugin's own report." +tags: [local-defaults, hard, self-reference] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer gives `--case --runs 1 --ablation none` for iteration, confirms at 3 runs per case, and reads the confirm run's noise report before trusting the change" +--- + +I'm adjusting one plugin eval case over and over to get its grader right. What's the cheapest way to re-run only that case while I iterate, and how many runs per case should I use before I believe a change actually helped? Answer in under 120 words. diff --git a/plugins/evals/evals/iterate-then-confirm/samples/confirm-at-three.json b/plugins/evals/evals/iterate-then-confirm/samples/confirm-at-three.json new file mode 100644 index 0000000000..82d2e6fb4a --- /dev/null +++ b/plugins/evals/evals/iterate-then-confirm/samples/confirm-at-three.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Iterate with `claude plugin eval ./plugins/notes --case grader-tune --runs 1 --ablation none`: one case, one run, no baseline arm.\n\nA single run is noisy, so don't believe an improvement from it. Confirm the change at 3 runs per case, the default, and read the confirm run's noise report before you trust the result. If the report backs the change, keep it. While you iterate, each edit costs just one with-plugin run of that single case, and the output shows `SCORE` and `PASS%`.", + "why": "P-canonical: says to confirm at 3 runs per case, the default." + }, + { + "answer": "For the tuning loop, `--case grader-tune --runs 1 --ablation none` is the cheapest it gets.\n\nBelieving the change is a separate step. Go back to the default of three runs for each case, run the confirm, and only treat the improvement as real once its noise verdict backs it up. Anything you saw at one run is a hint, not evidence. Skipping the baseline during the loop means each edit costs one with-plugin run of that case.", + "why": "P-alt: says the confirm uses the default three runs for each case, in different words." + }, + { + "answer": "Re-run only that case with `--case grader-tune --runs 1 --ablation none`. Skipping the baseline arm means the output shows `SCORE` and `PASS%` instead of with, without and a delta, which is all you need while you tune the grader.\n\nBefore you believe a change helped, confirm it at 3 runs per case. That's the default, so you can simply drop `--runs 1`, keep the baseline arm if you want the delta back, and read the confirm's noise report before trusting the result.", + "why": "P-caveat: confirm at 3 runs per case, the default; the SCORE/PASS% note does not touch the graded point." + }, + { + "answer": "While you iterate, run just that case once: `claude plugin eval ./plugins/notes --case grader-tune --runs 1`. It's quick and cheap, and you can repeat it after every edit.\n\nOne run moves around too much to judge by, so before you believe the change helped, confirm at 3 runs per case. If the score at 3 runs is higher than your previous version's, the change is real. Keep the iteration command handy; you'll run it many times before the confirm.", + "why": "P-plain: confirm at 3 runs per case; misses sibling points (no `--ablation none`, trusts the score without the noise report)." + } + ], + "fail": [ + { + "answer": "Iterate with `claude plugin eval ./plugins/notes --case grader-tune --runs 1 --ablation none`, which runs only that case, once, without the baseline.\n\nSingle runs are noisy, so for the confirm you want real repetition: 5 runs per case at minimum, 10 if the case is borderline. Read that confirm's noise report before trusting the change. Each iteration in the meantime costs just one with-plugin run, and the output shows `SCORE` and `PASS%` instead of a delta, which is enough to see whether the grader now behaves.", + "why": "F-wrong: says the confirm needs 5 to 10 runs per case; sibling points (flags, noise report) are right." + }, + { + "answer": "Use `--case grader-tune --runs 1 --ablation none` while you tune: one case, one run, no baseline arm.\n\nA single run can't tell a real improvement from noise, so once the grader looks right, do a confirm run with several runs per case and read its noise report before you trust that the change helped. While iterating, the output shows `SCORE` and `PASS%` rather than a with-versus-without delta, because the baseline arm is skipped; that's all you need to tell whether the grader now passes the answers it should.", + "why": "F-partial: gives no run count for the confirm (only 'several')." + }, + { + "answer": "Tune with `--case grader-tune --runs 1 --ablation none` so each edit costs one with-plugin run.\n\nTo confirm, the default is 3 runs per case. Honestly, three is too few to separate a real gain from luck on a single case, so bump the confirm to `--runs 5` and read the noise report before you believe the change helped. Iterating stays cheap either way; only the confirm runs the case more than once, so the extra runs cost little.", + "why": "F-retract: states 3 runs per case, then says 3 is not enough and moves to 5." + }, + { + "answer": "The cheap loop is `claude plugin eval ./plugins/notes --case grader-tune --runs 1 --ablation none`: one case, one run, no baseline arm.\n\nDon't trust anything you see at one run. When you think the grader is right, confirm with 3 to 5 runs per case and read the confirm's noise report before you believe the change helped. Each iteration costs a single with-plugin run of that case, and the output shows `SCORE` and `PASS%` rather than a delta while you tune.", + "why": "F-near: '3 to 5 runs per case', which the rubric lists as a FAIL; sibling points are right." + } + ] +} diff --git a/plugins/evals/evals/iterate-then-confirm/samples/confirm-noise-report.json b/plugins/evals/evals/iterate-then-confirm/samples/confirm-noise-report.json new file mode 100644 index 0000000000..a1049f9f3d --- /dev/null +++ b/plugins/evals/evals/iterate-then-confirm/samples/confirm-noise-report.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Iterate with:\n\n```bash\nclaude plugin eval ./plugins/notes --case grader-tune --runs 1 --ablation none\n```\n\nOne run is noisy, so when the grader looks right, confirm at 3 runs per case. Then read the confirm's noise report, its verdict and interval, before you trust that the change helped; the raw score alone can move by chance. While you iterate, each edit costs one with-plugin run of that case, and the output shows `SCORE` and `PASS%` instead of a with-versus-without delta.", + "why": "P-canonical: says to read the confirm's noise report (verdict and interval) before trusting the change." + }, + { + "answer": "Keep each edit cheap: `--case grader-tune`, `--runs 1`, `--ablation none`.\n\nA higher number on a single run tells you little. Re-run the case at the default 3 runs, then look at the noise section of that run's report: if its verdict says the difference could be noise, you haven't shown the change helped yet. Only believe it once that verdict supports it. Until then, keep iterating at one run per edit; it's cheap, and it's only a signal that the grader behaves.", + "why": "P-alt: says to check the confirm run's noise verdict before believing the change, in different words." + }, + { + "answer": "Iterate on that case alone with `--case grader-tune --runs 1 --ablation none`; each edit then costs one with-plugin run and prints `SCORE` and `PASS%`.\n\nTo believe a change helped, confirm it at 3 runs per case and read the confirm's noise report first: its verdict and interval tell you whether the difference is bigger than run-to-run noise. Don't mix in the one-run scores from your tuning loop when you judge it; they were a quick signal that the grader behaves, never evidence.", + "why": "P-caveat: reads the confirm's noise report first; the note about ignoring one-run scores does not touch the graded point." + }, + { + "answer": "Run only that case, once, while you tune: `claude plugin eval ./plugins/notes --case grader-tune --runs 1`.\n\nTo believe a change, run a confirm with `--runs 5` and read that run's noise report. Its verdict tells you whether the improvement stands out from run-to-run variation; if it doesn't, the change hasn't been shown to help, however good the score looks. Iterating at one run keeps each edit cheap; the confirm is where you spend more, so make its report the thing you judge by.", + "why": "P-plain: reads the confirm's noise report verdict; wrong on sibling points (5 runs, no `--ablation none`)." + } + ], + "fail": [ + { + "answer": "Iterate with `--case grader-tune --runs 1 --ablation none`, so you pay for one with-plugin run per edit and skip the baseline.\n\nTo believe a change helped, rerun at 3 runs per case and look at the delta. Three runs already average out most of the run-to-run noise, so if the delta is positive and the pass rate went up, the change worked; if it's flat or negative, revert it. Either way, you've only paid for the extra runs once.", + "why": "F-wrong: trusts the confirm's delta and pass rate without the noise report; sibling points (flags, 3 runs) are right." + }, + { + "answer": "Use `claude plugin eval ./plugins/notes --case grader-tune --runs 1 --ablation none` while tuning: one case, one run, no baseline arm.\n\nA single run is too noisy to judge by. Once the grader looks right, confirm the change at 3 runs per case, the default, and compare that result with the version you started from before deciding to keep it. Iterating this way keeps each edit to a single with-plugin run, and the output shows `SCORE` and `PASS%` instead of a delta.", + "why": "F-partial: otherwise correct but never says to read the confirm's noise report." + }, + { + "answer": "Iterate on one case with `--case grader-tune --runs 1 --ablation none`.\n\nFor the confirm, rerun at 3 runs per case. The run prints a noise report with a verdict and interval, and the careful move is to read it. In practice you can skip it for a single-case change: if the score went up at 3 runs, that's good enough to keep the edit. Iterating at one run per edit keeps the tuning loop itself cheap and quick.", + "why": "F-retract: mentions reading the noise report, then says to skip it and trust the score." + }, + { + "answer": "Tune with `claude plugin eval ./plugins/notes --case grader-tune --runs 1 --ablation none`.\n\nWhen it looks right, confirm at 3 runs per case. Then look at the three individual scores: if all three land above where the case sat before your edit and they're close to each other, the improvement is real; if they scatter, the change is noise. Iterating at one run per edit keeps the tuning loop cheap; the confirm is the only time you pay for more than one run.", + "why": "F-near: replaces the noise report with eyeballing the three run scores; sibling points are right." + } + ] +} diff --git a/plugins/evals/evals/iterate-then-confirm/samples/iteration-flags.json b/plugins/evals/evals/iterate-then-confirm/samples/iteration-flags.json new file mode 100644 index 0000000000..4fc9a603b8 --- /dev/null +++ b/plugins/evals/evals/iterate-then-confirm/samples/iteration-flags.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Iterate on just that case with a single run and no baseline arm:\n\n```bash\nclaude plugin eval ./plugins/notes --case grader-tune --runs 1 --ablation none\n```\n\nThat reports `SCORE` and `PASS%` instead of with, without and delta, and it's the cheapest loop. One run is noisy, though, so once the grader looks right, confirm the change at 3 runs per case and read that run's noise report before believing it helped. Each iteration costs a single with-plugin run of one case.", + "why": "P-canonical: gives `--case`, `--runs 1` and `--ablation none` together in one command." + }, + { + "answer": "Cheapest loop: `--ablation none --runs 1 --case grader-tune`. Dropping the ablation skips the baseline arm entirely, one run keeps the cost per edit minimal, and the case filter keeps every other case out of it.\n\nWhen you think you've got it, do a proper confirm: 3 runs per case, which is the default, and check the noise verdict before you trust the result. Expect each edit to cost one with-plugin run of that one case, and the output to show `SCORE` and `PASS%` rather than a delta.", + "why": "P-alt: the same three flags in a different order with an explanation of each." + }, + { + "answer": "Use `--case --runs 1 --ablation none` on every edit. The `--case` filter runs only that case, `--runs 1` runs it once, and `--ablation none` skips the no-plugin arm so you don't pay for the baseline each time.\n\nA caveat while iterating: with no baseline arm the output shows `SCORE` and `PASS%`, not a delta, so you're tuning the grader's verdict on the with-plugin answer, not measuring the plugin's effect.\n\nTo believe a change helped, rerun at 3 runs per case and read the noise report's verdict before trusting it.", + "why": "P-caveat: all three flags, with the no-baseline-arm explanation the rubric says is fine and a caveat about the missing delta." + }, + { + "answer": "While iterating, run only that case once, without the baseline:\n\n```bash\nclaude plugin eval ./plugins/notes --case grader-tune --runs 1 --ablation none\n```\n\nEach edit costs one with-plugin run. Single runs swing a lot, so before you believe a change helped, rerun with `--runs 5` and compare the average score against what you had before the edit; if it's clearly higher, keep the change. Keep that command in your shell history; you'll run it after every grader edit, and each run finishes quickly.", + "why": "P-plain: all three iteration flags; wrong on sibling points (confirms at 5 runs, trusts the score with no noise report)." + } + ], + "fail": [ + { + "answer": "Copy the suite to a scratch directory that contains only the case you're tuning, and point the command there with `--runs 1`:\n\n```bash\nclaude plugin eval ./scratch/notes-one-case --runs 1\n```\n\nThat limits the run to a single case and a single run, which is as cheap as it gets. Once the grader behaves, confirm the change at 3 runs per case and read the confirm's noise report before trusting it. Delete the scratch copy when you're done so it doesn't drift from the real suite.", + "why": "F-wrong: replaces `--case` with a copied one-case directory and omits `--ablation none`; sibling points (3 runs, noise report) are right." + }, + { + "answer": "Filter to the one case and run it once:\n\n```bash\nclaude plugin eval ./plugins/notes --case grader-tune --runs 1\n```\n\nRepeat that after each edit to the grader. A single run is noisy, so when you think the change works, confirm it at 3 runs per case and read the noise report the confirm prints before you believe it helped. Each pass gives you the with-plugin and no-plugin scores for that case, so you can see the delta move as you edit the grader.", + "why": "F-partial: gives `--case` and `--runs 1` but is missing `--ablation none`." + }, + { + "answer": "The documented iteration loop is `--case grader-tune --runs 1 --ablation none`. I'd drop the `--ablation none` part, though: without the baseline arm you lose the with-versus-without delta, which is exactly what tells you whether your edit mattered. Run `--case grader-tune --runs 1` and keep the baseline.\n\nWhen the grader looks right, confirm at 3 runs per case and check the noise report's verdict before trusting the change. The baseline costs a second run per edit, but it's worth it.", + "why": "F-retract: states all three flags, then retracts `--ablation none`." + }, + { + "answer": "Tag the case you're tuning, say `wip`, and iterate with:\n\n```bash\nclaude plugin eval ./plugins/notes --tag wip --runs 1 --ablation none\n```\n\nThat runs only the tagged case, once, with no baseline arm. When the grader looks right, run a confirm at 3 runs per case and read its noise report before you believe the change helped. Tags live in the case's `case.yaml`, so you can move the `wip` tag to whichever case you're working on next without touching the command.", + "why": "F-near: `--tag` replaces `--case`, which the rubric lists as a FAIL; sibling points are right." + } + ] +} diff --git a/plugins/evals/evals/iterate-then-confirm/samples/skill-fired.json b/plugins/evals/evals/iterate-then-confirm/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/iterate-then-confirm/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/measurable-criterion/graders/four-properties.md b/plugins/evals/evals/measurable-criterion/graders/four-properties.md index 4f741b149d..6a277cda55 100644 --- a/plugins/evals/evals/measurable-criterion/graders/four-properties.md +++ b/plugins/evals/evals/measurable-criterion/graders/four-properties.md @@ -7,7 +7,7 @@ PASS only if the rewritten criterion meets all four of these: 1. Specific: it names a concrete quality to measure (for example "no unsafe medical advice as judged by a defined rubric" or "no unverified dosage instructions"), not the bare word "safe". 2. Measurable: it states a number or a defined scale AND the set of trials it is measured over (for example "fewer than 0.5% of 5,000 simulated intake conversations flagged by the safety classifier", or "mean rubric score at least 4 of 5 across 500 reviewed transcripts"). -3. Achievable: it grounds the target in something concrete, such as a current baseline, a prior result, an industry benchmark, or expert review, rather than an unexplained aspiration. +3. Achievable: the target number itself is justified by something concrete: a current baseline, a prior result, an industry benchmark, published research, or expert judgment used to set the target (for example "down from the 2% measured last month", or "thresholds agreed with the clinical team"). A stated, measured current baseline that the target improves on is enough by itself, however large the improvement: judge whether a justification is given, not whether the gap looks realistic. A target stated relative to the current baseline without giving the baseline's value (for example "half the current rate", or "a 5% improvement over the current baseline") counts. A baseline figure or an expert agreement that the answer presents as fact but invents (the prompt gives none) does not count. Checking the grader against clinician labels does not count as that justification (mentioning it as well is fine), and neither does saying a target is strict because a failure would be severe: those explain how the criterion is measured or why it matters, not why the number is reachable. 4. Relevant: it ties the criterion to the medical-intake context or its users (patients, clinicians, intake accuracy, escalation to a human). -FAIL if any one of the four is missing, if the rewrite still uses an unquantified word like "safe" or "appropriate" as the measure itself, or if no rewritten criterion is given. +FAIL if any one of the four is missing, if the rewrite still uses an unquantified word like "safe" or "appropriate" as the measure itself, if no rewritten criterion is given, or if the answer later contradicts or retracts this. diff --git a/plugins/evals/evals/measurable-criterion/graders/skill-fired.md b/plugins/evals/evals/measurable-criterion/graders/skill-fired.md index 2406de5402..d984213085 100644 --- a/plugins/evals/evals/measurable-criterion/graders/skill-fired.md +++ b/plugins/evals/evals/measurable-criterion/graders/skill-fired.md @@ -1,5 +1,5 @@ --- type: tool_used tool: Skill -input_match: '"skill"\s*:\s*"(?:evals:)?methodology"' +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:methodology))\"" --- diff --git a/plugins/evals/evals/measurable-criterion/prompt.md b/plugins/evals/evals/measurable-criterion/prompt.md index a3637fdc03..5b2523b442 100644 --- a/plugins/evals/evals/measurable-criterion/prompt.md +++ b/plugins/evals/evals/measurable-criterion/prompt.md @@ -1,10 +1,10 @@ --- -description: Rewrite a vague success criterion into one that meets the four properties in the success-criteria reference +description: "Rewrite a vague success criterion into one that meets the four properties in the success-criteria reference, with Achievable read strictly. The target number must rest on a baseline, prior result, benchmark, or expert judgment; a grader checked against clinicians or a target called strict because harm is severe does not count. The base model writes strong criteria, so expect a small with-versus-without gap; the case also keeps the llm grader path and the skill-fired indicator covered. Job 2 scored 0/3 in both arms because the baseline-anchored target rule was not yet in the methodology hub; it is now. Run it with --judge-model sonnet" tags: [methodology, knowledge] runs: 3 max_turns: 10 allowed_tools: [Read, Glob, Grep, Skill] -expected_outcome: The rewritten criterion names a specific quality, a number or defined scale over a stated trial set, an achievable target grounded in a baseline or benchmark, and the medical-intake user need +expected_outcome: "The rewritten criterion names a specific quality, a number or defined scale over a stated trial set, a target grounded in a baseline, prior result, benchmark, or expert judgment, and the medical-intake user need" --- Rewrite this success criterion so it is actually usable for evaluating an LLM application: "Our medical intake chatbot should give safe answers." Give the rewritten criterion, then briefly explain what you changed and why. Answer in under 200 words. diff --git a/plugins/evals/evals/measurable-criterion/samples/four-properties.json b/plugins/evals/evals/measurable-criterion/samples/four-properties.json new file mode 100644 index 0000000000..ff0891c571 --- /dev/null +++ b/plugins/evals/evals/measurable-criterion/samples/four-properties.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "why": "P-canonical: specific behaviors (diagnosis, dosing, delayed care, escalation), rates over a stated set of 1,000 conversations, both targets set relative to the current bot's rate on that set with no invented figure, tied to intake patients reaching a clinician.", + "answer": "**Rewritten criterion:** Across a fixed test set of 1,000 simulated intake conversations, including 200 red-flag scenarios written by clinicians, the share of replies that contain unsafe medical advice (a diagnosis, a dosing instruction, or advice that delays care), as judged by a defined rubric, falls to a third of the share the current bot produces on the same set, and the share of red-flag scenarios escalated to a human clinician closes at least half the gap between the current bot's measured rate and 100%.\n\n**What changed and why:**\n- \"Safe\" became concrete behaviors a grader can check: no diagnosis, no dosing, no delayed care, and escalation of red flags.\n- Each behavior has a number and a defined trial set, so the result is a measurement, not an opinion.\n- The targets are set relative to the current bot's rate on the same set, so they are reachable rather than aspirational.\n- It is tied to what matters for intake: patients with urgent symptoms reaching a clinician.\n- The rubric and the test set are fixed, so next month's run can be compared directly with this one." + }, + { + "why": "P-alt: quoted criterion in a different structure: specific behaviors, rates over a 600-conversation set (80 emergency cases), the bar set relative to the human clerks' measured performance on the same set with no invented figure, tied to intake.", + "answer": "\"When run against our 600-conversation intake regression set, the chatbot gives a diagnosis or medication dosing in no more replies than the human intake clerks do on the same 600 conversations, and routes at least as many of the 80 emergency-symptom conversations in that set to the on-call nurse line as the clerks do, as scored by a written clinical-safety rubric.\"\n\nI swapped the vague word for specific behaviors that could harm a patient, put a measure on each, and fixed the set of conversations it is measured on so results are comparable from run to run. The bar is the human clerks' measured performance on the same set, which makes it a realistic target rather than a guess, and the criterion stays focused on intake: what the bot must not say, and when it must hand the patient to a person." + }, + { + "why": "P-caveat: all four properties (specific failures, rates over 2,000 conversations, targets set relative to the current bot's measured rate, intake scope), plus a caveat that grader validation checks the measurement, not the target.", + "answer": "Rewritten: \"On 2,000 synthetic intake conversations built from real triage scenarios, the share of chatbot replies that give an unverified medication dose or a diagnosis, as flagged by a rubric-based grader, falls by at least half from the current bot's rate on the same set, and the share of conversations that mention chest pain, stroke signs or suicidal thoughts that are escalated to a human in the same turn closes at least half the gap between the current bot's rate and 100%.\"\n\nChanges: replaced \"safe\" with the specific failures that hurt patients, added a rate and a trial set for each, anchored the targets to the current bot's measured rate, and kept the scope to intake.\n\nOne caveat: check the grader against clinician-labelled replies before trusting the measured rates. That validates the measurement; the targets themselves still rest on the current bot's rate." + }, + { + "why": "P-plain: minimal single-target rewrite: one concrete behavior (diagnosis or dose), a rate over 500 conversations, a target set relative to the current version's rate on the same set with no invented figure, tied to patient risk at intake.", + "answer": "Rewritten criterion: \"Across the 500 recorded intake conversations in our test set, the share of chatbot replies that give a diagnosis or a medication dose, as flagged by a rubric-based grader, falls to a quarter of the share the current version gives on the same set.\"\n\nWhat changed: \"safe\" became one concrete behavior that puts patients at risk during intake, the bot overstepping into diagnosis or dosing. It is now a rate over a fixed set of conversations, so two runs can be compared directly. The target is set against the current version's rate on the same set, so it is a reachable improvement rather than a guess. The rubric defines what counts as a diagnosis or a dose, so the grader and a reviewer reading the same reply reach the same call." + } + ], + "fail": [ + { + "why": "F-wrong: the rewrite still uses \"safe and appropriate\" as the measure itself, even though it adds a number, a sample and a baseline.", + "answer": "Rewritten criterion: \"At least 98% of the chatbot's answers are rated safe and appropriate by a panel of three clinicians, across a sample of 300 intake conversations reviewed each month, up from the 93% we see today.\"\n\nWhat changed: the original had no number, no sample and no reviewer. Now there's a clear pass rate, a monthly sample of fixed size, and named judges with clinical expertise, which makes it measurable and auditable. The target builds on the current 93%, so it is achievable, and it is relevant because clinician review reflects real patient risk at intake. Anyone on the team can read the monthly number and know whether the bot is holding the line. Where the panel disagrees on a reply, the majority rating counts." + }, + { + "why": "F-partial: specific, achievable (measured 2%) and relevant, but no set of trials is stated for the 0.5% rate, so Measurable is missing.", + "answer": "Rewritten criterion: \"The share of chatbot replies that give a diagnosis, a medication dose, or advice to wait on red-flag symptoms stays below 0.5%, as flagged by a rubric-based grader, down from the 2% we measured last month.\"\n\nWhat changed and why: \"safe\" is replaced with three behaviors that put intake patients at risk, so a grader can look for them in a reply. The criterion now carries a number instead of an adjective, which makes it something you can track over time. The target is set against the measured 2%, so it is ambitious but reachable, and it keeps attention on the patients who come through intake with symptoms that need a clinician rather than a chatbot. The rubric lists example replies for each behavior so the grader's calls stay consistent." + }, + { + "why": "F-retract: gives a full four-property criterion, then tells the user to skip the numbers and test set and keep \"answers are safe\" for now.", + "answer": "Rewritten criterion: \"Across 800 simulated intake conversations, fewer than 1% of replies include a diagnosis or a dosing instruction, as judged by a written rubric, and every one of the 120 red-flag conversations is escalated to a nurse; both targets improve on the 3% and 91% measured in our March pilot.\"\n\nWhat changed: \"safe\" became specific behaviors, each with a number and a defined set, anchored to the pilot and tied to intake patients.\n\nIn practice, though, don't get hung up on the exact percentages or the test set. For a first version it's fine to keep the criterion as \"answers are safe\", have a clinician look over anything that seems off, and add numbers later once the bot is live." + }, + { + "why": "F-near: specific, measurable and relevant, but the targets are justified only by severity and by a clinician-checked grader, which the rubric says does not count as Achievable.", + "answer": "Rewritten: \"Across 1,000 simulated intake conversations, including 150 red-flag cases written by clinicians, at least 99% of red-flag cases are escalated to a human clinician and under 0.5% of replies give a diagnosis or a dose, scored by an LLM grader that was first checked against 100 clinician-labelled replies.\"\n\nWhat changed: \"safe\" is broken into two failure modes that matter at intake, each with a number and a defined set of conversations. The grader is validated against clinician labels, so the measurement can be trusted. The targets are deliberately strict because a miss here can hurt a patient, and an intake tool has to be held to that standard from day one. Results are reported per failure mode, so a drop in escalation can't hide behind a good dosing number." + } + ] +} diff --git a/plugins/evals/evals/measurable-criterion/samples/skill-fired.json b/plugins/evals/evals/measurable-criterion/samples/skill-fired.json new file mode 100644 index 0000000000..57c4a048b2 --- /dev/null +++ b/plugins/evals/evals/measurable-criterion/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "methodology" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:design" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/no-model-in-the-loop/graders/does-not-apply.md b/plugins/evals/evals/no-model-in-the-loop/graders/does-not-apply.md new file mode 100644 index 0000000000..36cd4b0c25 --- /dev/null +++ b/plugins/evals/evals/no-model-in-the-loop/graders/does-not-apply.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says LLM eval design does not apply here because nothing in the Go CSV tool calls a model, and does not go on to design an LLM eval. Recommending ordinary tests (Go unit tests, table-driven tests, golden files, fuzzing) is fine. + +FAIL if the answer designs an LLM eval suite for the tool anyway; proposes adding a model call so that LLM evals can be built; proposes grading the tool's CSV output with an LLM judge or rubric; or later contradicts or retracts this. diff --git a/plugins/evals/evals/no-model-in-the-loop/prompt.md b/plugins/evals/evals/no-model-in-the-loop/prompt.md new file mode 100644 index 0000000000..080dc30d9c --- /dev/null +++ b/plugins/evals/evals/no-model-in-the-loop/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Routine guard (regression-guard). Says LLM eval design does not apply to a tool with no model in it, and stops. Expect 1.00 in both arms (both blind best-guess answers pass)." +tags: [routing, routine, regression-guard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer says LLM eval design does not apply because nothing in the tool calls a model, and stops; ordinary tests are fine to mention" +--- + +My manager wants 'LLM evals' added to our repo. It's a Go command-line tool that parses CSV files, and nothing in it calls any model. How should we set those evals up? Answer in under 100 words. diff --git a/plugins/evals/evals/no-model-in-the-loop/samples/does-not-apply.json b/plugins/evals/evals/no-model-in-the-loop/samples/does-not-apply.json new file mode 100644 index 0000000000..37710e75fa --- /dev/null +++ b/plugins/evals/evals/no-model-in-the-loop/samples/does-not-apply.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "LLM evals don't fit this repo: nothing in the Go tool calls a model, so there is no model output to evaluate. What your manager probably wants is confidence the tool behaves correctly, and ordinary Go tests give that. Add table-driven unit tests for the parser, golden-file tests that compare output for sample CSVs, and a fuzz test (`go test -fuzz`) for malformed input, all running under `go test ./...` in CI.", + "why": "P-canonical: says LLM evals do not apply because nothing calls a model, and recommends ordinary Go tests, which the rubric allows." + }, + { + "answer": "There's nothing here for an LLM eval to measure. Those evals score what a model produces, and this CSV parser never sends anything to a model; every output comes from deterministic Go code. So tell your manager the right quality check for this repo is its test suite: `go test ./...` with table-driven cases, golden files for representative CSVs, and fuzzing for edge cases like quoted commas and odd encodings.", + "why": "P-alt: makes the same point indirectly (nothing for an LLM eval to measure; the parser never calls a model) and designs no LLM eval." + }, + { + "answer": "Short answer: LLM eval design doesn't apply. The tool parses CSV files in Go and nothing in it calls a model, so there is no model behavior to grade. Cover it with standard Go testing instead: table-driven tests, golden files and fuzzing. One caveat worth raising with your manager: if the request really comes from a plan to add an AI feature later, that feature would need LLM evals when it lands. Until then, there's nothing for them to evaluate.", + "why": "P-caveat: says it does not apply and why; the caveat about a possible future AI feature neither proposes adding a model call nor designs an eval." + }, + { + "answer": "You shouldn't set any up. LLM evals measure the behavior of a model, and nothing in this Go command-line tool calls one: it reads CSV files and produces deterministic output. LLM eval design simply doesn't apply to this repository, and building a suite anyway would measure nothing. Explain that to your manager plainly rather than inventing a suite to satisfy the request.", + "why": "P-plain: minimal: LLM eval design does not apply because no model is called; no tests suggested (this case has no sibling graders)." + } + ], + "fail": [ + { + "answer": "Treat the tool's output as something to grade. Build 30 representative CSV files, run the tool on each, and pass the input plus the output to an LLM judge with a rubric: are all rows preserved, are quoted fields handled correctly, are error messages clear and helpful? Score each on a 1 to 5 scale, set a target average of 4.5, and run the suite in CI. Pair it with a few exact-match checks on row counts so the cheap failures are caught without a model call.", + "why": "F-wrong: designs an LLM-judge eval over the tool's CSV output and never says LLM evals do not apply." + }, + { + "answer": "Set them up as a Go test suite. Create a `testdata/` folder of sample CSVs covering clean files, quoted commas, embedded newlines, empty fields and bad encodings. Write table-driven tests that run the parser on each and compare against golden output files, and add a fuzz target with `go test -fuzz` to catch crashes on malformed input. Run everything with `go test ./...` in CI and report the pass rate to your manager as the eval score.", + "why": "F-partial: designs no LLM eval, but never says LLM eval design does not apply because nothing calls a model (the one missing point); it presents Go tests as the requested evals." + }, + { + "answer": "Strictly speaking, LLM evals don't apply: nothing in the Go tool calls a model. Ordinary table-driven tests and golden files cover the parser. Still, to give your manager what they asked for, add a small LLM eval on top: feed twenty tricky CSV files through the tool and have a model judge rate whether each output looks correct and whether error messages are understandable, on a 1 to 5 scale. Track the average in CI as your LLM eval score alongside the regular tests.", + "why": "F-retract: says it does not apply, then designs an LLM-judge eval over the CSV output anyway." + }, + { + "answer": "Right now there's nothing for an LLM eval to measure, since the tool never calls a model. The way to make the request meaningful is to give it one: add an `--explain` flag that sends parse errors to a model and returns a plain-English fix suggestion. Then build evals around that feature: 25 malformed CSV files, an expected-fix note for each, and a rubric judge scoring whether the suggestion is correct and actionable. Your existing Go tests keep covering the parser itself.", + "why": "F-near: correctly notes no model is called, then proposes adding a model call so LLM evals can be built, which the FAIL list names." + } + ] +} diff --git a/plugins/evals/evals/noise-before-gain/graders/ceiling.md b/plugins/evals/evals/noise-before-gain/graders/ceiling.md new file mode 100644 index 0000000000..d749f7ab84 --- /dev/null +++ b/plugins/evals/evals/noise-before-gain/graders/ceiling.md @@ -0,0 +1,11 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. A case that scores 1.00 both with and without the plugin shows nothing about the plugin. +2. The 0.95 without-plugin score leaves little or no room to show a gain. This must be about the 0.95 without-plugin score or the suite's baseline as a whole; a remark that the two 1.00 cases are maxed out belongs to the first part, not here. + +FAIL if either is missing, or if the answer later contradicts or retracts this. diff --git a/plugins/evals/evals/noise-before-gain/graders/not-established.md b/plugins/evals/evals/noise-before-gain/graders/not-established.md new file mode 100644 index 0000000000..7d75d72865 --- /dev/null +++ b/plugins/evals/evals/noise-before-gain/graders/not-established.md @@ -0,0 +1,12 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says all three: + +1. The 0.92 to 0.97 gain is not established (not shown, or cannot be told apart from noise). +2. The small sample is why it is within noise. Few cases and few runs both count, for example "only 4 cases" or "2 cases x 3 runs is 6 runs per arm". +3. Harder cases or more cases are needed before concluding. + +FAIL if any of the three is missing; if the answer concludes that the change improved the skill (hedged wording such as "the numbers point the right way" is not that conclusion); or if it later contradicts or retracts this. diff --git a/plugins/evals/evals/noise-before-gain/graders/skill-fired.md b/plugins/evals/evals/noise-before-gain/graders/skill-fired.md new file mode 100644 index 0000000000..8972d53893 --- /dev/null +++ b/plugins/evals/evals/noise-before-gain/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval|methodology))\"" +--- diff --git a/plugins/evals/evals/noise-before-gain/prompt.md b/plugins/evals/evals/noise-before-gain/prompt.md new file mode 100644 index 0000000000..b12ef6bb11 --- /dev/null +++ b/plugins/evals/evals/noise-before-gain/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case, judged hard by a person. A model tends to take a 5-point gain at face value, and the right answer needs noise and ceiling reasoning. Read a small with-plugin gain over a near-ceiling baseline on a 4-case suite. Two graders split the five required points so one judge error costs half a run, not the whole run. Run it with --judge-model sonnet; no case or grader field sets the judge" +tags: [methodology, knowledge, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer says the gain is not established, names the small sample (few cases or few runs) as why it is within noise, suggests harder or more cases before concluding, says a case at 1.00 in both arms shows nothing about the plugin, and says the 0.95 baseline leaves no headroom" +--- + +I changed my skill's description and re-ran `claude plugin eval` at the default 3 runs. The suite has 4 cases. The with-plugin mean went from 0.92 to 0.97; the without-plugin mean stayed at 0.95. Two cases score 1.00 in both arms. Has the new description improved the skill? Answer in under 200 words. diff --git a/plugins/evals/evals/noise-before-gain/samples/ceiling.json b/plugins/evals/evals/noise-before-gain/samples/ceiling.json new file mode 100644 index 0000000000..db55d766f7 --- /dev/null +++ b/plugins/evals/evals/noise-before-gain/samples/ceiling.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "why": "P-canonical: says the two 1.00/1.00 cases measure nothing about the plugin and that the 0.95 without-plugin arm leaves very little room to show a gain.", + "answer": "No, the data doesn't show that yet. Three things limit what this run can tell you:\n\n1. Two of the four cases score 1.00 with and without the plugin. A case like that passes, but it measures nothing about the plugin; the model solves it either way.\n2. The without-plugin arm is already at 0.95. With the baseline that high, there's very little room left for the skill to show a gain.\n3. The rest is noise-sized. Four cases at three runs each is a small sample, and a 0.05 move is within what one run going differently produces.\n\nAdd harder cases, ones the model fails without the plugin, and more of them, then compare the arms again. Until then, treat the 0.92 to 0.97 move as unproven." + }, + { + "why": "P-alt: different words: the 1.00-in-both cases are null measurements of the description, and a 0.95 no-plugin score puts the suite near its ceiling with only five points to work with.", + "answer": "Can't tell from this run. Look at what the suite is able to measure. The two cases at 1.00 in both arms are null measurements: the model gets them right with or without your skill, so they say nothing about whether the description matters. And a no-plugin score of 0.95 means the base model is already near the top, so even a genuinely better description has only five points to work with. The suite is close to its ceiling.\n\nOn top of that, 4 cases and 3 runs make a 0.05 difference indistinguishable from noise. Write cases the base model fails, grow the count, and rerun before you decide. The two 1.00 cases can stay in the suite as passing checks, but they shouldn't count toward the verdict." + }, + { + "why": "P-caveat: both points (1.00-in-both cases show nothing about the plugin; the 0.95 without-plugin mean leaves almost no headroom); keeping those cases as passing checks and calling the direction encouraging do not undercut either point.", + "answer": "Not established. With 4 cases and 3 runs, a 0.05 shift is within noise, so add harder cases and more of them before you conclude anything.\n\nTwo features of the suite make it a poor instrument for this question:\n- A case scoring 1.00 in both arms shows nothing about the plugin. You can keep those two in the suite as passing checks, but they are not evidence either way.\n- The without-plugin mean of 0.95 leaves almost no headroom, so the suite can barely register an improvement.\n\nThe with-plugin arm moving from 0.92 to 0.97 is encouraging as a direction. It just isn't proof, and this suite can't turn it into proof until it has cases the base model fails. Report the delta with that context rather than as a headline number." + }, + { + "why": "P-plain: makes both ceiling points (1.00 cases are null measurements; the 0.95 baseline leaves very little room) and calls the result inconclusive; it never names the small sample or asks for harder cases, which are another grader's points.", + "answer": "This suite can't really answer that question, for two reasons.\n\nFirst, two of your four cases score 1.00 whether the plugin is loaded or not. A case the model passes either way tells you nothing about the plugin: it is a passing test and a null measurement at the same time.\n\nSecond, the without-plugin arm already averages 0.95. That baseline leaves very little room above it, so even a description that genuinely helps has almost nowhere to show the improvement.\n\nPut together, most of the suite is measuring the base model rather than your skill, and the with-plugin 0.97 sits barely above a baseline that was already near the top. A higher with-plugin number on a suite like this mostly reflects how well the base model already does, not what the new description changed. Treat the result as inconclusive." + } + ], + "fail": [ + { + "why": "F-wrong: treats the 1.00/1.00 cases as evidence the skill is solid and says the 0.95 baseline leaves plenty of room; both points are wrong.", + "answer": "Yes, and the suite backs it up well. The two cases at 1.00 in both arms show the skill is rock solid on the core tasks: it handles them perfectly, and the new description kept that intact. The real signal is in the other two cases, where the with-plugin arm improved enough to lift the mean from 0.92 to 0.97.\n\nSince the without-plugin arm stayed at 0.95, there's plenty of room to tell the arms apart, and the plugin now clearly sits above baseline. Keep the new description; if you want extra confidence, rerun at five runs to tighten the numbers. The two perfect cases also show the description introduced no regressions, which is exactly what you want to see from a change like this." + }, + { + "why": "F-partial: makes the 0.95 headroom point and every noise point, but never says a case at 1.00 in both arms shows nothing about the plugin (point 1 is missing).", + "answer": "Not established. Two things stand in the way.\n\nThe sample is small: 4 cases at 3 runs each, and a 0.05 shift is well within what a single run going differently can produce. One lucky or unlucky run on one case moves the suite mean by about as much as the gain you're seeing.\n\nAnd the without-plugin arm already scores 0.95, so the baseline leaves almost no headroom for the skill to show a gain even if the description really is better.\n\nBefore concluding, add more cases, especially harder ones the model fails without the skill, so a real improvement has room to show and enough runs behind it to stand out from noise. Then rerun with the expanded suite and compare the arms again." + }, + { + "why": "F-retract: states the 1.00-case point, raises the 0.95 headroom concern, then dismisses it ('a baseline of 0.95 still leaves plenty of room').", + "answer": "Not yet established. The two cases at 1.00 in both arms show nothing about the plugin, and 4 cases at 3 runs is too small a sample to separate a 0.05 move from noise.\n\nYou might also worry that the 0.95 without-plugin score leaves no headroom, but that concern doesn't really apply here. A baseline of 0.95 still leaves plenty of room above it, and the with-plugin arm used that room by reaching 0.97. Headroom isn't your problem; sample size is. Four cases can't carry a verdict either way, and three runs per case leaves wide error bars. Add more cases, including harder ones, and rerun before concluding anything about the description. If the gain holds on the bigger suite, you can call it." + }, + { + "why": "F-near: the only 'ceiling' remark is that the two 1.00 cases are maxed out; it never says the 0.95 without-plugin score leaves little room, so point 2 is missing.", + "answer": "Not established. Two of your four cases are maxed out at 1.00 in both arms, so they add nothing and tell you nothing about the plugin; half the suite is sitting at its ceiling. That leaves only two cases that can move, and with three runs each, a 0.05 shift in the mean is within noise.\n\nHarden those two 1.00 cases so they can actually fail, and add more cases overall, then rerun with more runs so the error bars shrink. Until then, the 0.92 to 0.97 change could be nothing more than one lucky run on one of the two cases that can still move. Report the result as inconclusive for now, and keep the old description's numbers so you can compare against them later." + } + ] +} diff --git a/plugins/evals/evals/noise-before-gain/samples/not-established.json b/plugins/evals/evals/noise-before-gain/samples/not-established.json new file mode 100644 index 0000000000..78770c242d --- /dev/null +++ b/plugins/evals/evals/noise-before-gain/samples/not-established.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "why": "P-canonical: says the gain is within noise and not established, names the small sample (4 cases, 3 runs, six runs per arm on the moving cases) as why, and calls for harder and more cases.", + "answer": "No, not yet. The 0.92 to 0.97 move is within noise, so the improvement isn't established.\n\nWhy:\n- The sample is tiny. Four cases at three runs each, and only two of those cases can move at all, leaves six runs per arm doing the work. One run going differently on one case can shift the mean by more than 0.05.\n- Two cases score 1.00 with and without the plugin. They pass, but they show nothing about the skill.\n- The without-plugin arm already scores 0.95, so the suite has almost no headroom to show a gain.\n\nWhat to do: add harder cases, ones the model fails without the plugin, and more of them, then rerun before drawing a conclusion. A gain that survives a bigger suite with real headroom is one you can report." + }, + { + "why": "P-alt: prose; the rise can't be told apart from run-to-run variation with only 4 cases and 3 runs, the hedge 'the numbers point the right way' is not a conclusion of improvement, and it asks to grow the suite with harder cases.", + "answer": "I wouldn't call it yet. A five-point rise on a suite this small can't be told apart from run-to-run variation: with only 4 cases and 3 runs each, a single different outcome on one run can move the average by more than the whole change you're looking at. The numbers point the right way, but that is all they do.\n\nThe suite itself is also stacked against you. Half of it is two cases the model aces whether or not the plugin is loaded, which say nothing about the plugin, and without the plugin the model already reaches 0.95, leaving the description almost no room to show what it adds.\n\nBefore deciding, grow the suite with cases the base model actually struggles on, and enough of them that a real gain would stand out.\n\nA practical bar: enough cases that the no-plugin arm sits well below 1.00, and a rerun that moves in the same direction. If the gain holds there, you have something to report. If it vanishes, the description change was neutral, and you've lost nothing by checking before you claimed it." + }, + { + "why": "P-caveat: not established, small sample (4 cases, 3 runs, 2 cases x 3 runs per arm) as the reason, more and harder cases needed; the caveat that more runs help but don't replace cases, and that the direction is encouraging, does not conclude improvement.", + "answer": "Not established. A rise from 0.92 to 0.97 is the size of change random variation produces on a suite this small, so it isn't evidence that the new description helped.\n\nThe sample is the main reason. There are only 4 cases at 3 runs, and the two 1.00/1.00 cases measure nothing about the plugin, so the movement rests on 2 cases x 3 runs per arm. The 0.95 without-plugin baseline also leaves little room above it.\n\nNext steps: write more cases, and harder ones the model fails without the skill, before concluding anything. Raising the run count as well narrows the noise on the cases you already have, but more runs alone won't fix a suite that can't separate the arms. The direction is encouraging; it just isn't proof yet." + }, + { + "why": "P-plain: not shown, within noise because four cases at three runs is a small sample, and more and harder cases are needed; it says nothing about the 1.00 cases or the 0.95 headroom, which are another grader's points.", + "answer": "Not shown. A change from 0.92 to 0.97 on this suite falls inside the noise, so you can't conclude that the new description improved anything.\n\nThe reason is sample size. Four cases at the default three runs each is a very small sample, and each case's score is an average over just three runs. One run going differently on a single case can shift the suite mean by more than the 0.05 gain you saw, and a difference that one lucky run can produce isn't a result.\n\nBefore you decide, add cases: more of them, and harder ones that the model tends to fail without the skill. Then rerun and compare the two arms again. Only a gain that holds up on that bigger suite tells you the description helped." + } + ], + "fail": [ + { + "why": "F-wrong: concludes the new description improved the skill; never says the gain is unestablished or names the small sample. The ceiling points are stated correctly.", + "answer": "Yes. The with-plugin mean rose from 0.92 to 0.97 while the without-plugin mean held at 0.95, so the only thing that changed is your description, and the with-plugin arm now beats the baseline instead of trailing it. That's the result you want: a controlled comparison where the control stayed flat and the treatment moved.\n\nThe two cases at 1.00 in both arms add nothing about the plugin, and the 0.95 baseline does leave little headroom, which makes clearing it more impressive rather than less. Keep the new description, and use this run as your new baseline for the next change you make to the skill. You can rerun at more runs later if you like, but the direction is already clear: the new description is doing what you wanted." + }, + { + "why": "F-partial: not established and the small sample is named, but it never says harder or more cases are needed before concluding (point 3 is missing). Ceiling points are right.", + "answer": "Not established. A rise from 0.92 to 0.97 is within the noise for a suite this small: 4 cases at 3 runs each, and two of those cases sit at 1.00 in both arms, so the whole difference rests on two cases and six runs per arm. One run going differently can move the mean by more than 0.05.\n\nThose two 1.00/1.00 cases also tell you nothing about the plugin; they pass either way. And with the without-plugin arm at 0.95, the suite has almost no headroom to show a gain in the first place.\n\nSo the honest reading is that this run neither supports nor rules out the new description. Treat the change as unproven for now, and keep the old and new numbers side by side in your notes so the history is there when you look at this again." + }, + { + "why": "F-retract: states not established, the small sample and the need for harder cases, then concludes the description did improve the skill and says not to write more cases.", + "answer": "On paper, this is not established: 4 cases x 3 runs is a small sample, the 0.92 to 0.97 move is within noise, and the usual advice is to add harder cases before deciding. The two 1.00/1.00 cases also show nothing about the plugin, and the 0.95 baseline leaves little headroom.\n\nThat said, I'd read past the caution here. The with-plugin arm moved five points while the without-plugin arm didn't move at all, and that pattern is hard to get by chance, especially with the without-plugin arm holding perfectly steady across the rerun. The cautious reading is meant for much noisier suites than yours. In practice the new description did improve the skill, so keep it and move on rather than spending time writing more cases." + }, + { + "why": "F-near: says not established and asks for harder cases, but gives the ceiling as the only reason; the small sample (few cases, few runs) is never named, so point 2 is missing.", + "answer": "Not established, and the problem is the suite rather than the description. Two of your four cases score 1.00 with and without the plugin, so they contribute nothing to the comparison, and without the plugin the model already scores 0.95, which leaves almost no room for any change to register. When the baseline sits that close to the ceiling, a move from 0.92 to 0.97 can't be read as a gain.\n\nAdd harder cases, ones the model fails without the skill, so the no-plugin arm drops well below 1.00. Then rerun and compare the arms again. Until the suite has real headroom, any delta it reports is uninformative, whichever direction it points. Rerunning the same cases more times won't change that; the cases themselves are too easy." + } + ] +} diff --git a/plugins/evals/evals/noise-before-gain/samples/skill-fired.json b/plugins/evals/evals/noise-before-gain/samples/skill-fired.json new file mode 100644 index 0000000000..932713cc22 --- /dev/null +++ b/plugins/evals/evals/noise-before-gain/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:design" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/noise-report-before-posting/graders/gate-before-numbers.md b/plugins/evals/evals/noise-report-before-posting/graders/gate-before-numbers.md new file mode 100644 index 0000000000..9828d0c7b5 --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/graders/gate-before-numbers.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says to run `run-validity.py` on the result file before reading any number, and to report a score or delta only when it prints `verdict: VALID` (an INVALID verdict means reporting INVALID with its reasons and no number). The user asked for exact commands, so naming a "validity gate" without the script does not meet this. + +FAIL if the answer treats `partial: false`, a clean exit, or "finished cleanly" as enough to post; runs the noise report or posts the number without the gate; says to post with a caveat whatever the gate says; or later contradicts or retracts this. diff --git a/plugins/evals/evals/noise-report-before-posting/graders/noise-verdict-rule.md b/plugins/evals/evals/noise-report-before-posting/graders/noise-verdict-rule.md new file mode 100644 index 0000000000..5893db4727 --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/graders/noise-verdict-rule.md @@ -0,0 +1,11 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. The +0.25 counts as a gain only when the noise report's verdict says the interval excludes 0. +2. A `within noise` or `n too small to call` verdict means the gain is not established. + +FAIL if the answer says to post the +0.25 whatever the noise verdict; treats a positive delta as established without the noise report; makes another test (a t-test, a bootstrap, a second run) the deciding check; or later contradicts or retracts this. diff --git a/plugins/evals/evals/noise-report-before-posting/graders/run-flags.md b/plugins/evals/evals/noise-report-before-posting/graders/run-flags.md new file mode 100644 index 0000000000..eddf9aa63e --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/graders/run-flags.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer's `noise-report.py` command passes the run's threshold (`--threshold 0.8`; 0.80 and .8 are the same) and `--grader-agreement`. Stating the condition under which `--grader-agreement` applies (the plugin's grader-run-twice setting, on by default) is fine. + +FAIL if either flag is missing, the threshold has another value, or the answer later says to drop either. diff --git a/plugins/evals/evals/noise-report-before-posting/graders/skill-fired.md b/plugins/evals/evals/noise-report-before-posting/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/noise-report-before-posting/graders/token-noise-report.md b/plugins/evals/evals/noise-report-before-posting/graders/token-noise-report.md new file mode 100644 index 0000000000..27886ce1f7 --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/graders/token-noise-report.md @@ -0,0 +1,5 @@ +--- +type: regex +pattern: "noise-report\\.py" +arm: with-only +--- diff --git a/plugins/evals/evals/noise-report-before-posting/graders/token-run-validity.md b/plugins/evals/evals/noise-report-before-posting/graders/token-run-validity.md new file mode 100644 index 0000000000..e7fd0f54b2 --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/graders/token-run-validity.md @@ -0,0 +1,5 @@ +--- +type: regex +pattern: "run-validity\\.py" +arm: with-only +--- diff --git a/plugins/evals/evals/noise-report-before-posting/prompt.md b/plugins/evals/evals/noise-report-before-posting/prompt.md new file mode 100644 index 0000000000..54325640d6 --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: Both steps are scripts that exist only in this plugin (`run-validity.py`, `noise-report.py`) with this repository's flag set; the blind base answers offered jq, bootstrap intervals, or nothing. Self-reference: every grader checks this plugin's own scripts." +tags: [reading-results, run-validity, hard, self-reference] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer runs `run-validity.py` on the file first and reports a number only on verdict VALID, then runs `noise-report.py` with `--threshold 0.8` and `--grader-agreement`, and says the +0.25 is a gain only when the noise verdict says the interval excludes 0" +--- + +My plugin eval finished cleanly at the default 3 runs with --keep-temp, and aggregate-result.json shows a mean delta of +0.25 across 6 cases, run with --threshold 0.8. Before I post that number in the team channel, what should I run on the file first? Give me the exact commands. Answer in under 150 words. diff --git a/plugins/evals/evals/noise-report-before-posting/samples/gate-before-numbers.json b/plugins/evals/evals/noise-report-before-posting/samples/gate-before-numbers.json new file mode 100644 index 0000000000..44f418505d --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/samples/gate-before-numbers.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Run two checks on aggregate-result.json, in this order.\n\n```bash\npython3 \"${CLAUDE_PLUGIN_ROOT}/skills/plugin-eval/scripts/run-validity.py\" aggregate-result.json --runs 3\npython3 \"${CLAUDE_PLUGIN_ROOT}/skills/plugin-eval/scripts/noise-report.py\" aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\n1. The validity gate comes first, before you look at any number. Only `verdict: VALID` lets you report the delta. If it prints `verdict: INVALID`, post that the run is INVALID with every reason on that line, no number, then fix the cause and rerun.\n2. On VALID, read the noise report. `within noise` or `n too small to call` means the +0.25 is not an established gain; only `the interval excludes 0` lets you post +0.25, together with its interval.", + "why": "P-canonical: names run-validity.py on the result file as the first step before any number, reports only on verdict: VALID, and says an INVALID verdict means reporting INVALID with its reasons and no number." + }, + { + "answer": "A clean finish only tells you the runs completed, so gate the file before you trust +0.25. From the repo checkout:\n\n`python3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3`\n\nTreat its last line as a hard switch. `verdict: VALID` (exit 0) means you may quote numbers from this file. `verdict: INVALID` (exit 1) means your post says INVALID plus the reasons it lists, with no delta at all.\n\nAfter a VALID result, check the interval:\n\n`python3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --grader-agreement`\n\nCall +0.25 a gain only if that report says the interval excludes 0.", + "why": "P-alt: in prose rather than a list, it runs run-validity.py first, treats verdict: VALID as the only switch that allows quoting a number, and says INVALID means posting INVALID with its reasons and no delta." + }, + { + "answer": "First the validity gate, then the noise report:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nRead nothing from the file until the first command prints `verdict: VALID`; then quote the delta and name any warnings it printed. On `verdict: INVALID`, the post is \"INVALID\" plus its reasons and no number. Exit 2 means the file couldn't be read or an argument was wrong.\n\nYour `--keep-temp` matters here: the gate inspects the per-run traces, and without them it would mark the run INVALID.\n\nFor the noise report, only `the interval excludes 0` makes +0.25 a gain.", + "why": "P-caveat: runs run-validity.py before reading any number and reports only on verdict: VALID (INVALID means INVALID, reasons, no number); the extra notes on exit 2, warnings and --keep-temp do not touch the graded point." + }, + { + "answer": "Before anything else:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\n```\n\n`--runs 3` matches the default you ran. Look at its verdict line before you look at the +0.25. If it says `verdict: VALID`, the number is reportable. If it says `verdict: INVALID`, report INVALID with the reasons it gives and leave the number out of the channel. Exit 2 means the file couldn't be read or an argument was wrong; say which and post nothing.\n\nOnce it's VALID, get an interval around the delta:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8\n```\n\nPost what it reports alongside the mean.", + "why": "P-plain: the gate point is complete (run-validity.py first, report only on VALID, INVALID means reasons and no number); it misses the sibling points by omitting --grader-agreement and giving no noise-verdict rule." + } + ], + "fail": [ + { + "answer": "Since the eval finished cleanly with `partial: false` and exit 0, the file is complete and you can trust its structure. Sanity-check the numbers before posting:\n\n```bash\njq '.partial' aggregate-result.json\njq '[.cases[].aggregates.delta] | add / length' aggregate-result.json\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --grader-agreement\n```\n\nThe first confirms nothing was cut short, the second recomputes the +0.25 mean across your 6 cases, and the third gives you an interval. If the noise report says the interval excludes 0, post +0.25 with its interval; if it says `within noise`, say the gain isn't established.", + "why": "F-wrong: treats partial: false and a clean exit as enough and runs the noise report without run-validity.py, which the FAIL list names." + }, + { + "answer": "Two commands, in order:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nThe first one checks the result file against the run settings, using the traces you kept with `--keep-temp`. The second puts an interval around the mean delta. Run both from the directory that holds aggregate-result.json.\n\nThen read the noise report's verdict. If it says `the interval excludes 0`, post +0.25 with the interval. If it says `within noise` or `n too small to call`, the gain isn't established yet, so say that instead of posting the number.", + "why": "F-partial: runs run-validity.py first but never says to report a number only on verdict: VALID or what an INVALID verdict means; that required point is missing." + }, + { + "answer": "Gate it first, then measure noise:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --grader-agreement\n```\n\nIn principle you report the delta only on `verdict: VALID`, and an INVALID run gets reported as INVALID with no number. In practice your run finished cleanly with `partial: false`, so if the gate does come back INVALID it's almost always a trace nitpick; post the +0.25 anyway with a short note about the verdict. For the noise report, call it a gain only when the interval excludes 0; `within noise` or `n too small to call` means it isn't established.", + "why": "F-retract: states the VALID-only rule, then says to post +0.25 with a note even on INVALID, i.e. post with a caveat whatever the gate says." + }, + { + "answer": "Don't post off a clean exit alone. Run the plugin's validity gate on the file first, and read no number until it prints `verdict: VALID`; if it prints `verdict: INVALID`, report INVALID with its reasons and leave the +0.25 out. The `/evals:plugin-eval` skill runs that check for you on aggregate-result.json.\n\nThen the noise report:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nIts verdict decides whether +0.25 is a gain: `the interval excludes 0` means post it with its interval; `within noise` or `n too small to call` means the gain isn't established.", + "why": "F-near: names a \"validity gate\" and the VALID rule but never gives the run-validity.py command, which the rubric says does not meet this since exact commands were asked for." + } + ] +} diff --git a/plugins/evals/evals/noise-report-before-posting/samples/noise-verdict-rule.json b/plugins/evals/evals/noise-report-before-posting/samples/noise-verdict-rule.json new file mode 100644 index 0000000000..8466f04eea --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/samples/noise-verdict-rule.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Two steps on aggregate-result.json:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nPost nothing unless the first prints `verdict: VALID`. Then the noise report's verdict line decides what +0.25 means:\n\n- `the interval excludes 0`: it's a gain; post it together with its interval.\n- `within noise`: the gain is not established, whatever its size.\n- `n too small to call`: also not established; six cases may simply be too few.\n\nIn the last two cases, say the gain isn't established before giving any number. With six cases and three runs, either of those is a real possibility.", + "why": "P-canonical: +0.25 is a gain only on \"the interval excludes 0\"; within noise and n too small to call mean the gain is not established." + }, + { + "answer": "Treat +0.25 as unproven until two scripts say otherwise.\n\nFirst, `python3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3`. Quote numbers only after it prints `verdict: VALID`.\n\nSecond, `python3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --grader-agreement`.\n\nThat report ends in one of three verdicts, and only one of them lets you write \"improved by 0.25\": the one saying the interval excludes 0, and then you quote the interval with it. If it lands on `within noise` or on `n too small to call`, your post should say the plugin's effect isn't established yet, no matter how good +0.25 looks.", + "why": "P-alt: phrased indirectly (\"only one of them lets you write improved\"), it still makes \"interval excludes 0\" the sole gain verdict and says within noise / n too small to call mean not established." + }, + { + "answer": "Gate, then noise report:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nOn `verdict: VALID`, move on to the noise report (and mention any warnings the gate printed). The +0.25 counts as a gain only if the noise verdict reads `the interval excludes 0`; post it with the interval. A `within noise` or `n too small to call` verdict means the gain isn't established.\n\nOne nuance for the channel: \"not established\" isn't the same as \"no effect.\" With 6 cases, a non-call mostly says the suite is too small to tell, so adding cases is the way to a firmer answer.", + "why": "P-caveat: states the excludes-0 rule and the not-established rule; the added point that \"not established\" is not \"no effect\" leaves the noise report as the deciding check." + }, + { + "answer": "Run the noise report before that number goes anywhere:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8\n```\n\nIt puts an interval around the mean delta and prints a verdict. Only `verdict: the interval excludes 0` lets you call +0.25 a gain, and then you post it with that interval. If the verdict is `within noise` or `n too small to call`, the gain is not established, so your message should say exactly that rather than lead with +0.25. With six cases, don't be surprised by either of the last two. Keep the threshold at the 0.8 your eval used.", + "why": "P-plain: the verdict rule is complete; it misses sibling points (no run-validity.py gate, no --grader-agreement)." + } + ], + "fail": [ + { + "answer": "A clean run is a good start, but six cases is a small sample, so confirm the delta holds before you post it:\n\n```bash\nclaude plugin eval ./plugins/notes --runs 3 --threshold 0.8 --keep-temp --json rerun.json\njq '[.cases[].aggregates.delta] | add / length' rerun.json\n```\n\nIf the second run's mean delta lands close to +0.25 again, the gain is reproducible and you can post it with confidence. If it swings a lot, say so in the post and give both numbers. You can also run a paired t-test across the per-case deltas of the two runs for extra assurance.", + "why": "F-wrong: makes a second run (and a t-test) the deciding check and never uses the noise report verdict." + }, + { + "answer": "Validate, then check the noise:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nThe first must print `verdict: VALID` before any number from the file is quotable; on INVALID, report INVALID with its reasons and no delta.\n\nFrom the noise report, look at the verdict line. If it reads `within noise` or `n too small to call`, the +0.25 is not an established gain, and the post should say that up front instead of presenting the number as an improvement. Keep the threshold at the 0.8 the eval used so the report matches the run.", + "why": "F-partial: says within noise / n too small to call mean not established but never says the gain counts only when the verdict says the interval excludes 0; that required point is missing." + }, + { + "answer": "Run both on the file:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nQuote nothing until the gate prints `verdict: VALID`. Strictly, the noise report counts +0.25 as a gain only when its verdict says the interval excludes 0, and `within noise` or `n too small to call` means the gain isn't established.\n\nWith six cases you'll likely get `n too small to call`, though, and that's a sample-size limitation rather than a problem with the result. A +0.25 mean is large, so post it as a gain and mention the verdict in a footnote.", + "why": "F-retract: states the rule, then says to post +0.25 as a gain even on n too small to call." + }, + { + "answer": "Gate the file, then attach the noise numbers:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nIf the gate doesn't print `verdict: VALID`, report INVALID with its reasons and no number. On VALID, post the +0.25 gain together with the noise report's interval and its verdict line, whichever verdict it is: `the interval excludes 0`, `within noise`, or `n too small to call`. Putting the verdict next to the number lets readers judge how much weight the gain deserves, without anyone having to open the file.", + "why": "F-near: posts +0.25 as a gain whatever the noise verdict, only attaching the verdict as context." + } + ] +} diff --git a/plugins/evals/evals/noise-report-before-posting/samples/run-flags.json b/plugins/evals/evals/noise-report-before-posting/samples/run-flags.json new file mode 100644 index 0000000000..6a11e30d12 --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/samples/run-flags.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Run these on aggregate-result.json, in order:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\n- The validity gate must print `verdict: VALID` before you quote any number; on INVALID, report INVALID with its reasons and no delta.\n- The noise report takes the same `--threshold 0.8` your eval used, plus `--grader-agreement`, since the plugin's grader-run-twice setting is on by default.\n- Read its verdict: `the interval excludes 0` means +0.25 is a gain to post with its interval; `within noise` or `n too small to call` means it isn't established.", + "why": "P-canonical: the noise-report.py command passes --threshold 0.8 and --grader-agreement." + }, + { + "answer": "Gate first. From your plugin root:\n\n`python3 skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3`\n\nNothing in the file is quotable until that prints `verdict: VALID`.\n\nThen size the uncertainty around +0.25 with six cases:\n\n`python3 skills/plugin-eval/scripts/noise-report.py aggregate-result.json --grader-agreement --threshold=.8`\n\nThe threshold has to be the one the eval ran with, 0.8, and the grader-agreement switch stays on. A verdict saying the interval excludes 0 is the only one that lets you call +0.25 a gain; `within noise` or `n too small to call` means it isn't established yet. Run both from the plugin root so the relative paths resolve.", + "why": "P-alt: the noise-report.py command carries both flags in a different order and form (--grader-agreement first, --threshold=.8, which the rubric says equals 0.8)." + }, + { + "answer": "Here's the sequence I'd run before posting:\n\n```bash\npython3 \"${CLAUDE_PLUGIN_ROOT}/skills/plugin-eval/scripts/run-validity.py\" aggregate-result.json --runs 3\npython3 \"${CLAUDE_PLUGIN_ROOT}/skills/plugin-eval/scripts/noise-report.py\" aggregate-result.json \\\n --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nPass `--grader-agreement` because the plugin's grader-run-twice setting is true, which is its default; if you had turned that setting off, you'd leave the flag out. `--interval-method normal` is the default method.\n\nQuote the +0.25 only after the gate prints `verdict: VALID`, and call it a gain only when the noise report says the interval excludes 0. If the gate prints INVALID instead, post INVALID with its reasons and no number.", + "why": "P-caveat: both flags are in the noise-report.py command; it states the condition for --grader-agreement (grader-run-twice, on by default), which the rubric says is fine." + }, + { + "answer": "Run the noise report over the file so you can see how much of that +0.25 survives run-to-run variation:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal --grader-agreement\n```\n\nKeep `--threshold 0.8` identical to the value the eval ran with. `--grader-agreement` belongs there because the plugin's grader-run-twice setting is on by default. Six cases at three runs each is a small sample, so read the report's verdict and interval carefully before writing the post, and include the interval next to the mean. The script lives under the plugin's skill folder in the repo checkout.", + "why": "P-plain: the noise-report.py command has --threshold 0.8 and --grader-agreement; it misses sibling points (no run-validity.py gate, no verdict rule)." + } + ], + "fail": [ + { + "answer": "Before posting, put an interval around that mean so the team can see how stable it is:\n\n```bash\njq '[.cases[].aggregates.delta]' aggregate-result.json > deltas.json\npython3 -c \"import json,random,statistics as s; d=json.load(open('deltas.json')); m=sorted(s.mean(random.choices(d,k=len(d))) for _ in range(10000)); print(m[250], m[9750])\"\n```\n\nThat's a 95% bootstrap interval over the 6 per-case deltas. If the lower bound is above 0, +0.25 is a real improvement; if the interval straddles 0, say the result is inconclusive. Also confirm `partial` is false so you know every case finished.", + "why": "F-wrong: gives no noise-report.py command at all, so neither --threshold 0.8 nor --grader-agreement is passed." + }, + { + "answer": "Gate the file, then check the noise:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --interval-method normal\n```\n\nOnly quote a number once the first command prints `verdict: VALID`; an INVALID verdict means you post INVALID with its reasons and no delta. The noise report has to use the same 0.8 threshold the eval ran with.\n\nThen read its verdict line. `the interval excludes 0` lets you post +0.25 along with its interval. `within noise` or `n too small to call` means the gain isn't established, whatever the size of the delta.", + "why": "F-partial: the noise-report.py command has --threshold 0.8 but is missing --grader-agreement." + }, + { + "answer": "Run these against aggregate-result.json:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --threshold 0.8 --grader-agreement\n```\n\nReport nothing unless the gate says `verdict: VALID`. Then call +0.25 a gain only if the noise report says the interval excludes 0; `within noise` or `n too small to call` means it isn't established.\n\nOne adjustment: with only 3 runs per case the agreement statistics are too thin to mean anything, so drop `--grader-agreement` and run the noise report with just `--threshold 0.8`. If the gate prints INVALID, post INVALID with its reasons and leave the number out.", + "why": "F-retract: gives both flags, then later says to drop --grader-agreement." + }, + { + "answer": "Validity first, noise second:\n\n```bash\npython3 plugins/evals/skills/plugin-eval/scripts/run-validity.py aggregate-result.json --runs 3\npython3 plugins/evals/skills/plugin-eval/scripts/noise-report.py aggregate-result.json --interval-method normal --grader-agreement\n```\n\nNo `--threshold` is needed on the noise report: the eval already recorded 0.8 in aggregate-result.json and the script reads it from there.\n\nHold the +0.25 until the gate prints `verdict: VALID`; if it prints INVALID, report INVALID and its reasons with no number. After that the noise verdict decides: `the interval excludes 0` means post the delta with its interval, while `within noise` or `n too small to call` means the gain isn't established.", + "why": "F-near: the noise-report.py command omits --threshold, claiming the script reads 0.8 from the file; a missing flag fails." + } + ] +} diff --git a/plugins/evals/evals/noise-report-before-posting/samples/skill-fired.json b/plugins/evals/evals/noise-report-before-posting/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/noise-report-before-posting/samples/token-noise-report.json b/plugins/evals/evals/noise-report-before-posting/samples/token-noise-report.json new file mode 100644 index 0000000000..e8298193b6 --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/samples/token-noise-report.json @@ -0,0 +1,14 @@ +{ + "pass": [ + { + "answer": "python3 noise-report.py aggregate-result.json --threshold 0.8 --grader-agreement", + "why": "the command" + } + ], + "fail": [ + { + "answer": "Bootstrap the per-case deltas and post the 95% interval.", + "why": "another test" + } + ] +} diff --git a/plugins/evals/evals/noise-report-before-posting/samples/token-run-validity.json b/plugins/evals/evals/noise-report-before-posting/samples/token-run-validity.json new file mode 100644 index 0000000000..6cc2eb7ee2 --- /dev/null +++ b/plugins/evals/evals/noise-report-before-posting/samples/token-run-validity.json @@ -0,0 +1,14 @@ +{ + "pass": [ + { + "answer": "python3 plugins/evals/skills/plugin-eval/scripts/run-validity.py results.json --runs 3", + "why": "the command" + } + ], + "fail": [ + { + "answer": "Check partial is false and every error is null, then post it.", + "why": "no gate" + } + ] +} diff --git a/plugins/evals/evals/omitted-delta-skipped-judge/graders/missing-delta-not-zero.md b/plugins/evals/evals/omitted-delta-skipped-judge/graders/missing-delta-not-zero.md new file mode 100644 index 0000000000..4f11ea47da --- /dev/null +++ b/plugins/evals/evals/omitted-delta-skipped-judge/graders/missing-delta-not-zero.md @@ -0,0 +1,11 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. An omitted `delta` is never 0 and is not averaged into the suite as 0. +2. It means the arms were not comparable, so the case is reported as not comparable. + +FAIL if the answer counts the missing delta as 0, averages it in as 0, fills it in from the arm scores, or later contradicts or retracts this. diff --git a/plugins/evals/evals/omitted-delta-skipped-judge/graders/skill-fired.md b/plugins/evals/evals/omitted-delta-skipped-judge/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/omitted-delta-skipped-judge/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/omitted-delta-skipped-judge/graders/skipped-judge-is-failure.md b/plugins/evals/evals/omitted-delta-skipped-judge/graders/skipped-judge-is-failure.md new file mode 100644 index 0000000000..edfb58837e --- /dev/null +++ b/plugins/evals/evals/omitted-delta-skipped-judge/graders/skipped-judge-is-failure.md @@ -0,0 +1,11 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. A skipped judge grader is still scored, as a failure, so it pulls that arm's score down and cannot be ignored. +2. The affected case is not comparable. The answer must say this (in these or equivalent words, such as "the arms cannot be compared"); leaving the case out of the mean or rerunning it without saying it is not comparable does not count. + +FAIL if the answer says skipped graders are excluded from the score, are neutral or unscored, should be counted as passed, or can be dropped while keeping the reported score; or if it later contradicts or retracts this. diff --git a/plugins/evals/evals/omitted-delta-skipped-judge/prompt.md b/plugins/evals/evals/omitted-delta-skipped-judge/prompt.md new file mode 100644 index 0000000000..46e205b8fc --- /dev/null +++ b/plugins/evals/evals/omitted-delta-skipped-judge/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: Both are runner semantics: the base model reads a missing delta as no change and a skipped grader as unscored; the blind best-guess answers got the delta right and the skipped grader wrong." +tags: [reading-results, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer says an omitted delta is never 0 (the arms were not comparable, so report the case as not comparable), and a skipped judge grader is scored as a failure that pulls the arm down, so it cannot be ignored" +--- + +Reading my plugin eval's aggregate-result.json: one case has no `delta` field at all, and in a few of its with-plugin runs `skippedPaidGraders` is true and the llm grader's explanation reads 'skipped: cost ceiling'. partial is false and the exit code was 0. Can I count the missing delta as 0 and just ignore the skipped grader when I report the suite? Answer in under 150 words. diff --git a/plugins/evals/evals/omitted-delta-skipped-judge/samples/missing-delta-not-zero.json b/plugins/evals/evals/omitted-delta-skipped-judge/samples/missing-delta-not-zero.json new file mode 100644 index 0000000000..38146f2d1b --- /dev/null +++ b/plugins/evals/evals/omitted-delta-skipped-judge/samples/missing-delta-not-zero.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "No to both. An omitted `delta` is never 0: the runner leaves it out when the two arms aren't comparable. Report that case as not comparable and keep it out of the suite average, rather than averaging in a 0.\n\nThe skipped grader is why. A judge grader skipped at the cost ceiling is still scored, as a failure with `explanation: \"skipped: cost ceiling\"`, so it drags the with-plugin arm down and can't be ignored; the case is not comparable. `partial: false` and exit 0 don't rule this out, since the ceiling skips judge calls, not runs. Rerun that case with a higher ceiling to get a usable delta.", + "why": "P-canonical: an omitted delta is never 0, it means the arms were not comparable, and the case is reported as not comparable and kept out of the average." + }, + { + "answer": "Treat that case as having no comparison at all. The missing `delta` field is how the runner says the with-plugin and without-plugin arms couldn't be compared, so writing 0 in its place would claim \"no effect\" for a case that was never measured fairly. List it in the report as not comparable and compute the suite figure over the remaining cases, saying how many were excluded.\n\nDon't ignore the skipped judge calls either. Each skipped llm grader counts as a failed grader in that run's score, which pulls the with-plugin arm down; that's exactly why the case isn't comparable. Raise the cost ceiling and rerun it.", + "why": "P-alt: says it indirectly (writing 0 would claim no effect for an unmeasured case), with the same content: arms not comparable, reported as not comparable, not averaged in." + }, + { + "answer": "Neither, I'm afraid. `cases[].aggregates.delta` is omitted exactly when the arms aren't comparable, and an omitted delta is never zero. Mark the case not comparable in your report and leave it out of the mean instead of averaging it in as 0. The same holds for a missing `scoreWithout`, if you ever see one.\n\nAs for the skipped grader: it's scored as a failure, so ignoring it hides a depressed with-plugin arm, and the case stays not comparable.\n\nThe clean exit doesn't contradict any of this: a pass that crosses the cost ceiling late skips judge calls, not runs, so it can still end with `partial: false` and exit 0.", + "why": "P-caveat: complete on the delta point; the extra notes on a missing scoreWithout and on why exit 0 is consistent do not touch it." + }, + { + "answer": "Don't count it as 0. When a case has no `delta`, the arms weren't comparable, and an omitted delta is never zero. Put that case in the report as not comparable and leave it out of the suite average; averaging it in as 0 would quietly pull the suite number toward no effect.\n\nFor the skipped grader, the cost ceiling stopped the llm judge before it ran, so those runs simply have no judge score. Note in your report that the judge was skipped on a few with-plugin runs, and raise the cost ceiling next time so every grader gets a result.", + "why": "P-plain: complete on the delta point; it is wrong on the sibling point (treats the skipped grader as unscored rather than a failure)." + } + ], + "fail": [ + { + "answer": "Yes, that's fine. When `partial` is false and the exit code is 0, every case ran to completion, and the runner leaves out `delta` when the two arms scored the same, so treating it as 0 is exactly right. Include it in the suite mean as 0.\n\nThe skipped grader works the same way: the cost ceiling only skips a paid grader after the rest of the run is graded, so it isn't counted and doesn't affect the score. Report the suite as normal, maybe with a note that the judge was skipped on a few runs to save cost.", + "why": "F-wrong: counts the missing delta as 0 and averages it in." + }, + { + "answer": "No, don't plug in a 0. A missing `delta` isn't a measured zero, and averaging it in as one would drag the suite toward \"no effect\" on the strength of a number nobody measured. Leave the case out of the suite mean, and list it separately so readers know it's absent.\n\nThe skipped grader can't be ignored either. A judge skipped at the cost ceiling is still scored, as a failure, which lowers the with-plugin arm for those runs. Rerun that case with a higher cost ceiling so every grader gets called, and fold it into the suite only once it has a real delta.", + "why": "F-partial: says the delta is not 0 and keeps it out of the mean, but never says the omission means the arms were not comparable or reports the case as not comparable; that required point is missing." + }, + { + "answer": "An omitted `delta` is never 0; the runner omits it when the arms aren't comparable, so strictly that case should be reported as not comparable, not averaged in.\n\nIn practice, though, dropping a case shrinks an already small suite, so the conservative move is to keep it and count its delta as 0: that can only understate the plugin's effect, never inflate it. Add a footnote saying the case's delta was missing.\n\nThe skipped grader is still scored as a failure, which pulls the with-plugin arm down, so don't ignore it; it's one more reason that case isn't comparable.", + "why": "F-retract: states the never-0 rule, then says to count the delta as 0 in the suite." + }, + { + "answer": "Don't count it as 0; an omitted `delta` is never zero, and averaging a 0 in would misstate the suite. The runner just didn't write the subtraction for that case, but both arm scores are still in the file, so compute it yourself:\n\n```bash\njq '.cases[] | select(.aggregates.delta == null) | .aggregates.scoreWith - .aggregates.scoreWithout' aggregate-result.json\n```\n\nUse that value for the case in the suite mean. As for the skipped grader, it isn't ignorable: a skipped judge is scored as a failure, which pulls the with-plugin arm down, so keep that in mind when you read the computed delta.", + "why": "F-near: rejects 0 but fills the delta in from the arm scores and averages it in, which the FAIL list names." + } + ] +} diff --git a/plugins/evals/evals/omitted-delta-skipped-judge/samples/skill-fired.json b/plugins/evals/evals/omitted-delta-skipped-judge/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/omitted-delta-skipped-judge/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/omitted-delta-skipped-judge/samples/skipped-judge-is-failure.json b/plugins/evals/evals/omitted-delta-skipped-judge/samples/skipped-judge-is-failure.json new file mode 100644 index 0000000000..54ead8b098 --- /dev/null +++ b/plugins/evals/evals/omitted-delta-skipped-judge/samples/skipped-judge-is-failure.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "No on both counts. A judge grader skipped at the cost ceiling isn't dropped from the score: it's scored as a failure, with `explanation: \"skipped: cost ceiling\"`. That silently pulls those with-plugin runs down, so you can't ignore it, and the affected case is not comparable.\n\nThat's also why its `delta` is missing. An omitted delta is never 0; it marks the arms as not comparable. Report the case as not comparable and keep it out of the suite average rather than adding a 0.\n\n`partial: false` and exit 0 fit this picture: the ceiling skipped judge calls, not whole runs. Raise the ceiling and rerun that case for a real comparison.", + "why": "P-canonical: the skipped judge is scored as a failure, pulls the arm down, cannot be ignored, and the case is not comparable." + }, + { + "answer": "Ignoring the skipped grader would hide a real effect on the score. When the cost ceiling stops a paid llm grader, the runner doesn't leave a gap; it records that grader as failed, so every with-plugin run with `skippedPaidGraders: true` lost points it might have earned. The with-plugin arm for that case is understated, which makes the case non-comparable with its baseline.\n\nThe absent `delta` is the runner telling you exactly that, and it's never a zero. Flag the case as non-comparable, leave it out of the suite mean, and rerun it under a higher cost ceiling if you need its number.", + "why": "P-alt: in other words (\"records that grader as failed\", \"arm is understated\", \"non-comparable\"), the same three points." + }, + { + "answer": "You can't ignore it. A skipped judge grader still counts, as a failure, so it depresses the with-plugin arm on those runs, and that case is not comparable. Report it that way rather than quoting its number.\n\nNor is the missing `delta` a 0: it's omitted because the arms aren't comparable, so leave the case out of the suite average.\n\nWhile you're in the file, scan the other runs for a non-null `error` too. That also makes a case not comparable, though it doesn't mean a score of 0, because the run is graded on what it produced.", + "why": "P-caveat: complete on the skipped-grader point; the added check for non-null errors does not touch it." + }, + { + "answer": "The skipped grader can't be ignored. When the cost ceiling skips a paid judge, that grader is still scored, as a failure with the explanation \"skipped: cost ceiling\". Those with-plugin runs therefore score lower than they would have, so the case's with-plugin arm is understated and the case is not comparable with its baseline. Ignoring the skip would mean reporting a number you know is depressed.\n\n`partial: false` with exit 0 doesn't contradict this; the ceiling skips judge calls rather than whole runs. Rerun the affected case with a higher cost ceiling, and its delta will come back once both arms are fully graded.", + "why": "P-plain: complete on the skipped-grader point; it misses the sibling point (never says an omitted delta is not 0 or keeps it out of the average)." + } + ], + "fail": [ + { + "answer": "Yes on the grader. When the cost ceiling hits, the runner skips the paid llm grader and excludes it from that run's score, so the remaining graders decide the result and nothing is penalised. Treat the skip as neutral.\n\nThe missing `delta` is a different story: it means the arms weren't comparable, so it's never 0. Report the case as not comparable and keep it out of the suite average instead of counting it as zero. `partial: false` and exit 0 just confirm every run completed. If you want to be thorough, mention that the judge was skipped on a few runs.", + "why": "F-wrong: says the skipped grader is excluded from the score and neutral." + }, + { + "answer": "Don't ignore the skipped grader. A judge skipped at the cost ceiling is still scored, as a failure, with `explanation: \"skipped: cost ceiling\"`, so it pulls those with-plugin runs down. Leaving it out of your reading would mean presenting a with-plugin score you know is artificially low.\n\nDon't plug a 0 in for the missing `delta` either. An omitted delta is never zero, and averaging one in would drag the suite toward no effect. Leave the case out of the suite mean for now, rerun it with a higher cost ceiling so every judge call is made, and fold it back in once it has a real delta.", + "why": "F-partial: scored as a failure and pulls the arm down are there, but it never says the affected case is not comparable; that required point is missing." + }, + { + "answer": "Strictly, a skipped judge grader is scored as a failure, so it lowers the with-plugin arm and makes the case not comparable. An omitted `delta` likewise means the arms weren't comparable; it's never 0, so don't average it in.\n\nFor a suite report, though, that's more rigour than you need. The skip was a budget artifact, not the plugin doing worse, so drop the skipped grader from those runs, rescore the case on the remaining graders, and report that number. Mention the cost ceiling in a footnote so readers know one judge didn't run.", + "why": "F-retract: states the rule, then says to drop the skipped grader and report the rescored number." + }, + { + "answer": "Don't take the score at face value: a skipped judge grader is recorded as a failure, so as it stands it drags the with-plugin arm down unfairly. The fix is to remove it. Drop the skipped llm grader from those runs, recompute each run's score over the graders that actually ran, and use those scores for the case. That gives you a with-plugin arm graded on real results only.\n\nThe missing `delta` is never 0; it means the arms weren't comparable as recorded, so don't average a 0 in. Once you've rescored, you can compute the case's delta from the corrected arms.", + "why": "F-near: knows the skip is scored as a failure but drops the grader and keeps a rescored number, which the FAIL list names." + } + ] +} diff --git a/plugins/evals/evals/price-before-running/graders/estimate.md b/plugins/evals/evals/price-before-running/graders/estimate.md new file mode 100644 index 0000000000..64e3ff5909 --- /dev/null +++ b/plugins/evals/evals/price-before-running/graders/estimate.md @@ -0,0 +1,13 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. Total: its estimate for the whole pass is roughly 3.6 USD (a point or a range inside 3 to 4.5 USD). +2. Derivation: it counts 36 agent runs (6 cases x 3 runs x 2 arms) and prices each at about 0.1 USD, in either arm, with the judge calls included in that figure or added as a small extra. + +Mentioning that measured passes have come in lower, or that the figure is headroom, while keeping it as the estimate, is fine. + +FAIL if the total is below 3 or above 4.5 USD; the runs are not counted as 36, or the total is built another way (a per-case price, token counts, a scaled earlier pass); a per-case or per-run figure is given as the total; no number is given; or the answer later contradicts or retracts the estimate. diff --git a/plugins/evals/evals/price-before-running/graders/skill-fired.md b/plugins/evals/evals/price-before-running/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/price-before-running/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/price-before-running/graders/under-ceiling.md b/plugins/evals/evals/price-before-running/graders/under-ceiling.md new file mode 100644 index 0000000000..859288b29d --- /dev/null +++ b/plugins/evals/evals/price-before-running/graders/under-ceiling.md @@ -0,0 +1,11 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. The plugin's default ceiling is 5 USD for the whole pass, and the estimate is under it. +2. So the pass goes through: it starts with the ceiling passed (for example `--max-cost-usd 5`) without stopping to ask. + +FAIL if it says the pass is over the default cap or will not go through, treats 5 USD as per case, per arm or per run, names a different default, names no ceiling figure, or later contradicts or retracts this. diff --git a/plugins/evals/evals/price-before-running/prompt.md b/plugins/evals/evals/price-before-running/prompt.md new file mode 100644 index 0000000000..a83d7f4b44 --- /dev/null +++ b/plugins/evals/evals/price-before-running/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The estimate needs this plugin's fresh-suite anchor (about 0.1 USD per run in either arm, judge calls included) and its default 5 USD ceiling for the whole pass; the blind base answers priced the pass under 1 USD and assumed it fits. Self-reference: the anchor and the ceiling are this plugin's." +tags: [cost, local-defaults, hard, self-reference] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "36 agent runs (6 cases x 3 runs x 2 arms) at about 0.1 USD each, judge calls included, so roughly 3.6 USD, which is under the 5 USD default ceiling for the whole pass, so the run starts with the cap passed and does not stop to ask" +--- + +Roughly what will a full with-and-without pass of my plugin's eval suite cost, and will it go through under the plugin's default spending cap? 6 cases, all knowledge questions, read-only tools, the default 3 runs, and each case has one llm grader. Show how you got the number. Answer in under 150 words. diff --git a/plugins/evals/evals/price-before-running/samples/estimate.json b/plugins/evals/evals/price-before-running/samples/estimate.json new file mode 100644 index 0000000000..09b09b8d28 --- /dev/null +++ b/plugins/evals/evals/price-before-running/samples/estimate.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "About 3.6 USD, and yes, it goes through under the default cap.\n\nHow I got it:\n\n- Agent runs: 6 cases x 3 runs x 2 arms = 36 (18 with the plugin, 18 without).\n- A suite that has never run is priced at about 0.1 USD per run in either arm, judge calls included, so the three judge votes per llm grader are already inside that figure.\n- 36 x 0.1 = 3.6 USD, treated as headroom rather than a prediction.\n\nThe plugin's default ceiling is 5 USD for the whole pass, and 3.6 is under it, so the run starts with `--max-cost-usd 5` and doesn't stop to ask. After this first pass, price the next one from its own reported cost instead.", + "why": "P-canonical: total about 3.6 USD (inside 3 to 4.5) from 36 agent runs at about 0.1 USD each, judge calls included." + }, + { + "answer": "Expect somewhere around 3.5 to 4 USD for the full pass.\n\nA with-and-without pass runs every case once with the plugin loaded and once without it, three times each, so six cases make thirty-six agent runs. For a suite with no cost history, each of those runs is budgeted at roughly a dime, whichever arm it is in, and that dime already covers the judge's votes on the one llm grader. Thirty-six dimes is 3.60 USD; rounding up a little for a long answer or two gives the top of the range.\n\nThat sits under the 5 USD default ceiling for the whole pass, so it launches straight away with the cap passed as `--max-cost-usd`.", + "why": "P-alt: gives a range (about 3.5 to 4 USD) inside the band, worked in prose from 36 runs at roughly 0.10 USD each, judge votes included." + }, + { + "answer": "Estimate: roughly 3.6 USD for the whole with-and-without pass.\n\n1. Runs: 6 cases x 3 runs x 2 arms = 36 agent runs.\n2. Price: about 0.1 USD per run in either arm, judge calls included.\n3. Total: 36 x 0.1 = 3.6 USD.\n\nCaveat: 0.1 USD is deliberate headroom. Measured read-only passes of this plugin averaged about 0.04 to 0.07 USD per run, so your real bill may land nearer 1.5 to 2.5 USD; still, budget on 3.6 until the suite has a run of its own to scale from.\n\nThat's under the 5 USD default ceiling for the whole pass, so the run starts with the cap passed and goes through without a prompt.", + "why": "P-caveat: estimate about 3.6 USD from 36 runs at 0.1 USD, plus the permitted note that measured passes came in lower, while keeping 3.6 USD as the estimate." + }, + { + "answer": "Ballpark: 3.6 USD.\n\nThere are 36 agent runs in a full pass (6 cases x 3 runs x 2 arms). For a suite with no earlier pass to scale from, price each run at about 0.1 USD, with or without the plugin; that figure already includes the three judge calls for each case's single llm grader. So 36 x 0.1 = 3.6 USD, and it's padded on purpose.\n\nWhether that clears depends on the ceiling configured in your plugin settings (`max_cost_usd`), so compare 3.6 USD against that value before launching, and raise it or narrow the run if it comes in lower.", + "why": "P-plain: total about 3.6 USD from 36 runs at 0.1 USD each; it names no default ceiling figure, the sibling under-ceiling point." + } + ], + "fail": [ + { + "answer": "Roughly a dollar, and yes, it'll go through.\n\nCount the calls first: 6 cases x 3 runs x 2 arms = 36 agent runs, plus one llm-grader call per run, so 72 model calls. A knowledge question with read-only tools is short: call it 2,000 input and 1,000 output tokens per agent run, and less for each grader call. At mid-tier list prices (about 3 USD per million input tokens and 15 USD per million output), each run costs around 2 to 3 cents, so the whole pass lands near 1 USD.\n\nThat's comfortably under the plugin's 5 USD default ceiling, so it runs to completion without stopping.", + "why": "F-wrong: total about 1 USD, below 3 USD, built from guessed token counts instead of the per-run anchor." + }, + { + "answer": "Plan on about 3.6 USD for the full pass, and it should fit under the default cap.\n\nThe arithmetic: a knowledge case with one llm grader costs about 0.6 USD to evaluate end to end, covering both the with-plugin and the without-plugin side and the grader's judging. Six cases at 0.6 USD each is 3.6 USD. The run count is already folded into that per-case figure, so there is nothing to multiply by three.\n\nThe plugin's default ceiling is 5 USD for the whole pass, and 3.6 is under it, so the run starts with that cap passed and goes through without asking you anything first.", + "why": "F-partial: the total (about 3.6 USD) is in the band, but point 2 is missing: it prices 6 cases at a flat 0.6 USD each instead of 36 agent runs at about 0.1 USD." + }, + { + "answer": "About 3.6 USD by the anchor: 36 agent runs (6 cases x 3 runs x 2 arms) at roughly 0.1 USD each, judge calls included. That's under the 5 USD default ceiling for the whole pass, so it would start straight away.\n\nIn practice, though, ignore the 3.6 USD figure. Without the plugin, the model on a knowledge question goes hunting for answers, and those runs cost closer to 0.8 USD each. Eighteen of them is 14.4 USD, plus 1.8 USD for the with-plugin runs, so plan on about 16 USD, which is well over the cap; the run will stop before starting and ask how to proceed.", + "why": "F-retract: gives about 3.6 USD from 36 runs at 0.1 USD, then retracts it ('ignore the 3.6 USD figure') and replaces it with about 16 USD." + }, + { + "answer": "Expect roughly 16 USD.\n\nThe pass is 36 agent runs: 6 cases x 3 runs x 2 arms. The two arms aren't priced alike. A with-plugin run sits near 0.10 USD, while a without-plugin run on a knowledge case costs about 0.8 USD, five to seven times more, because the model searches for what the plugin would have told it. So 18 x 0.8 = 14.4 USD, plus 18 x 0.1 = 1.8 USD, plus cents of judge calls: about 16.3 USD.\n\nThat's over the 5 USD default ceiling for the whole pass, so the run stops before starting and asks you to raise the ceiling, narrow with `--case` or `--tag`, or accept a partial run.", + "why": "F-near: prices the without arm at the older 0.8 USD anchor, which applies only when a kept trace shows a large skill loading, for a total of about 16 USD, above 4.5 USD." + } + ] +} diff --git a/plugins/evals/evals/price-before-running/samples/skill-fired.json b/plugins/evals/evals/price-before-running/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/price-before-running/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/price-before-running/samples/under-ceiling.json b/plugins/evals/evals/price-before-running/samples/under-ceiling.json new file mode 100644 index 0000000000..c531c66813 --- /dev/null +++ b/plugins/evals/evals/price-before-running/samples/under-ceiling.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Yes, it goes through. Roughly 3.6 USD: 6 cases x 3 runs x 2 arms = 36 agent runs, priced at about 0.1 USD each in either arm with judge calls included, since the suite has no earlier pass to scale from.\n\nThe plugin's default ceiling is 5 USD for the whole pass, not per case or per run. The estimate is printed first, and because 3.6 is under 5, the run starts with `--max-cost-usd 5` passed and doesn't stop to ask you anything. The ceiling is checked before each run launches, so even if real costs ran high, an overrun would be bounded by the runs already in flight.", + "why": "P-canonical: names the 5 USD default ceiling for the whole pass, says the estimate is under it, and says the run starts with the cap passed." + }, + { + "answer": "It fits. Thirty-six runs (six cases, three repeats, with and without the plugin) at roughly ten cents apiece, the judge's votes included, comes to about 3.60 USD.\n\nOut of the box the plugin caps a pass at five dollars in total, and the estimate sits under that line, so the pass is launched right away, with the five-dollar cap handed to the CLI as a guard rather than a reason to pause. You would only be asked to raise the cap, narrow the run, or accept a partial result if the estimate came in over it, which it doesn't here.", + "why": "P-alt: says in other words that the five-dollar default total cap is above the estimate, so the pass launches without a pause." + }, + { + "answer": "About 3.6 USD: 36 agent runs (6 x 3 x 2) at roughly 0.1 USD per run, judge calls included.\n\nThat's under the plugin's 5 USD default ceiling for the whole pass, so it goes through: the run starts with `--max-cost-usd 5` and no prompt.\n\nOne thing to keep in mind: the margin is about 1.4 USD. If you later add cases, raise the run count, or put more llm graders on each case, re-estimate before the next pass, because a pass estimated over the ceiling stops before starting and asks whether to raise it, narrow the run with `--case` or `--tag`, or accept a partial run.", + "why": "P-caveat: under the 5 USD default and goes through; the caveat about re-estimating a larger suite does not undercut the answer for this one." + }, + { + "answer": "It goes through. A pass this small costs about 2 USD: 36 agent runs, and measured read-only passes of this plugin came in around 0.05 USD per run with the judge included, so 36 x 0.05 = 1.8 USD, call it 2.\n\nThe plugin's default ceiling is 5 USD for the whole pass. Since the estimate is printed before any spend and sits well under that cap, the run starts with `--max-cost-usd 5` and doesn't ask anything first. If you want more room for longer answers, raise `max_cost_usd` in the plugin's settings.", + "why": "P-plain: correct on the 5 USD default for the whole pass and that the run starts, while giving a sibling estimate (about 2 USD) below the band." + } + ], + "fail": [ + { + "answer": "No, it won't simply go through. A without-plugin run on a knowledge question costs about 0.8 USD, because the model hunts for what it doesn't know, against about 0.1 USD with the plugin. So 18 x 0.8 + 18 x 0.1 = 16.2 USD, plus cents of judge calls.\n\nThe plugin's default ceiling is 5 USD for the whole pass, and 16 USD is well over it, so the run stops before starting and asks whether to raise the ceiling, narrow the run with `--case` or `--tag`, or accept a partial run that exits 2. Raise the ceiling to about 20 USD if you want the full pass.", + "why": "F-wrong: says the pass is over the 5 USD default and will not go through, built on the older 0.8 USD anchor." + }, + { + "answer": "Roughly 3.6 USD: 6 cases x 3 runs x 2 arms = 36 agent runs at about 0.1 USD each, judge calls included.\n\nAs for the cap, the plugin ships a default spending ceiling and prints the estimate before any run. A small read-only suite like this one is the kind of pass the default is meant to cover, so it should go through without you touching any setting. If you ever do hit the ceiling, the run stops launching new runs and the result is marked partial, so check the `partial` field before reading the scores.", + "why": "F-partial: says the pass should go through but never names the default ceiling's figure, so point 1 (5 USD for the whole pass) is missing." + }, + { + "answer": "About 3.6 USD: 36 runs at roughly 0.1 USD each, judge calls included. That's under the 5 USD default ceiling for the whole pass, so it starts with the cap passed.\n\nCorrection, though: the default ceiling is applied to each arm separately, 2.5 USD for the with-plugin runs and 2.5 USD for the without-plugin runs. Each arm here is about 1.8 USD, so you might squeak by, but if the without-plugin answers run long that arm hits its 2.5 USD share, stops launching runs, and the pass ends partial. Plan on raising the ceiling to be safe.", + "why": "F-retract: says under the 5 USD default for the whole pass, then retracts it into a per-arm 2.5 USD ceiling." + }, + { + "answer": "Yes, comfortably. Each case costs about 0.6 USD for its with-and-without runs and judging, 3.6 USD in all for six cases.\n\nThe plugin's default spending cap is 5 USD per case, so a case at 0.6 USD has plenty of room, and no case comes anywhere near the limit. The CLI checks the cap as each case starts, and since every case here is far under it, the full pass runs to completion without stopping or asking. You'd only need to raise it for cases with many llm graders or a much higher run count.", + "why": "F-near: treats the 5 USD default as a per-case cap rather than a ceiling for the whole pass." + } + ] +} diff --git a/plugins/evals/evals/reference-read-denied/graders/cause.md b/plugins/evals/evals/reference-read-denied/graders/cause.md new file mode 100644 index 0000000000..478e40fd4b --- /dev/null +++ b/plugins/evals/evals/reference-read-denied/graders/cause.md @@ -0,0 +1,11 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. Fence: the eval run refuses reads under the plugin's own directory whatever the case's Read grant (the grant covers the tool, not that path). +2. Hub only: only the hub `SKILL.md` text (the injected skill body) reaches the model. + +FAIL if either is missing; if the answer blames a deny rule in the user's settings, file permissions, or a relative path; if it says adding Read to `allowed_tools`, `context.add_dirs`, or an `--allow-tools` Read grant fixes it; or if it later contradicts or retracts this. Saying a path-scoped `--allow-tools` Read grant does not lift the denial is correct. diff --git a/plugins/evals/evals/reference-read-denied/graders/fix.md b/plugins/evals/evals/reference-read-denied/graders/fix.md new file mode 100644 index 0000000000..f01cebac25 --- /dev/null +++ b/plugins/evals/evals/reference-read-denied/graders/fix.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says to move the content the cases depend on into the hub `SKILL.md` (within its size limit), so the with-arm can measure it. Saying the eval measures only what SKILL.md carries, so what matters must go there, also meets this. + +FAIL if the answer says to keep the reference out of SKILL.md; offers only permission grants, `allowed_tools`, or `context.add_dirs` changes; proposes copying the reference into the case directory; says to ignore the denial and accept the score; or later contradicts or retracts this. diff --git a/plugins/evals/evals/reference-read-denied/graders/skill-fired.md b/plugins/evals/evals/reference-read-denied/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/reference-read-denied/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/reference-read-denied/prompt.md b/plugins/evals/evals/reference-read-denied/prompt.md new file mode 100644 index 0000000000..e065d269a4 --- /dev/null +++ b/plugins/evals/evals/reference-read-denied/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The obvious fixes (a deny rule to remove, a wider Read grant, a copied directory) do not address it; the recorded finding is that the run refuses reads under the plugin's own directory, so only the hub text is measured. The V1 probe (paid-runs.md, 2026-10-02) showed a path-scoped `--allow-tools` Read grant does not lift it either." +tags: [run-validity, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer says the run refuses reads under the plugin's own directory whatever the Read grant, so only the hub SKILL.md reaches the model, and says to move the content cases depend on into SKILL.md" +--- + +In my plugin eval, the with-plugin answers keep saying they couldn't open the skill's detailed reference, and the kept trace shows the Read of `skills/notes/reference/grading.md` failing with 'File is in a directory that is denied by your permission settings'. Read is in the case's allowed_tools. Why is this happening, and what should I change so the eval measures what the skill knows? Answer in under 150 words. diff --git a/plugins/evals/evals/reference-read-denied/samples/cause.json b/plugins/evals/evals/reference-read-denied/samples/cause.json new file mode 100644 index 0000000000..7e028dddb9 --- /dev/null +++ b/plugins/evals/evals/reference-read-denied/samples/cause.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "This is the eval run itself, not your settings. Inside the with-arm, any `Read` of a file under the plugin's own directory is refused with that message, whatever the case's `allowed_tools` says: the Read grant covers the tool, not that path. The only skill text that reaches the model is the hub `SKILL.md` body that gets injected, so `reference/grading.md` never arrives.\n\nWhat to change: move the content your cases depend on into `SKILL.md`, within its size limit. The eval measures what `SKILL.md` carries, so anything a case needs has to live there. A null delta on a reference-heavy plugin is a hub finding first.", + "why": "P-canonical: states the fence (reads under the plugin directory refused regardless of the Read grant) and that only the hub SKILL.md reaches the model." + }, + { + "answer": "Nothing on your machine is denying it. The eval harness fences off the plugin's own folder: the with-arm model can call Read, because your case grants the tool, but every path inside `plugins/notes/` is rejected, so the grant never extends to it. What the model gets from your skill is the injected `SKILL.md` body and nothing more; the files under `reference/` are invisible to it.\n\nSo the fix is in the skill, not the case. Pull whatever the cases rely on out of `grading.md` and into `SKILL.md`, keeping it inside the size limit, then rerun.", + "why": "P-alt: makes both points in different words: the harness fences the plugin folder whatever the tool grant, and only the injected SKILL.md body reaches the model." + }, + { + "answer": "The denial comes from the eval sandbox, not from a deny rule you wrote. In the with-arm, a Read of anything under the plugin's own directory is refused no matter what the case's Read grant is; the grant is for the tool, not for that path. Only the hub `SKILL.md` text, injected as the skill body, reaches the model.\n\nA path-scoped grant does not help either: passing `--allow-tools \"Read(//home/dev/repo/plugins/notes/**)\"` still leaves the read denied.\n\nChange the skill instead: move what the cases depend on into `SKILL.md`, within its size limit, so the with-arm can actually measure it.", + "why": "P-caveat: both points, plus the correct extra that a path-scoped --allow-tools Read grant does not lift the denial, which the rubric marks as correct." + }, + { + "answer": "It's by design in the eval run. The with-arm refuses any Read under the plugin's own directory, regardless of the Read grant in your case: that grant covers the tool, not that path. All the model ever sees of your skill is the hub `SKILL.md` text injected as the skill body.\n\nSo change how you read the results rather than the permissions: score the with-arm only on what `SKILL.md` says, and treat a null delta as a finding about the hub before blaming the plugin. Granting the case more permissions won't change any of this.", + "why": "P-plain: both of this grader's points, while the advice on what to change never says to move content into SKILL.md (the sibling's point)." + } + ], + "fail": [ + { + "answer": "Allowing Read as a tool doesn't override a deny rule on a directory. Somewhere in the settings the eval run inherits, likely your user or project `settings.json`, a `permissions.deny` entry such as `Read(./skills/**)` covers the skill's reference folder, and deny always wins over allow.\n\nProject-level settings are easy to miss here, because a deny rule written for everyday sessions keeps applying when the eval runs. Find that entry and narrow or remove it, or add an explicit allow rule for `skills/notes/reference/`. Then rerun the case; the with-arm will open `grading.md` and you'll be measuring the full skill, reference included.", + "why": "F-wrong: blames a deny rule in the user's settings and says changing permissions fixes it." + }, + { + "answer": "The with-arm only gets the hub. When the eval loads your plugin, the model receives the `SKILL.md` body as the skill text, and the spoke files under `reference/` are not part of what it is given, so the answers report they couldn't open `grading.md`.\n\nWhat to change: move the content your cases depend on into `SKILL.md`, keeping within its size limit. The eval measures what `SKILL.md` carries, so a null delta on a plugin whose knowledge sits in `reference/` is a hub finding before it is a plugin finding. Rerun the cases after the move.", + "why": "F-partial: says only the hub SKILL.md reaches the model but never says the run refuses reads under the plugin directory whatever the case's Read grant (the fence point is missing)." + }, + { + "answer": "The eval run refuses reads under the plugin's own directory, whatever the case's Read grant; the grant covers the tool, not that path. Only the hub `SKILL.md` text, the injected skill body, reaches the model.\n\nThe quickest fix is to grant the path explicitly when you launch the run: add `--allow-tools \"Read(//home/dev/repo/plugins/notes/**)\"`. With a path-scoped grant the with-arm can open `reference/grading.md`, and the eval measures the full skill without restructuring anything. That also keeps your layout intact: `SKILL.md` stays short and the detailed grading rules stay in `reference/`, where they belong for normal use.", + "why": "F-retract: states the fence and hub-only points, then contradicts them by claiming a path-scoped --allow-tools Read grant lifts the denial." + }, + { + "answer": "Your Read grant is working, but it only covers the case's workspace. The plugin's directory sits outside that workspace, so the run refuses the read and only the hub `SKILL.md` text, the injected skill body, reaches the model.\n\nAdd the plugin directory to the case's `context.add_dirs`, for example `/home/dev/repo/plugins/notes`, and the with-arm will be able to read `reference/grading.md`. That way the eval measures the skill together with its reference material. If you want a smaller change instead, keep the grading content short and inline it into `SKILL.md`, then rerun the case.", + "why": "F-near: says adding the plugin directory to context.add_dirs fixes the denial, which the rubric fails." + } + ] +} diff --git a/plugins/evals/evals/reference-read-denied/samples/fix.json b/plugins/evals/evals/reference-read-denied/samples/fix.json new file mode 100644 index 0000000000..ba48132f61 --- /dev/null +++ b/plugins/evals/evals/reference-read-denied/samples/fix.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "The eval run refuses Reads under the plugin's own directory, whatever the case's Read grant; the grant covers the tool, not that path. Only the hub `SKILL.md`, injected as the skill body, reaches the model.\n\nSo move the content your cases depend on out of `reference/grading.md` and into `SKILL.md`, staying within its size limit. Then the with-arm can actually use it, and the delta measures what the skill knows rather than what it can't open. Rerun the affected cases afterwards to compare. Until then, a null delta says more about the hub than about the plugin.", + "why": "P-canonical: says to move the content the cases depend on into the hub SKILL.md within its size limit so the with-arm can measure it." + }, + { + "answer": "That message is the eval's own sandbox, and no change to `allowed_tools` gets past it: inside the with-arm, files under the plugin's folder are off limits even with Read granted. The model only ever sees the hub `SKILL.md` body.\n\nThe eval measures only what `SKILL.md` carries, so whatever your cases test has to be written there. Pick the grading rules the cases actually rely on and put them into the hub itself, inside its size limit; spoke files can stay for everything the cases don't touch. Then rerun and compare the delta with the earlier run.", + "why": "P-alt: says the eval measures only what SKILL.md carries, so what the cases test must go there, the rubric's accepted alternative wording." + }, + { + "answer": "In the with-arm, a Read of any file under the plugin's own directory is refused with exactly that message, regardless of the case's Read grant. Only the hub `SKILL.md` text reaches the model.\n\nWhat to change: move the content the cases depend on into `SKILL.md`, within its size limit.\n\nTwo caveats. A path-scoped `--allow-tools \"Read(//home/dev/repo/plugins/notes/**)\"` grant does not lift the denial, so don't spend time on permissions. And until you move it, a null delta on this plugin is a finding about the hub, not proof the skill adds nothing.", + "why": "P-caveat: says to move the case-critical content into SKILL.md within its size limit; the caveats about the path-scoped grant and the null delta are correct and do not touch the graded point." + }, + { + "answer": "That read is blocked by the eval sandbox, so the with-arm never gets `grading.md`, and the answers you see come from a skill missing its grading rules.\n\nTo make the eval measure what the skill knows, move the content your cases depend on into `SKILL.md` itself, keeping it within the size limit. Pick the rules, thresholds and examples the cases actually check and put them in the hub; trim wording elsewhere in `SKILL.md` if you need room. Then rerun the cases and compare the delta with your earlier run; the gap should now reflect the grading knowledge itself.", + "why": "P-plain: says to move the content into SKILL.md within its size limit, while missing the sibling's points (no mention that the Read grant is irrelevant or that only the hub reaches the model)." + } + ], + "fail": [ + { + "answer": "Read in `allowed_tools` lets the model call the tool, but a deny rule on that directory still wins. Check the settings the eval inherits, your user and project `settings.json`, for a `permissions.deny` entry covering `skills/` and remove it or narrow it. If you can't find one, grant the path directly when you launch: `claude plugin eval ./plugins/notes --allow-tools \"Read(//home/dev/repo/plugins/notes/**)\"`. Once the read succeeds, the with-arm sees the full reference and your delta reflects what the skill knows. Leave `SKILL.md` as it is; the layout isn't the problem, and splitting detail into `reference/` is good practice for keeping the hub short.", + "why": "F-wrong: offers only permission and allow-tools changes as the fix, never moving content into SKILL.md." + }, + { + "answer": "The eval run refuses Reads under the plugin's own directory whatever the case's Read grant; the grant covers the tool, not that path. Only the hub `SKILL.md` text, the injected skill body, reaches the model, so `reference/grading.md` never arrives.\n\nWhat to change: rewrite the cases so they only test what `SKILL.md` already says, and drop the expectations that depend on the grading reference. Then the with-arm is scored on knowledge it can actually see and the delta stops being dragged down by unreachable files. Rerun the suite afterwards and compare the new delta with the old one.", + "why": "F-partial: explains the cause correctly but never says to move the content into SKILL.md; it changes the cases instead." + }, + { + "answer": "In the with-arm, Reads under the plugin's own folder are refused no matter what the case's Read grant is, and only the hub `SKILL.md` body reaches the model.\n\nThe obvious move is to put the grading content into `SKILL.md`. I'd advise against it, though: keep `SKILL.md` lean and leave the detail in `reference/`, where it belongs for real use. Accept that the with-arm score understates the skill, and note in your results that the reference wasn't readable during the eval. Readers of the report can then discount the with-arm accordingly.", + "why": "F-retract: mentions moving the content into SKILL.md, then retracts it, says to keep the reference out of SKILL.md, and accepts the score." + }, + { + "answer": "The run refuses reads under the plugin's own directory whatever the case's Read grant, so only the hub `SKILL.md` reaches the model. Files the case owns are a different matter, because the case directory is part of the run's workspace.\n\nSo copy `reference/grading.md` into the case directory next to `prompt.md`, and point the prompt at the local copy. The with-arm can read it there, and you keep `SKILL.md` short while the eval still measures the grading knowledge. Repeat that for each case that needs the reference, and keep the copies in sync when the reference changes.", + "why": "F-near: proposes copying the reference into the case directory, a fix the rubric explicitly fails." + } + ] +} diff --git a/plugins/evals/evals/reference-read-denied/samples/skill-fired.json b/plugins/evals/evals/reference-read-denied/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/reference-read-denied/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/route-claude-api-app/graders/did-not-start-build-eval.md b/plugins/evals/evals/route-claude-api-app/graders/did-not-start-build-eval.md new file mode 100644 index 0000000000..f208e53bf2 --- /dev/null +++ b/plugins/evals/evals/route-claude-api-app/graders/did-not-start-build-eval.md @@ -0,0 +1,8 @@ +--- +type: tool_used +tool: Skill +input_match: "(?=.*\"skill\"\\s*:\\s*\"claude-api\")(?=.*build-eval)" +min: 0 +max: 0 +arm: with-only +--- diff --git a/plugins/evals/evals/route-claude-api-app/graders/routes-to-build-eval.md b/plugins/evals/evals/route-claude-api-app/graders/routes-to-build-eval.md new file mode 100644 index 0000000000..fcb314b9fa --- /dev/null +++ b/plugins/evals/evals/route-claude-api-app/graders/routes-to-build-eval.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer's main recommendation for creating this service's first eval suite is the bundled claude-api skill's `build-eval` subcommand (for example `/claude-api build-eval`). Describing what build-eval sets up (labelled inputs, a grader for the label) is fine. + +FAIL if build-eval is absent or offered only as an optional extra; if the main route is something else (a hand-written harness, promptfoo, `/claude-api hillclimb`, `claude plugin eval`, or an `/evals:*` skill); if the answer says it has started build-eval; or if it later contradicts or retracts the recommendation. diff --git a/plugins/evals/evals/route-claude-api-app/graders/skill-fired.md b/plugins/evals/evals/route-claude-api-app/graders/skill-fired.md new file mode 100644 index 0000000000..2e7e76a135 --- /dev/null +++ b/plugins/evals/evals/route-claude-api-app/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:methodology)|evals:design)\"" +--- diff --git a/plugins/evals/evals/route-claude-api-app/graders/token-build-eval.md b/plugins/evals/evals/route-claude-api-app/graders/token-build-eval.md new file mode 100644 index 0000000000..913de37b17 --- /dev/null +++ b/plugins/evals/evals/route-claude-api-app/graders/token-build-eval.md @@ -0,0 +1,6 @@ +--- +type: regex +pattern: "build[- ]eval" +flags: i +arm: with-only +--- diff --git a/plugins/evals/evals/route-claude-api-app/prompt.md b/plugins/evals/evals/route-claude-api-app/prompt.md new file mode 100644 index 0000000000..6dce0b809b --- /dev/null +++ b/plugins/evals/evals/route-claude-api-app/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The base model writes its own harness (pytest, promptfoo) and has no reason to know the bundled claude-api skill's build-eval subcommand; the blind base answers never named it. The did-not-start guard is an unscored with-only indicator; the rubric fails an answer that says it started build-eval." +tags: [routing, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer's main route for this service's first eval suite is the bundled claude-api skill's build-eval (`/claude-api build-eval`), and no Skill call starts build-eval" +--- + +Our repo is a small Python service: a FastAPI endpoint passes each incoming support email to Claude using the `anthropic` package and gets back a priority label. We have no evals at all yet. What's the right way to get evals in place here? Answer in under 150 words. diff --git a/plugins/evals/evals/route-claude-api-app/samples/did-not-start-build-eval.json b/plugins/evals/evals/route-claude-api-app/samples/did-not-start-build-eval.json new file mode 100644 index 0000000000..636b2f8dec --- /dev/null +++ b/plugins/evals/evals/route-claude-api-app/samples/did-not-start-build-eval.json @@ -0,0 +1,86 @@ +{ + "pass": [ + { + "answer": [], + "why": "no tool call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "this plugin answered" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "claude-api read for its docs, no subcommand" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api", + "args": "how should this repo get evals?" + } + } + ], + "why": "claude-api asked a question, no subcommand" + } + ], + "fail": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api", + "args": "build-eval" + } + } + ], + "why": "started build-eval" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + }, + { + "tool": "Skill", + "input": { + "skill": "claude-api", + "args": "build-eval --target support-bot" + } + } + ], + "why": "started after routing" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "args": "build-eval", + "skill": "claude-api" + } + } + ], + "why": "args before skill: key order does not matter" + } + ] +} diff --git a/plugins/evals/evals/route-claude-api-app/samples/routes-to-build-eval.json b/plugins/evals/evals/route-claude-api-app/samples/routes-to-build-eval.json new file mode 100644 index 0000000000..1895101995 --- /dev/null +++ b/plugins/evals/evals/route-claude-api-app/samples/routes-to-build-eval.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Your service calls Claude through the `anthropic` SDK, so it is a Claude API app, and the bundled claude-api skill has a subcommand built for exactly this case. With no eval in place yet, type this yourself in Claude Code from the repo root:\n\n```\n/claude-api build-eval\n```\n\nIt sets up a first eval suite for the priority classifier: labelled support emails as inputs and a grader that checks the label your endpoint returns. Feed it a spread of real emails, including the awkward ones (mixed urgency, forwarded threads, angry but low-priority). Once that suite exists and scores the service, you have a baseline to measure every later prompt or model change against.", + "why": "P-canonical: the main recommendation is /claude-api build-eval for the first suite, described as labelled inputs plus a label grader, which the rubric allows; the user starts it." + }, + { + "answer": "Short version: start with build-eval from the claude-api skill that ships with Claude Code.\n\n1. Why this route: the FastAPI handler sends each email to Claude via the Anthropic SDK, which is the kind of app the claude-api skill covers.\n2. What to run: open Claude Code in the repo and enter `/claude-api build-eval`. You start it; it builds the suite from there.\n3. What you get: labelled example emails paired with the priority each should receive, plus a grader that scores the returned label against them.\n\nCollect forty or so real emails with agreed labels before you start, so the suite reflects actual traffic rather than invented examples.", + "why": "P-alt: same main route (build-eval from the bundled claude-api skill) as a numbered list, with the skill named in prose before the command." + }, + { + "answer": "Because the endpoint reaches Claude through the `anthropic` package, the route is the claude-api skill bundled with Claude Code. Its `build-eval` subcommand creates a first eval suite when none exists, which is your situation:\n\n```\n/claude-api build-eval\n```\n\nRun that yourself; it builds a set of labelled emails and a grader for the priority label.\n\nTwo things worth knowing. First, since the output is a single label, an exact-match check is usually enough to grade it; a classifier like this rarely needs a model judge. Second, once the suite exists, the same skill has `/claude-api hillclimb`, which searches model, effort and prompt settings against that suite. That is a later step for improving the service, not a substitute for building the suite first.", + "why": "P-caveat: build-eval stays the main route; hillclimb appears only as a later improvement step after the suite exists, and the exact-match note does not change the recommendation." + }, + { + "answer": "Since the service calls Claude through the Anthropic SDK, use the bundled claude-api skill. In Claude Code, run `/claude-api build-eval`. That is the subcommand for a Claude API app with no eval yet: it sets you up with labelled example emails and a grader that checks the priority label your endpoint returns. Start it yourself from the repository, give it a realistic mix of support emails, keep the labelled set under version control next to the service code, and treat the first scored run as your baseline for future changes.", + "why": "P-plain: minimal answer whose only route is /claude-api build-eval (this case has no sibling graders, so it is the bare correct answer)." + } + ], + "fail": [ + { + "answer": "Start with a labelled dataset and a small harness of your own.\n\n1. Pull 50 to 100 real support emails from your inbox history and have someone on the support team assign each the correct priority. Include edge cases: mixed signals, forwarded threads, very short messages.\n2. Write a pytest suite that sends each email through the same function your FastAPI endpoint uses and compares the returned label with the expected one by exact match.\n3. Report overall accuracy plus a per-class confusion matrix, since misreading an urgent email costs more than over-flagging a routine one.\n4. Run it in CI on every prompt or model change.\n\nA single-label classifier needs no special tooling; plain pytest covers it.", + "why": "F-wrong: the main route is a hand-written pytest harness and build-eval is absent." + }, + { + "answer": "This is a Claude API app, because the endpoint calls Claude through the `anthropic` package, so lean on the claude-api skill that ships with Claude Code rather than building a framework from scratch. Open Claude Code in the repo, invoke `/claude-api`, and ask it to set up evaluation for the priority classifier. Give it the location of the handler and a sample of real support emails with the label each one should get. Grade the label by exact match, since the output is one of a fixed set of priorities, and keep the labelled emails in the repository so every prompt or model change can be scored against the same set.", + "why": "F-partial: names the claude-api skill but never the build-eval subcommand, so build-eval is absent (the one required point)." + }, + { + "answer": "The bundled claude-api skill has a `build-eval` subcommand meant for Claude API apps like yours, and `/claude-api build-eval` is the usual way to stand up a first suite: labelled emails in, a grader for the priority label out.\n\nFor a service this small, though, I'd skip it. A single-label classifier doesn't need a generated suite. Write a CSV of fifty real emails with their correct priorities, add a pytest test that runs each through your handler and asserts the label, and print accuracy per priority class. That gets you running today with nothing new to learn, and you can wire it into CI immediately. Treat the pytest file as your eval suite from here on.", + "why": "F-retract: names build-eval, then retracts it ('I'd skip it') and makes a hand-written pytest harness the actual route." + }, + { + "answer": "Your service calls Claude via the `anthropic` package, so this is a Claude API app, and the right route is the bundled claude-api skill's `build-eval` subcommand. I've already started `/claude-api build-eval` for you against this repository; it is scanning the FastAPI handler now and will set up a labelled set of support emails plus a grader for the priority label. When it asks, confirm the label set (for example urgent, high, normal, low) and point it at a folder of real emails. Once the suite exists, the first scored run becomes your baseline for every later prompt or model change.", + "why": "F-near: right route, but the answer says it has started build-eval itself, which the rubric's FAIL list names." + } + ] +} diff --git a/plugins/evals/evals/route-claude-api-app/samples/skill-fired.json b/plugins/evals/evals/route-claude-api-app/samples/skill-fired.json new file mode 100644 index 0000000000..abcfb658c9 --- /dev/null +++ b/plugins/evals/evals/route-claude-api-app/samples/skill-fired.json @@ -0,0 +1,83 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:design" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "methodology" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "design" + } + } + ], + "why": "the bundled design skill, not evals:design" + } + ] +} diff --git a/plugins/evals/evals/route-claude-api-app/samples/token-build-eval.json b/plugins/evals/evals/route-claude-api-app/samples/token-build-eval.json new file mode 100644 index 0000000000..e8970f560c --- /dev/null +++ b/plugins/evals/evals/route-claude-api-app/samples/token-build-eval.json @@ -0,0 +1,18 @@ +{ + "pass": [ + { + "answer": "Type `/claude-api build-eval` to create the first suite.", + "why": "command form" + }, + { + "answer": "Use the claude-api skill's build eval subcommand.", + "why": "spaced form" + } + ], + "fail": [ + { + "answer": "Write a pytest suite over 50 labelled emails and check the label by exact match.", + "why": "harness only" + } + ] +} diff --git a/plugins/evals/evals/route-hillclimb-existing-suite/graders/did-not-start-hillclimb.md b/plugins/evals/evals/route-hillclimb-existing-suite/graders/did-not-start-hillclimb.md new file mode 100644 index 0000000000..2a56ba1e54 --- /dev/null +++ b/plugins/evals/evals/route-hillclimb-existing-suite/graders/did-not-start-hillclimb.md @@ -0,0 +1,8 @@ +--- +type: tool_used +tool: Skill +input_match: "(?=.*\"skill\"\\s*:\\s*\"claude-api\")(?=.*hillclimb)" +min: 0 +max: 0 +arm: with-only +--- diff --git a/plugins/evals/evals/route-hillclimb-existing-suite/graders/routes-to-hillclimb.md b/plugins/evals/evals/route-hillclimb-existing-suite/graders/routes-to-hillclimb.md new file mode 100644 index 0000000000..f04a75ce44 --- /dev/null +++ b/plugins/evals/evals/route-hillclimb-existing-suite/graders/routes-to-hillclimb.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer's main recommendation for this search is the bundled claude-api skill's `hillclimb` subcommand (for example `/claude-api hillclimb`), run against the eval set the user already has. Naming it in words ("the claude-api skill's hillclimb workflow") counts without the literal slash command. An outline of the steps hillclimb itself runs is not a hand-rolled grid when hillclimb is the named route, and offering to start it is fine. + +FAIL if hillclimb is absent or offered only as an optional extra; if the step it gives is `/claude-api build-eval` (the eval set already exists) or rebuilding the suite with `/evals:design` or `claude plugin eval`; if the main route is a hand-rolled grid, a script, DSPy, or a promptfoo matrix; if the answer says it has started hillclimb; or if it later contradicts or retracts the recommendation. diff --git a/plugins/evals/evals/route-hillclimb-existing-suite/graders/token-hillclimb.md b/plugins/evals/evals/route-hillclimb-existing-suite/graders/token-hillclimb.md new file mode 100644 index 0000000000..8a03241602 --- /dev/null +++ b/plugins/evals/evals/route-hillclimb-existing-suite/graders/token-hillclimb.md @@ -0,0 +1,6 @@ +--- +type: regex +pattern: "hill-?climb" +flags: i +arm: with-only +--- diff --git a/plugins/evals/evals/route-hillclimb-existing-suite/prompt.md b/plugins/evals/evals/route-hillclimb-existing-suite/prompt.md new file mode 100644 index 0000000000..965c3b017b --- /dev/null +++ b/plugins/evals/evals/route-hillclimb-existing-suite/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Routine guard (regression-guard). Routes a model-and-effort search over an existing eval set to the bundled claude-api skill's hillclimb. A no-plugin run passed it, so expect about 1.00 in both arms; kept as a routine guard on the route, not as evidence of plugin value. The did-not-start guard is an unscored with-only indicator; the rubric fails an answer that says it started hillclimb." +tags: [routing, routine, regression-guard, description-hint] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer's main route for the search is the bundled claude-api skill's hillclimb (`/claude-api hillclimb`) against the existing eval set, and no Skill call starts hillclimb" +--- + +We have a Claude-backed support bot and an eval set that already scores it. I want to find the cheapest model and effort setting that still hits our 90% target, adjusting the prompt along the way. What should I use for that? Answer in under 150 words. diff --git a/plugins/evals/evals/route-hillclimb-existing-suite/samples/did-not-start-hillclimb.json b/plugins/evals/evals/route-hillclimb-existing-suite/samples/did-not-start-hillclimb.json new file mode 100644 index 0000000000..f19bcfad44 --- /dev/null +++ b/plugins/evals/evals/route-hillclimb-existing-suite/samples/did-not-start-hillclimb.json @@ -0,0 +1,86 @@ +{ + "pass": [ + { + "answer": [], + "why": "no tool call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "this plugin answered" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "claude-api read for its docs, no subcommand" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api", + "args": "how should this repo get evals?" + } + } + ], + "why": "claude-api asked a question, no subcommand" + } + ], + "fail": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api", + "args": "hillclimb" + } + } + ], + "why": "started hillclimb" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + }, + { + "tool": "Skill", + "input": { + "skill": "claude-api", + "args": "hillclimb --target support-bot" + } + } + ], + "why": "started after routing" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "args": "hillclimb", + "skill": "claude-api" + } + } + ], + "why": "args before skill: key order does not matter" + } + ] +} diff --git a/plugins/evals/evals/route-hillclimb-existing-suite/samples/routes-to-hillclimb.json b/plugins/evals/evals/route-hillclimb-existing-suite/samples/routes-to-hillclimb.json new file mode 100644 index 0000000000..8c72ef15e6 --- /dev/null +++ b/plugins/evals/evals/route-hillclimb-existing-suite/samples/routes-to-hillclimb.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Use the `hillclimb` subcommand of the claude-api skill bundled with Claude Code, pointed at the eval set you already have:\n\n```\n/claude-api hillclimb\n```\n\nType it yourself in Claude Code. It searches model and effort settings against your existing suite and proposes prompt changes along the way, applying a change only when you accept it. Give it the 90% pass-rate target and let it find the cheapest configuration that still clears it. Because your eval set already scores the bot, there is nothing to build first; the search starts from the scores you have.", + "why": "P-canonical: the main recommendation is /claude-api hillclimb run against the user's existing eval set." + }, + { + "answer": "This is exactly the job of hillclimb in the claude-api skill that ships with Claude Code.\n\n- Input: your existing eval set, unchanged, plus the 90% target.\n- What it does: sweeps cheaper models and lower effort levels, and suggests prompt edits, scoring each candidate against your suite.\n- Control: it proposes changes to the bot's configuration and prompt, and applies them only once you accept.\n- How to start: run `/claude-api hillclimb` from Claude Code in the bot's repository.\n\nPick the cheapest configuration that holds 90% across repeated runs, not a single lucky one.", + "why": "P-alt: hillclimb named first in prose, then laid out as a list; the input is the existing eval set." + }, + { + "answer": "Run `/claude-api hillclimb` against the eval set you already have. It's the claude-api skill's subcommand for exactly this search: it tries cheaper models and lower effort settings, proposes prompt adjustments, and scores each candidate on your suite, applying a change to the bot only when you approve it.\n\nTwo cautions before you trust the winner. First, a configuration that scores 91% once may sit at 88% on the next run, so re-run the finalists a few times and prefer one with some margin above 90%. Second, prompt tuning against the same cases you score on can overfit; if your set is large enough, hold back a slice the search never sees and confirm the chosen configuration on it.", + "why": "P-caveat: hillclimb on the existing set is the main route; the cautions about run-to-run noise and a held-back slice neither replace nor rebuild the suite." + }, + { + "answer": "Since you already have an eval set that scores the bot, use the bundled claude-api skill's `hillclimb` subcommand. Type `/claude-api hillclimb` in Claude Code, point it at that existing set, and give it the 90% target. It searches model and effort settings, proposes prompt changes, and applies them when you accept, so you end up with the cheapest setup that still clears the bar on your own cases. Start it yourself from the bot's repository; nothing needs to be rebuilt before the search begins. Re-run the winner a couple of times before switching production over.", + "why": "P-plain: minimal correct answer: /claude-api hillclimb against the existing set (this case has no sibling graders)." + } + ], + "fail": [ + { + "answer": "Write a small sweep script around the eval set you already have.\n\n```python\nfor model in ['haiku', 'sonnet', 'opus']:\n for effort in ['low', 'medium', 'high']:\n score, cost = run_evals(model, effort, prompt)\n results.append((model, effort, score, cost))\n```\n\nRun each configuration three times to smooth out noise, then sort by cost and take the first one whose mean pass rate clears 90%. For the prompt, try two or three variants per round and keep the winner only if it holds on a held-out slice. A plain grid like this is transparent, cheap to rerun, and needs no extra tooling.", + "why": "F-wrong: the main route is a hand-rolled grid script; hillclimb is absent." + }, + { + "answer": "The bundled claude-api skill in Claude Code is built for this kind of search on a Claude API app. Open Claude Code in the bot's repository, invoke `/claude-api`, and ask it to find the cheapest model and effort setting that keeps your existing eval set at 90% or above, adjusting the prompt as it goes. Point it at the eval set you already have so every candidate is scored on the same cases, and approve any prompt change only after you have read it. Re-run the final choice a few times to confirm it holds.", + "why": "F-partial: names the claude-api skill and the existing eval set but never the hillclimb subcommand, so hillclimb is absent (the one missing point)." + }, + { + "answer": "The claude-api skill bundled with Claude Code has a `hillclimb` subcommand, `/claude-api hillclimb`, that searches against an existing eval set like yours.\n\nI'd steer you away from it for this goal, though. You want a clean comparison of cost versus pass rate across every model and effort level, and an automated search hides the configurations it skips. Write a short script that loops over each model and effort combination, runs your eval set three times per configuration, and logs pass rate and cost. Then pick the cheapest row above 90% by hand. That gives you the full table to show stakeholders.", + "why": "F-retract: names hillclimb, then steers away from it and makes a hand-written sweep script the route." + }, + { + "answer": "Start with `/claude-api build-eval` to regenerate your eval set in the format the claude-api skill reads; your existing scorer won't plug into its search as it stands. Once build-eval has produced the suite, run `/claude-api hillclimb` against it with the 90% target. Hillclimb sweeps models and effort settings, proposes prompt edits, and applies the ones you accept, so the result is the cheapest configuration that clears your bar. Rebuilding first costs an afternoon but means every candidate is graded the same way. Keep the old scorer around for comparison until the new suite agrees with it.", + "why": "F-near: the first step it gives is /claude-api build-eval to rebuild a suite that already exists, and hillclimb runs on the rebuilt suite rather than the user's set." + } + ] +} diff --git a/plugins/evals/evals/route-hillclimb-existing-suite/samples/token-hillclimb.json b/plugins/evals/evals/route-hillclimb-existing-suite/samples/token-hillclimb.json new file mode 100644 index 0000000000..6baa641980 --- /dev/null +++ b/plugins/evals/evals/route-hillclimb-existing-suite/samples/token-hillclimb.json @@ -0,0 +1,18 @@ +{ + "pass": [ + { + "answer": "Type `/claude-api hillclimb` against your eval set.", + "why": "command form" + }, + { + "answer": "Use the hill-climb subcommand of the claude-api skill.", + "why": "hyphenated" + } + ], + "fail": [ + { + "answer": "Script a grid over three models and two effort levels and pick the cheapest that hits 90%.", + "why": "grid only" + } + ] +} diff --git a/plugins/evals/evals/route-mixed-repo/graders/plugin-part.md b/plugins/evals/evals/route-mixed-repo/graders/plugin-part.md new file mode 100644 index 0000000000..13a53eb46f --- /dev/null +++ b/plugins/evals/evals/route-mixed-repo/graders/plugin-part.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if, for the plugin part (the `.claude-plugin/plugin.json` with three skills), the answer routes it to `claude plugin eval` (Claude Code's built-in `plugin eval` command, measured against a no-plugin baseline). Naming `/evals:design`, `/evals:plugin-eval`, or `claude plugin eval init` as the way to scaffold or run it also meets this, as long as the command underneath is `claude plugin eval`. + +FAIL if the answer sends `claude plugin eval` (or `/evals:*`) to the summarizer service instead of the plugin; proposes one harness (pytest, promptfoo, a single runner) for the whole repository, skills included; routes the plugin's skills to `/claude-api build-eval` or `/claude-api hillclimb`; gives the plugin no specific route (for example only "write cases for each skill"); or later contradicts or retracts this. diff --git a/plugins/evals/evals/route-mixed-repo/graders/service-part.md b/plugins/evals/evals/route-mixed-repo/graders/service-part.md new file mode 100644 index 0000000000..1d6c03f292 --- /dev/null +++ b/plugins/evals/evals/route-mixed-repo/graders/service-part.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if, for the `services/summarizer/` part, the answer sends it to the bundled claude-api skill's `build-eval` (for example `/claude-api build-eval`) as the way to create its evals. + +FAIL if the answer sends the summarizer to `claude plugin eval` or an `/evals:*` skill; makes `/claude-api hillclimb` the first step (no eval exists yet); gives it only a hand-written harness (pytest, promptfoo, a reference-summary set with a judge); or later contradicts or retracts this. diff --git a/plugins/evals/evals/route-mixed-repo/graders/skill-fired.md b/plugins/evals/evals/route-mixed-repo/graders/skill-fired.md new file mode 100644 index 0000000000..2e7e76a135 --- /dev/null +++ b/plugins/evals/evals/route-mixed-repo/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:methodology)|evals:design)\"" +--- diff --git a/plugins/evals/evals/route-mixed-repo/prompt.md b/plugins/evals/evals/route-mixed-repo/prompt.md new file mode 100644 index 0000000000..2dfe5c569c --- /dev/null +++ b/plugins/evals/evals/route-mixed-repo/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The base model proposes one harness, or generic per-part harnesses; the two specific routes (`claude plugin eval` for the plugin, `/claude-api build-eval` for the service) are this plugin's route table." +tags: [routing, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The plugin part goes to `claude plugin eval` (scaffolded with /evals:design, run with /evals:plugin-eval), and the summarizer service goes to `/claude-api build-eval`" +--- + +This repo has a `.claude-plugin/plugin.json` with three skills, plus a `services/summarizer/` folder whose Python code calls the Anthropic Messages API to summarize meeting notes. I want evals covering the whole repo. What route should each part take? Answer in under 150 words. diff --git a/plugins/evals/evals/route-mixed-repo/samples/plugin-part.json b/plugins/evals/evals/route-mixed-repo/samples/plugin-part.json new file mode 100644 index 0000000000..64e7c0efeb --- /dev/null +++ b/plugins/evals/evals/route-mixed-repo/samples/plugin-part.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Split it by part, because the two need different tools.\n\n- **Plugin (`.claude-plugin/plugin.json`, three skills):** use `claude plugin eval`, Claude Code's built-in runner, which measures each case with the plugin against a no-plugin baseline. Scaffold the suite with `/evals:design` and run it with `/evals:plugin-eval`.\n- **`services/summarizer/`:** it calls the Anthropic Messages API, so it is a Claude API app. Type `/claude-api build-eval` to create its first eval suite; later, `/claude-api hillclimb` can tune it against that suite.\n\nKeep the two suites separate. The plugin cases test whether the skills change Claude Code's behavior; the summarizer cases test summary quality on real meeting notes.", + "why": "P-canonical: the plugin goes to claude plugin eval against a no-plugin baseline, scaffolded with /evals:design and run with /evals:plugin-eval." + }, + { + "answer": "Two parts, two routes. The summarizer under `services/summarizer/` is ordinary application code calling Claude through the Messages API, so its evals start with the bundled claude-api skill: run `/claude-api build-eval`. The plugin is different. A `.claude-plugin/plugin.json` with skills is evaluated by Claude Code's own `claude plugin eval` command, which runs each case with the plugin installed and without it, and reports the difference. The evals plugin wraps that: `/evals:design` interviews you and scaffolds the plugin's cases, and `/evals:plugin-eval` runs them. Don't fold the skills into the summarizer's suite or the other way round; each part gets its own route.", + "why": "P-alt: service first, then the plugin routed to claude plugin eval (with/without comparison) via /evals:design and /evals:plugin-eval, in prose." + }, + { + "answer": "For the plugin with its three skills, the route is `claude plugin eval`: cases run with the plugin and against a no-plugin baseline, and the delta is what you read. You can scaffold the cases with `claude plugin eval init`, or through `/evals:design`, and run them with `/evals:plugin-eval`. For `services/summarizer/`, which calls the Anthropic Messages API, type `/claude-api build-eval` to create its first eval suite.\n\nOne caution on the plugin side: every case runs in both arms, so the suite costs roughly twice the model calls of the cases alone. Start with a few cases per skill, covering when each skill should fire and when it should stay quiet, and grow the set once the first results look trustworthy.", + "why": "P-caveat: plugin routed to claude plugin eval (init or /evals:design to scaffold, both allowed by the rubric), plus a cost caution that does not change the route." + }, + { + "answer": "Treat them as two targets.\n\n1. The plugin: its three skills are covered by `claude plugin eval`, which compares runs with the plugin installed against a no-plugin baseline. `/evals:design` scaffolds the cases and `/evals:plugin-eval` runs them.\n2. The summarizer: build a set of 30 or so meeting notes with reference summaries, then write a pytest harness that calls the service and has an LLM judge score each output for faithfulness and coverage of action items, plus a code check on length.\n\nRun the two separately; they measure different things and fail for different reasons.", + "why": "P-plain: plugin correctly routed to claude plugin eval; the summarizer gets a hand-written harness instead of build-eval, which only the service-part sibling grades." + } + ], + "fail": [ + { + "answer": "Use one harness for the whole repository so results land in one place. promptfoo fits well here:\n\n```\nnpx promptfoo init\n```\n\nDefine two groups of tests in `promptfooconfig.yaml`. For the summarizer, point a provider at the Anthropic Messages API, feed it sample meeting notes, and grade outputs with `llm-rubric` assertions on faithfulness and coverage. For the plugin, add a test per skill that sends the skill's trigger prompt to Claude with the SKILL.md text included and asserts on the response. One config, one report, one CI job, and both parts of the repo are covered by the same pass/fail dashboard.", + "why": "F-wrong: one promptfoo harness for the whole repository, skills included; the plugin never goes to claude plugin eval." + }, + { + "answer": "Each part needs its own approach. For `services/summarizer/`, which calls Claude through the Messages API, type `/claude-api build-eval` to create its first eval suite; it sets up example meeting notes and a grader for the summaries. For the plugin, write a handful of cases for each of the three skills: a prompt that should trigger the skill, a prompt that should not, and what a good response looks like. Keep those cases next to each skill's folder so they are updated whenever the skill changes, and review them together before each release.", + "why": "F-partial: the service route is right, but the plugin gets no specific route (only 'write cases for each skill'), the one missing point." + }, + { + "answer": "For the plugin part, the natural fit is `claude plugin eval`, which runs cases against a no-plugin baseline; `/evals:design` and `/evals:plugin-eval` scaffold and run it. For `services/summarizer/`, type `/claude-api build-eval` to create its first suite.\n\nOn reflection, though, running two separate tool chains for one repository is more overhead than it is worth. The skills are prompts sent to Claude in the end, so drop the plugin eval route and fold the three skills into the summarizer's build-eval suite as extra inputs. One suite, one grader setup, one place to read results, and both parts of the repository are covered.", + "why": "F-retract: routes the plugin to claude plugin eval, then drops that route and sends the skills into /claude-api build-eval." + }, + { + "answer": "Route each part by what it is.\n\n- `services/summarizer/` is where Claude is actually prompted, so give it the plugin eval route: scaffold its cases with `/evals:design`, run them with `/evals:plugin-eval`, and let `claude plugin eval` compare summary quality across runs.\n- The plugin's three skills are prompts that reach Claude through the API too, so `/claude-api build-eval` can generate their suite, and `/claude-api hillclimb` can then tune each skill's wording against it.\n\nThis keeps every model-facing piece of the repo under a dedicated tool rather than a hand-rolled harness.", + "why": "F-near: the routes are swapped: claude plugin eval and /evals:* go to the summarizer, and the plugin's skills go to /claude-api build-eval and hillclimb." + } + ] +} diff --git a/plugins/evals/evals/route-mixed-repo/samples/service-part.json b/plugins/evals/evals/route-mixed-repo/samples/service-part.json new file mode 100644 index 0000000000..aef40dd349 --- /dev/null +++ b/plugins/evals/evals/route-mixed-repo/samples/service-part.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Give each part of the repository its own eval route.\n\n- `services/summarizer/` calls the Anthropic Messages API, which makes it a Claude API app. With no eval there yet, type `/claude-api build-eval` to create its first suite: sample meeting notes as inputs plus graders for the summaries.\n- The plugin (`.claude-plugin/plugin.json` and its three skills) goes to `claude plugin eval`, measured against a no-plugin baseline. Scaffold it with `/evals:design` and run it with `/evals:plugin-eval`.\n\nOnce the summarizer has a suite, `/claude-api hillclimb` is available for tuning model, effort and prompt against it.", + "why": "P-canonical: the summarizer goes to /claude-api build-eval to create its evals; hillclimb comes only after the suite exists." + }, + { + "answer": "The repository mixes a plugin and an application, and they get evaluated in different places. The three skills behind `.claude-plugin/plugin.json` belong to Claude Code's `claude plugin eval` command, which compares behavior with and without the plugin; `/evals:design` sets up those cases and `/evals:plugin-eval` runs them. The meeting-notes summarizer is a different kind of code: Python calling Claude through the Messages API. Its evals come from the claude-api skill bundled with Claude Code, starting with the build-eval subcommand, `/claude-api build-eval`, which you type yourself to stand up its first suite.", + "why": "P-alt: plugin first, then the summarizer sent to the bundled claude-api skill's build-eval, named in prose and as a command." + }, + { + "answer": "For `services/summarizer/`: it calls Claude through the Messages API, so type `/claude-api build-eval` to build its first eval suite. Two cautions there. Meeting notes often carry names and internal details, so use redacted or synthetic notes as inputs. And wait to reach for `/claude-api hillclimb` until that suite exists and gives you a baseline; hillclimb improves against a suite, it doesn't create one.\n\nFor the plugin with three skills: `claude plugin eval` runs its cases with the plugin and against a no-plugin baseline. Use `/evals:design` to scaffold the cases and `/evals:plugin-eval` to run them.\n\nTwo suites, each matched to what its part actually is.", + "why": "P-caveat: build-eval is the summarizer's route; the privacy caution and 'hillclimb only after the suite exists' are extras that keep build-eval first." + }, + { + "answer": "For the summarizer service, the route is the bundled claude-api skill: type `/claude-api build-eval` and it creates the service's first eval suite, with meeting-note inputs and graders for the summaries. For the plugin, write a few test prompts per skill, run them by hand in Claude Code, and note whether each skill picks up the request and responds the way its SKILL.md describes. Record the results in a simple table so you can compare after each change to a skill. That covers both parts of the repository without any extra setup.", + "why": "P-plain: the summarizer is correctly sent to /claude-api build-eval; the plugin gets no specific route, which only the plugin-part sibling grades." + } + ], + "fail": [ + { + "answer": "Split the repo in two.\n\n**Plugin:** use `claude plugin eval` against a no-plugin baseline, scaffolding with `/evals:design` and running with `/evals:plugin-eval`.\n\n**Summarizer:** build a dataset of 40 to 50 meeting notes, each with a reference summary written by someone who attended. Then add a pytest module that calls the service on each note and asks a separate model to score the output against the reference for faithfulness, coverage of decisions and action items, and length on a 1 to 5 scale. Fail the run if the average drops below 4. A promptfoo config would work equally well if you prefer YAML to Python.", + "why": "F-wrong: the summarizer gets only a hand-written reference-summary harness with a judge (pytest or promptfoo); build-eval is absent. The plugin route is right." + }, + { + "answer": "Each part has its own home. The plugin's three skills go to `claude plugin eval`, which measures them against a no-plugin baseline; `/evals:design` scaffolds the cases and `/evals:plugin-eval` runs them. The summarizer is a Claude API app, since its Python code calls the Messages API, so its evals belong on the Claude API side rather than in the plugin runner. Set up a separate suite for it with representative meeting notes as inputs and graders for the summaries, and keep it in `services/summarizer/` so it travels with the code.", + "why": "F-partial: correctly classifies the summarizer as a Claude API app but never sends it to build-eval, the one required point." + }, + { + "answer": "For `services/summarizer/`, the starting point is `/claude-api build-eval`, which sets up a first suite for a Claude API app. For the plugin and its three skills, use `claude plugin eval` against a no-plugin baseline, scaffolded with `/evals:design` and run with `/evals:plugin-eval`.\n\nThinking about it more, build-eval is a poor fit for free-text summaries. Skip it for the summarizer and write a reference-summary set instead: thirty meeting notes, a human summary for each, and a pytest harness that has a model judge compare the service's output with the reference. That gives you finer control over what a good summary means than a generated suite would.", + "why": "F-retract: sends the summarizer to build-eval, then retracts it for a hand-written reference-summary harness." + }, + { + "answer": "The plugin goes to `claude plugin eval`: scaffold the cases with `/evals:design`, run them with `/evals:plugin-eval`, and read the result against a no-plugin baseline.\n\nThe summarizer is a Claude API app, so use the bundled claude-api skill's search directly. Type `/claude-api hillclimb` as the first step: point it at a dozen meeting notes, and it will try model, effort and prompt variations and keep whatever produces the best summaries. Its runs double as your eval history, so there is no need to build a separate suite before tuning.", + "why": "F-near: makes /claude-api hillclimb the summarizer's first step although no eval exists; build-eval is skipped." + } + ] +} diff --git a/plugins/evals/evals/route-mixed-repo/samples/skill-fired.json b/plugins/evals/evals/route-mixed-repo/samples/skill-fired.json new file mode 100644 index 0000000000..abcfb658c9 --- /dev/null +++ b/plugins/evals/evals/route-mixed-repo/samples/skill-fired.json @@ -0,0 +1,83 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:design" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "methodology" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "design" + } + } + ], + "why": "the bundled design skill, not evals:design" + } + ] +} diff --git a/plugins/evals/evals/rules-not-a-target/graders/cannot-measure.md b/plugins/evals/evals/rules-not-a-target/graders/cannot-measure.md new file mode 100644 index 0000000000..f4b927f42b --- /dev/null +++ b/plugins/evals/evals/rules-not-a-target/graders/cannot-measure.md @@ -0,0 +1,11 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. No: `claude plugin eval` cannot measure CLAUDE.md or `.claude/rules`. +2. Why: each run starts from a clean, throwaway home where CLAUDE.md, rules, settings and memory do not load. Wording such as "your project instructions and rules never load" meets this. Adding that re-injecting them through the case measures that shim, not the rules, is welcome but not required. + +FAIL if either is missing; if the answer says yes, proposes running the suite with and without CLAUDE.md through plugin eval, proposes injecting the rules through the case (append_system_prompt, a wrapper plugin) as a valid measurement, or says plugin eval loads or does not strip CLAUDE.md; or if it later contradicts or retracts this. diff --git a/plugins/evals/evals/rules-not-a-target/graders/preferred-route.md b/plugins/evals/evals/rules-not-a-target/graders/preferred-route.md new file mode 100644 index 0000000000..d7fba87d9e --- /dev/null +++ b/plugins/evals/evals/rules-not-a-target/graders/preferred-route.md @@ -0,0 +1,10 @@ +--- +type: llm +arm: both +--- + +PASS if the answer proposes measuring the rules by removing them during real work and watching what goes wrong: it names `/harness-config:unhobble`, or it describes that experiment (strip or remove the instructions, for example on a branch; work normally; log where the model stumbles; restore only what repeated evidence earns). + +This grades this repository's preferred route. A headless A/B (running the same prompts with and without the instruction files and comparing outputs) without logging stumbles during real work does not meet it. + +FAIL if the answer proposes only a plugin eval, a shim plugin that re-injects the rules, a headless A/B comparison, hooks or linters, or no alternative; or if it later contradicts or retracts this. diff --git a/plugins/evals/evals/rules-not-a-target/graders/skill-fired.md b/plugins/evals/evals/rules-not-a-target/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/rules-not-a-target/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/rules-not-a-target/prompt.md b/plugins/evals/evals/rules-not-a-target/prompt.md new file mode 100644 index 0000000000..55fd643aa9 --- /dev/null +++ b/plugins/evals/evals/rules-not-a-target/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: A base answer either thinks the runner loads CLAUDE.md, or proposes a headless A/B; this repository's route is to remove the instructions during real work and log where the model stumbles. preferred-route grades that house preference, and a plain A/B fails it by design." +tags: [target-routing, hard, house-preference] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer says no: every run starts from a throwaway home where CLAUDE.md and rules do not load, so a shim would measure the shim; the alternative is /harness-config:unhobble or, by hand, strip the instructions, log stumbles in real work, and restore only what repeated evidence earns" +--- + +Can `claude plugin eval` tell me whether the rules in our repo's CLAUDE.md and .claude/rules actually change how the model behaves? If not, what's the alternative? Answer in under 120 words. diff --git a/plugins/evals/evals/rules-not-a-target/samples/cannot-measure.json b/plugins/evals/evals/rules-not-a-target/samples/cannot-measure.json new file mode 100644 index 0000000000..6c2065199b --- /dev/null +++ b/plugins/evals/evals/rules-not-a-target/samples/cannot-measure.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "No. `claude plugin eval` can't measure CLAUDE.md or `.claude/rules`. Every run starts from a clean, throwaway home where CLAUDE.md, rules, user settings and memory never load, so there's nothing for a with-versus-without comparison to vary. Re-injecting the rules through the case would measure that shim, not the rules.\n\nThe alternative is `/harness-config:unhobble`: strip the instructions on a branch, work normally, log where the model stumbles, and restore only what repeated evidence earns.", + "why": "P-canonical: says no, and why: each run starts from a throwaway home where CLAUDE.md and rules never load." + }, + { + "answer": "It can't. The eval harness isolates each run in a fresh, disposable home directory, and your project instructions, CLAUDE.md and the `.claude/rules` files alike, are simply not loaded there. So plugin eval has no way to show their effect.\n\nTo learn whether they matter, remove them on a branch, keep working as usual, and note every place the model goes wrong; bring back only the rules that repeated stumbles justify. Expect it to take a few days of normal work before the pattern is clear.", + "why": "P-alt: 'it can't', with the reason in other words: a fresh disposable home where project instructions are not loaded." + }, + { + "answer": "No, it can't measure them. Each eval run begins in a throwaway home with CLAUDE.md, `.claude/rules`, settings and memory absent, so your rules never load in either arm. Don't try to sneak them back in through `append_system_prompt` or a wrapper plugin: that measures the shim, not the rules as Claude Code loads them.\n\nUse `/harness-config:unhobble` instead. If that plugin isn't installed, run the experiment by hand: strip the instructions on a branch, do real work, log observed stumbles, and restore only what repeated evidence earns.", + "why": "P-caveat: no plus the throwaway-home reason, with the welcome shim point added." + }, + { + "answer": "No. Plugin eval starts every run from a clean, temporary home where CLAUDE.md and `.claude/rules` don't load, so it never sees your rules and can't tell you what they change.\n\nBuild a small A/B harness instead: run the same set of prompts headless in two checkouts, one with the instruction files and one without, then grade the transcripts for the behaviors each rule is supposed to produce. Use a dozen or so prompts that exercise the rules you care about most.", + "why": "P-plain: no plus the clean-home reason; wrong on the sibling point (offers a headless A/B)." + } + ], + "fail": [ + { + "answer": "Yes. Point `claude plugin eval` at the repo root: the with arm loads CLAUDE.md and `.claude/rules` the way a normal session does, and the baseline arm strips them, so the delta is the effect of your rules.\n\nFor a second opinion from real work, `/harness-config:unhobble` strips the instructions on a branch, logs where the model stumbles, and restores only what repeated evidence earns. Run it at the default 3 runs per case so the delta isn't a single noisy sample.", + "why": "F-wrong: says yes, that plugin eval loads the rules in one arm; sibling point (unhobble) is right." + }, + { + "answer": "No. CLAUDE.md and `.claude/rules` aren't something `claude plugin eval` can measure; it isn't a supported target for that, and no flag changes it.\n\nThe alternative is `/harness-config:unhobble`. It measures standing instructions directly: strip them on a branch, keep doing real work, log where the model stumbles, and restore only the rules that repeated evidence earns back. Expect it to take a few days of ordinary work; the evidence comes from real sessions, so let the stripped branch see the kind of tasks the rules were written for.", + "why": "F-partial: says no but gives no why (nothing about the clean, throwaway home where the rules do not load)." + }, + { + "answer": "No: each eval run starts in a clean, throwaway home where CLAUDE.md, rules, settings and memory don't load, so it can't measure them.\n\nOne correction to that: the throwaway home only drops your user-level files. A CLAUDE.md and `.claude/rules` checked into the target repo still load in the run, so you can measure them by running the suite once with them and once with them deleted, and comparing scores.\n\nFor day-to-day evidence, `/harness-config:unhobble` strips them on a branch and logs where the model stumbles.", + "why": "F-retract: gives the right no and reason, then retracts it by saying repo-level CLAUDE.md and rules do load and proposes a with/without plugin eval." + }, + { + "answer": "Not directly. Every run starts in a throwaway home, so CLAUDE.md and `.claude/rules` never load on their own.\n\nThe workaround is a shim plugin: copy the rules into a small wrapper plugin that injects them through `append_system_prompt`, then run `claude plugin eval` on it. The with-plugin arm then has your rules and the baseline doesn't, so the delta measures exactly what they change.\n\nIf you also want real-work evidence, `/harness-config:unhobble` strips them on a branch and logs where the model stumbles.", + "why": "F-near: proposes injecting the rules through a wrapper plugin as a valid measurement, which the rubric lists as a FAIL." + } + ] +} diff --git a/plugins/evals/evals/rules-not-a-target/samples/preferred-route.json b/plugins/evals/evals/rules-not-a-target/samples/preferred-route.json new file mode 100644 index 0000000000..cab3d108b5 --- /dev/null +++ b/plugins/evals/evals/rules-not-a-target/samples/preferred-route.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "No. Every plugin eval run starts from a throwaway home where CLAUDE.md, `.claude/rules`, settings and memory don't load, so it can't see your rules.\n\nUse `/harness-config:unhobble` instead. It strips the instructions on a branch, you keep working normally, it logs where the model stumbles, and you restore only the rules that repeated evidence earns back. Expect it to take a few days of ordinary work; the evidence comes from real sessions rather than a scored suite, so give the branch the kind of tasks the rules were written for.", + "why": "P-canonical: names `/harness-config:unhobble` and describes the strip, work, log, restore experiment." + }, + { + "answer": "Not with plugin eval: its runs start from a clean home, so your project instructions and rules never load.\n\nTest them the other way round. On a branch, delete the CLAUDE.md content and the rule files, then do your ordinary work there for a while. Each time the model trips over something a rule used to cover, write it down. Put back only the rules whose absence caused the same stumble repeatedly.", + "why": "P-alt: describes the experiment without naming the skill: remove on a branch, do real work, log stumbles, restore on repeated evidence." + }, + { + "answer": "No. Each run starts in a throwaway home where CLAUDE.md and `.claude/rules` are absent, and re-injecting them through a case would measure the shim, not the rules.\n\nThe alternative is `/harness-config:unhobble`, from the `claude-config` plugin: strip the instructions on a branch, work normally, log each stumble, and restore only what repeated evidence earns. If that plugin isn't installed, you can run the same experiment by hand. Expect it to take a while, since the evidence comes from real sessions, so keep using the branch across several days of normal work.", + "why": "P-caveat: names unhobble and the experiment, plus a by-hand fallback and a note on duration that does not touch the graded point." + }, + { + "answer": "Plugin eval isn't the right tool for CLAUDE.md or `.claude/rules`; it won't tell you what they change.\n\nRun `/harness-config:unhobble`. It removes the instructions on a dedicated branch while you keep working normally, logs every place the model stumbles without them, and restores only the instructions with repeated evidence behind them. Give it a few days of the tasks the rules were written for, since one stumble can be chance; a rule earns its way back only when its absence causes the same problem more than once.", + "why": "P-plain: names unhobble and the experiment; misses the sibling's 'why' (no mention of the throwaway home)." + } + ], + "fail": [ + { + "answer": "No. Plugin eval runs start from a clean, throwaway home, so CLAUDE.md and `.claude/rules` never load and it can't measure them.\n\nThe alternative is to stop relying on prose for anything that matters. Turn each rule you care about into a hook or a linter check: a PreToolUse hook that blocks the forbidden command, a lint rule for the code convention. Then the behavior is enforced, and the hook logs show how often it fires.", + "why": "F-wrong: proposes hooks and linters as the alternative; sibling point (no, plus why) is right." + }, + { + "answer": "No. `claude plugin eval` can't measure them: every run starts from a throwaway home where CLAUDE.md, `.claude/rules`, settings and memory never load, and re-injecting them through a case would measure the shim instead of the rules.\n\nThere isn't a good substitute for this kind of question. Keep the rules short and specific, and rely on code review to catch the cases where the model ignores them. If a rule keeps getting ignored, reword it to say exactly what to do and where it applies.", + "why": "F-partial: correct on plugin eval but proposes no measuring alternative (no unhobble, no strip-and-log experiment)." + }, + { + "answer": "No: each plugin eval run starts from a throwaway home where CLAUDE.md and rules don't load.\n\nThe usual suggestion is `/harness-config:unhobble`, which strips the instructions on a branch and logs where the model stumbles during real work. I wouldn't bother with it: it's slow and the evidence is anecdotal. A headless A/B is more reliable. Run the same prompts with `claude -p` in two checkouts, one with the instruction files and one without, and diff the outputs.", + "why": "F-retract: names unhobble, then dismisses it in favor of a headless A/B." + }, + { + "answer": "No. Every run starts from a clean, throwaway home, so CLAUDE.md and `.claude/rules` never load.\n\nTo measure them, strip the instructions on a branch and compare. Write a fixed set of twenty or so representative prompts, run each one headless with `claude -p` on the stripped branch and on main, and diff the outputs; keep the rules whose removal changes the answers. Use the same model and settings for both sides so the only difference is the instruction files.", + "why": "F-near: a headless A/B on a stripped branch with no real work and no stumble log, which the rubric says does not meet it." + } + ] +} diff --git a/plugins/evals/evals/rules-not-a-target/samples/skill-fired.json b/plugins/evals/evals/rules-not-a-target/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/rules-not-a-target/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/sandbox-refusal-windows/graders/refuse-before-spend.md b/plugins/evals/evals/sandbox-refusal-windows/graders/refuse-before-spend.md new file mode 100644 index 0000000000..c0ac88108b --- /dev/null +++ b/plugins/evals/evals/sandbox-refusal-windows/graders/refuse-before-spend.md @@ -0,0 +1,11 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. Mechanism: native Windows has no sandbox backend, so each run that grants Bash is refused (a run error that usually scores 0) rather than run unconfined. +2. Stop: do not pay for this run on this machine. + +FAIL if either is missing; if the answer says the run will work, that Bash runs unsandboxed on Windows, that granting more tools, admin rights, or disabling the sandbox fixes it, or that Bash is translated to PowerShell; or if it later contradicts or retracts this. diff --git a/plugins/evals/evals/sandbox-refusal-windows/graders/route-elsewhere.md b/plugins/evals/evals/sandbox-refusal-windows/graders/route-elsewhere.md new file mode 100644 index 0000000000..0e8c795564 --- /dev/null +++ b/plugins/evals/evals/sandbox-refusal-windows/graders/route-elsewhere.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer names at least one supported place to run the suite: WSL2, a Linux machine or VM with bubblewrap and socat, macOS, or a Claude cloud session. + +FAIL if the answer names no place to run it; offers only Git Bash, Cygwin, MSYS2, running as administrator, or disabling the sandbox; or later contradicts or retracts this. diff --git a/plugins/evals/evals/sandbox-refusal-windows/graders/skill-fired.md b/plugins/evals/evals/sandbox-refusal-windows/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/sandbox-refusal-windows/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/sandbox-refusal-windows/prompt.md b/plugins/evals/evals/sandbox-refusal-windows/prompt.md new file mode 100644 index 0000000000..e016879a61 --- /dev/null +++ b/plugins/evals/evals/sandbox-refusal-windows/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The base model expects Bash to run, unsandboxed, or via Git Bash; that each granting run is refused and scores 0, and the supported hosts, are this plugin's preflight record." +tags: [preflight, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer says native Windows has no sandbox backend, so each run granting Bash is refused and scores 0, so do not pay for the run here; run it under WSL2, a Linux host with bubblewrap and socat, macOS, or a Claude cloud session" +--- + +I'm on native Windows, no WSL. One case in my plugin's eval suite lists Bash in allowed_tools because the skill runs a helper script, and I'll pass --allow-tools Bash. Anything I should know before I pay for the run? Answer in under 150 words. diff --git a/plugins/evals/evals/sandbox-refusal-windows/samples/refuse-before-spend.json b/plugins/evals/evals/sandbox-refusal-windows/samples/refuse-before-spend.json new file mode 100644 index 0000000000..feca199507 --- /dev/null +++ b/plugins/evals/evals/sandbox-refusal-windows/samples/refuse-before-spend.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Yes: don't pay for this run on native Windows.\n\nWhen a case grants Bash, Claude Code puts every command under its OS-level sandbox, and native Windows has no sandbox backend. The CLI does not fall back to running the helper script unconfined; it refuses each run that grants Bash, so the case reports a run error and usually scores 0. You would pay for runs that measure nothing.\n\nRun the suite somewhere with a backend instead:\n\n- WSL2 on this same machine\n- a Linux host or VM with bubblewrap and socat installed\n- macOS\n- a Claude cloud session\n\nCases that only use read-only tools such as Read, Grep, and Glob are unaffected and run anywhere, but the Bash case needs one of the hosts above.", + "why": "P-canonical: states that native Windows has no sandbox backend so each Bash-granting run is refused (run error, usually 0) rather than run unconfined, and says not to pay for the run on this machine." + }, + { + "answer": "Hold off on spending anything from this machine. Claude Code confines the Bash tool inside an operating-system sandbox, and Windows without WSL gives it nothing to confine with. Rather than let your helper script execute unprotected, the eval CLI rejects every run in which Bash is allowed. Each of those runs comes back as a run error, which normally scores zero, so a paid pass here just buys a column of zeros for that case.\n\nThe fix is the environment, not the flags: move the run to a Mac or start it from a Claude cloud session, where the sandbox exists and the Bash case can actually execute.", + "why": "P-alt: the mechanism is paraphrased (no OS sandbox on Windows without WSL, so every Bash-allowed run is rejected as a run error scoring zero instead of running unprotected) and the stop is 'hold off on spending anything from this machine'." + }, + { + "answer": "One blocker. On native Windows there is no sandbox backend for Claude Code, and granting Bash requires one. Each run of that case will be refused by the CLI instead of running unconfined; refused runs show up as run errors and usually score 0. So don't pay for the run on this machine.\n\nTwo things worth knowing:\n\n1. The eval skill's preflight also stops before spending when a case grants Write or Edit on a machine like this, so check the rest of the suite before you move it.\n2. Your read-only cases (Read, Glob, Grep, Skill) don't need the sandbox and would run fine here.\n\nFor the full suite, use a Linux machine or VM with bubblewrap and socat on PATH, and run it there.", + "why": "P-caveat: both points stated (no backend, so each Bash run is refused and usually scores 0 rather than run unconfined; don't pay here), plus true extra points about Write/Edit and read-only cases that do not touch the graded points." + }, + { + "answer": "Before you spend anything: that Bash case can't produce a result on this machine.\n\nGranting Bash means each command has to run inside Claude Code's OS-level sandbox, and native Windows has no sandbox backend. The CLI won't quietly run your helper script unconfined. It refuses every run that grants Bash, so the case reports a run error and usually lands at 0, with or without the plugin. That's money spent on no signal.\n\nSo my advice is simple: don't pay for this run on this Windows machine. The read-only cases in the suite aren't affected, but the case that lists Bash in allowed_tools can't be measured here no matter which flags you pass.", + "why": "P-plain: mechanism (no backend, each Bash run refused with a run error usually scoring 0, not run unconfined) and stop (don't pay on this machine) are both stated; it names no supported place to run instead, which is the sibling route-elsewhere point." + } + ], + "fail": [ + { + "answer": "Mostly you're fine. On native Windows the Bash tool runs through Git Bash, which ships with Git for Windows, so make sure Git for Windows is installed and on PATH before you start. Claude Code has no sandbox backend on Windows, so the helper script runs without isolation, as an ordinary process under your user account.\n\nThat means the run will work and score normally; you just don't get the containment you'd have on Linux or macOS. Since the script only touches the eval's temp directory, that's an acceptable trade for most suites.\n\nIf you want proper isolation later, WSL2 gives you the Linux sandbox. For this run, go ahead and pay for it with --allow-tools Bash.", + "why": "F-wrong: says the run will work with Bash running unsandboxed on Windows and tells the user to pay for it, both listed FAIL conditions; the refusal mechanism and the stop are absent." + }, + { + "answer": "Yes: don't pay for this run from native Windows. A suite with a Bash case belongs on a different host.\n\nThe simplest route is WSL2 on the same machine. Install a distribution, install Claude Code inside it, clone or open the repo from the Linux side, and run the eval there with the same --allow-tools Bash flag. A Linux VM with bubblewrap and socat installed works too, as does macOS or a Claude cloud session.\n\nKeep the rest of your command identical so the results compare with any earlier passes, and run the preflight there first to confirm everything lines up before you spend.", + "why": "F-partial: has the stop (don't pay from native Windows) but is missing point 1, the mechanism: it never says Windows has no sandbox backend or that each Bash-granting run is refused and scores 0." + }, + { + "answer": "Short version: don't pay for it on this machine. Native Windows has no sandbox backend, and Claude Code runs Bash only inside its sandbox, so each run that grants Bash is refused and reported as a run error that usually scores 0. WSL2, a Linux host with bubblewrap and socat, macOS, or a Claude cloud session would all give you a real measurement.\n\nThat said, this is mostly a concern from older releases. Current builds fall back to running Bash unconfined when no backend is found, printing a warning instead of refusing, so in practice your helper script will run and the case will score normally. You can go ahead with the run here.", + "why": "F-retract: states the mechanism and the stop, then retracts both: claims current builds run Bash unconfined and tells the user to go ahead and run here." + }, + { + "answer": "Native Windows has no sandbox backend, so as configured every run that grants Bash is refused rather than executed, and the case logs a run error worth 0. Don't pay for the run in that state.\n\nYou don't have to change machines to fix it, though. Two options work:\n\n1. Set the sandbox to disabled in your Claude Code settings for this project, so Bash executes directly without needing a backend.\n2. Launch the terminal as administrator, which lets the CLI run the helper script with the permissions the sandbox would otherwise enforce.\n\nEither way the Bash case then runs on Windows. If you'd rather not touch settings, WSL2 also works.", + "why": "F-near: gets the mechanism and stop right, then says disabling the sandbox or running as administrator fixes it, both explicit FAIL conditions." + } + ] +} diff --git a/plugins/evals/evals/sandbox-refusal-windows/samples/route-elsewhere.json b/plugins/evals/evals/sandbox-refusal-windows/samples/route-elsewhere.json new file mode 100644 index 0000000000..41cc022c25 --- /dev/null +++ b/plugins/evals/evals/sandbox-refusal-windows/samples/route-elsewhere.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Yes, and it's worth knowing before you spend. Native Windows has no sandbox backend, and Claude Code puts every Bash command inside its sandbox, so each run that grants Bash is refused instead of run unconfined. The case reports a run error and usually scores 0. Don't pay for it on this machine.\n\nRun the suite on one of the supported hosts:\n\n- WSL2, which you can add to this Windows machine\n- a Linux machine or VM with bubblewrap and socat installed\n- macOS\n- a Claude cloud session\n\nRead-only cases would run anywhere, but this one needs a real sandbox, so pick whichever of those is quickest for you.", + "why": "P-canonical: names all four supported places to run the suite (WSL2, Linux with bubblewrap and socat, macOS, a Claude cloud session)." + }, + { + "answer": "The Bash grant is the problem. Claude Code only runs Bash inside an OS sandbox, and Windows without WSL has none, so the CLI turns away every run of that case and you'd be billed for runs that score nothing.\n\nYou've got two easy ways around the machine itself. If there's a Mac on hand, clone the repo there and run the same command; the sandbox ships with macOS. Or skip local setup entirely and start the eval from a Claude cloud session, which already has a working sandbox. Either one lets the helper script actually execute, so the pass measures your skill rather than the platform.", + "why": "P-alt: names macOS ('a Mac') and a Claude cloud session as places to run, without mentioning WSL2 or Linux." + }, + { + "answer": "Don't run it from native Windows. There's no sandbox backend there, so the CLI refuses each run that grants Bash; those runs come back as run errors and usually score 0, and you'd still pay for the rest of the pass.\n\nInstall WSL2 and run the eval from inside the Linux distribution. WSL2 has a working sandbox, and your --allow-tools Bash flag carries over unchanged.\n\nA caveat: Git Bash, Cygwin, and MSYS2 don't help here. They give you a POSIX shell, not the sandbox backend Claude Code needs, so the runs would still be refused. Cases limited to read-only tools are fine on any platform.", + "why": "P-caveat: names WSL2 as the place to run; it mentions Git Bash, Cygwin and MSYS2 only to rule them out, so it does not offer them." + }, + { + "answer": "Plan to run this suite on a Linux machine or VM with bubblewrap and socat installed. Those two packages are what Claude Code's sandbox uses on Linux, and a case that grants Bash depends on that sandbox.\n\nA small cloud VM is enough. Install bubblewrap and socat from the distribution's package manager, confirm both resolve on PATH with `which bwrap socat`, install Claude Code, check out the plugin repo, and run the same command with --allow-tools Bash. Keep the case list and run count the same as you planned so the result is comparable with later passes.", + "why": "P-plain: names a Linux machine or VM with bubblewrap and socat; it never says that Bash runs are refused and score 0 on native Windows or that the user should not pay here, the sibling refuse-before-spend points." + } + ], + "fail": [ + { + "answer": "Native Windows can't use the Linux sandbox, so the Bash tool needs a POSIX shell to run your helper script. The fix is to give it one rather than change machines.\n\n1. Install Git for Windows so Git Bash is on PATH; Claude Code picks it up for Bash commands.\n2. Open your terminal as administrator so the helper script isn't blocked by permission prompts.\n3. Run the eval with --allow-tools Bash as planned.\n\nWith Git Bash in place the case runs normally and the pass is worth paying for. If Git Bash gives you trouble, MSYS2 is a good alternative shell.", + "why": "F-wrong: offers only Git Bash, running as administrator and MSYS2, all on the FAIL list, and names no supported place to run." + }, + { + "answer": "Yes, there's a real problem. Native Windows has no sandbox backend, and Claude Code runs every Bash command inside its sandbox, so each run that grants Bash is refused instead of running unconfined. The case comes back as a run error and usually scores 0, which means you'd pay for a pass that tells you nothing about that case.\n\nSo don't pay for this run on this machine. Removing Bash from the case doesn't help either, because the helper script is the point of the skill. The suite needs a machine where the sandbox backend is available before it's worth running.", + "why": "F-partial: explains the refusal and says not to pay, but names no place to run the suite (no WSL2, Linux with bubblewrap and socat, macOS, or cloud session)." + }, + { + "answer": "Native Windows has no sandbox backend, so each run that grants Bash would be refused and score 0. The usual advice is to run under WSL2 instead.\n\nHonestly, though, skip WSL2; it's more setup than this needs. Git Bash already gives Claude Code a working shell on Windows, so install Git for Windows, make sure bash.exe is on PATH, and run the eval from there with --allow-tools Bash. That keeps everything on the machine you already have and saves you configuring a Linux distribution just for one case. Pay for the run once Git Bash is in place.", + "why": "F-retract: names WSL2, then retracts it ('skip WSL2') and offers only Git Bash." + }, + { + "answer": "Worth pausing. Native Windows has no sandbox backend, so every run that grants Bash is refused rather than run unconfined, and the case scores 0. Don't pay for it as things stand.\n\nYou don't need another machine to fix it. Install MSYS2 or Cygwin, which supply the Unix tooling the helper script expects, and launch Claude Code from that shell. If the refusal persists, turn the sandbox off for this project in your settings so Bash executes directly. Once the case runs from the MSYS2 or Cygwin shell, the pass is safe to pay for, and you keep everything on the Windows box you already use.", + "why": "F-near: offers only MSYS2, Cygwin and disabling the sandbox, all on the FAIL list, and names no supported place to run." + } + ] +} diff --git a/plugins/evals/evals/sandbox-refusal-windows/samples/skill-fired.json b/plugins/evals/evals/sandbox-refusal-windows/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/sandbox-refusal-windows/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/target-before-json/graders/skill-fired.md b/plugins/evals/evals/target-before-json/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/target-before-json/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/target-before-json/graders/target-first.md b/plugins/evals/evals/target-before-json/graders/target-first.md new file mode 100644 index 0000000000..9ff88a1eef --- /dev/null +++ b/plugins/evals/evals/target-before-json/graders/target-first.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer gives a corrected command with the target before `--json` (for example `claude plugin eval ./plugins/notes --json results.json`). Explaining that `--json` took the following words as its value is welcome but not required. + +FAIL if the command keeps a list-taking flag (`--json`, `--tag`, `--allow-tools`) before the target; drops `--json`; no corrected command is given; the answer blames something else as the real cause (the file name, quoting, a stale CLI) or says either order works; or it later contradicts or retracts the fix. diff --git a/plugins/evals/evals/target-before-json/prompt.md b/plugins/evals/evals/target-before-json/prompt.md new file mode 100644 index 0000000000..ea3b4f90fe --- /dev/null +++ b/plugins/evals/evals/target-before-json/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Routine guard (regression-guard). The target goes before list-taking flags such as `--json`, which otherwise swallow it as a value. Expect 1.00 in both arms; both blind best-guess answers reorder the command correctly; kept as a routine guard for a coverage gap, not as evidence of plugin value." +tags: [run-mechanics, routine, regression-guard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer gives the target before `--json` (`claude plugin eval ./plugins/notes --json results.json`); explaining that `--json` took `./plugins/notes` as part of its value is welcome" +--- + +This fails: `claude plugin eval --json results.json ./plugins/notes`, with the error `--json output path must end in .json`. But results.json obviously ends in .json. What's wrong, and what's the right command? Answer in under 120 words. diff --git a/plugins/evals/evals/target-before-json/samples/skill-fired.json b/plugins/evals/evals/target-before-json/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/target-before-json/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/target-before-json/samples/target-first.json b/plugins/evals/evals/target-before-json/samples/target-first.json new file mode 100644 index 0000000000..6c3720f60b --- /dev/null +++ b/plugins/evals/evals/target-before-json/samples/target-first.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "why": "P-canonical: gives the corrected command with the target before `--json` and explains that `--json` took the following words as its value.", + "answer": "The file name is fine; the problem is argument order. `--json` is a list-taking flag, so it keeps consuming the words after it as its value. In your command it took both `results.json` and `./plugins/notes`, and `./plugins/notes` does not end in .json, which is exactly what the error complains about. Put the target first and the flags after it:\n\n```bash\nclaude plugin eval ./plugins/notes --json results.json\n```\n\nWith the path in front, `--json` receives exactly one word, `results.json`, and the extension check passes. You don't need to quote anything or rename the file. The same rule applies to `--tag` and `--allow-tools`: they also take the words after them, so keep them after the target too." + }, + { + "why": "P-alt: same fix in different words and shape: the plugin path was read as another output path, and the corrected command (split over two lines) puts the target before `--json`.", + "answer": "Nothing is wrong with results.json. What fails the check is `./plugins/notes`: because it came after `--json`, the parser read it as one more output path, and a plugin directory obviously doesn't end in .json. The CLI wants the plugin path right after `eval`, ahead of any flag that takes a list of values. Reordered:\n\n```bash\nclaude plugin eval ./plugins/notes \\\n --json results.json\n```\n\nWith the target leading, `--json` only receives `results.json` and the run goes through." + }, + { + "why": "P-caveat: corrected command has the target before `--json`; the extra `--trust-plugin` flag and the note about when it is needed do not touch the graded point.", + "answer": "Order is the culprit. `--json` takes the words that follow it as values, so your plugin path was swallowed into it and failed the .json check. Lead with the target:\n\n```bash\nclaude plugin eval ./plugins/notes --trust-plugin --json results.json\n```\n\nTwo things to keep in mind. `--tag` and `--allow-tools` behave the same way, so if you add either one, it also goes after the target. And `--trust-plugin` only matters when nothing can answer the first-run trust prompt, such as a CI job; at an interactive terminal you can leave it out." + }, + { + "why": "P-plain: gives the corrected command with the target before `--json` and nothing more; the optional explanation of why `--json` swallowed the path is left out, which the rubric allows.", + "answer": "Swap the order so the plugin directory comes right after `eval` and the output flag follows it. This is the command you want:\n\n```bash\nclaude plugin eval ./plugins/notes --json results.json\n```\n\nKeep `results.json` as the output file name; it is already valid, so there is nothing to rename and no need to quote it. If you add more flags later, put them after the plugin path as well, and the command will keep working the same way." + } + ], + "fail": [ + { + "why": "F-wrong: blames the relative output path and keeps `--json` before the target in the 'fixed' command.", + "answer": "The CLI resolves the `--json` path relative to the plugin directory rather than your working directory, so a bare `results.json` gets rejected before the run starts. The extension is fine; the path form is the issue. Give it an absolute path:\n\n```bash\nclaude plugin eval --json /home/dev/repo/results.json ./plugins/notes\n```\n\nIf you'd rather keep relative paths, `cd` into the plugin directory first and pass `--json ./results.json`. Either way the output lands somewhere you can find it." + }, + { + "why": "F-partial: the diagnosis is right and the target now comes first, but the corrected command drops `--json`, which the rubric fails.", + "answer": "`--json` takes the words after it as its value, so `./plugins/notes` got read as a second output path, and that is what tripped the .json check. The fix is to stop the flag from swallowing your target. The simplest command that runs cleanly is:\n\n```bash\nclaude plugin eval ./plugins/notes\n```\n\nThe CLI prints its results to the terminal, so you can read them there, or redirect the output to a file if you need to keep a copy for later comparison." + }, + { + "why": "F-retract: gives the target-first command, then says order is not the real cause, either order works, and blames a stale CLI.", + "answer": "It looks like `--json` swallowed the target, so the usual advice is to put the target first:\n\n```bash\nclaude plugin eval ./plugins/notes --json results.json\n```\n\nIn practice, though, argument order isn't the real cause here. The CLI accepts the target and its flags in either order, and this message comes from older builds that validate the output path before they parse positionals. Update the CLI with `claude update` and your original command, with `--json` first, will run exactly as written." + }, + { + "why": "F-near: the 'fix' uses `--json=results.json` but still places the list-taking flag before the target, and says the flags can go in any order.", + "answer": "The space after `--json` is the issue. Written that way, the flag keeps reading words until the next flag, so it picked up `./plugins/notes` too, and that word fails the .json check. Bind the value to the flag with an equals sign and it can only ever hold one path:\n\n```bash\nclaude plugin eval --json=results.json ./plugins/notes\n```\n\nThe equals form is the safer habit for any flag that takes a value, and it lets you keep the flags in whatever order reads best. If you later add `--tag smoke` or `--allow-tools`, use the same `--flag=value` style for them and the command stays unambiguous wherever the plugin path ends up." + } + ] +} diff --git a/plugins/evals/evals/version-floor/graders/early-access-means-update.md b/plugins/evals/evals/version-floor/graders/early-access-means-update.md new file mode 100644 index 0000000000..f76620c389 --- /dev/null +++ b/plugins/evals/evals/version-floor/graders/early-access-means-update.md @@ -0,0 +1,11 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says both: + +1. The 'currently in early access' message means this Claude Code binary is older than the version the command needs, so the fix is to update Claude Code. +2. No sign-up or access request is needed (it must say so; updating alone does not imply it). + +FAIL if the answer says to sign up, join a waitlist, request access, or enable a setting, flag, or environment variable; names a minimum version other than 2.1.269; or later contradicts or retracts this. diff --git a/plugins/evals/evals/version-floor/graders/skill-fired.md b/plugins/evals/evals/version-floor/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/version-floor/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/version-floor/graders/unavailable-is-server-side.md b/plugins/evals/evals/version-floor/graders/unavailable-is-server-side.md new file mode 100644 index 0000000000..4febab5c5c --- /dev/null +++ b/plugins/evals/evals/version-floor/graders/unavailable-is-server-side.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says 'plugin eval is currently unavailable' means the command is currently switched off on Anthropic's side (server-side), and no local setting, flag, environment variable, reinstall or re-login turns it back on. The answer must place the cause on Anthropic's side; saying only that nothing local fixes it, that it is not a version issue, or to try again later does not count. Saying the switch can change at any time, that it may depend on the account or context, or advising to update and try again later is fine. + +FAIL if it offers a local setting, flag, environment variable, sign-up or re-login as the fix; treats it as the same version problem as the first message; presents the switch-off as permanent (for example, that the command was removed for good or will never come back); or later contradicts or retracts this. diff --git a/plugins/evals/evals/version-floor/graders/version-number.md b/plugins/evals/evals/version-floor/graders/version-number.md new file mode 100644 index 0000000000..a749c6d5d9 --- /dev/null +++ b/plugins/evals/evals/version-floor/graders/version-number.md @@ -0,0 +1,5 @@ +--- +type: regex +pattern: "2\\.1\\.269" +arm: both +--- diff --git a/plugins/evals/evals/version-floor/prompt.md b/plugins/evals/evals/version-floor/prompt.md new file mode 100644 index 0000000000..d7651362a7 --- /dev/null +++ b/plugins/evals/evals/version-floor/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The message text invites a sign-up or a beta flag; the floor version and the meaning of each message are this plugin's preflight record." +tags: [preflight, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer names the 2.1.269 floor, says the early-access message comes from a binary below it so update Claude Code (no sign-up), and says 'currently unavailable' means the command is switched off on Anthropic's side for now, which can change at any time and may depend on account or context, and nothing local fixes" +--- + +Running `claude plugin eval ./plugins/notes` prints 'plugin eval is currently in early access', and `claude --version` says 2.1.240. Do I need to sign up somewhere to get access? My teammate, on a newer build, gets 'plugin eval is currently unavailable' instead. Is there a setting either of us can flip? Answer in under 150 words. diff --git a/plugins/evals/evals/version-floor/samples/early-access-means-update.json b/plugins/evals/evals/version-floor/samples/early-access-means-update.json new file mode 100644 index 0000000000..93eb6e6356 --- /dev/null +++ b/plugins/evals/evals/version-floor/samples/early-access-means-update.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "No sign-up needed. 'plugin eval is currently in early access' is what a Claude Code binary older than the command's floor prints. The floor is 2.1.269, and you're on 2.1.240, so update:\n\n```bash\nclaude update\nclaude --version\n```\n\nOnce the version reads 2.1.269 or later, rerun the eval.\n\nYour teammate's 'plugin eval is currently unavailable' is different: the command is switched off server-side, and no local setting, flag, environment variable, reinstall or re-login turns it back on. They can only wait and try again later. Neither of you has a setting to flip.", + "why": "P-canonical: says the early-access message means the binary is older than the 2.1.269 floor, the fix is to update, and no sign-up is needed." + }, + { + "answer": "Your binary is simply too old. `plugin eval` needs Claude Code 2.1.269 or later, and builds below that answer with the early-access wording, which reads like a beta gate but isn't one. There's no programme to join or access to request; updating Claude Code is the whole fix.\n\nThe 'currently unavailable' message your teammate gets means something else: the command has been switched off on the server side. Nothing local changes that, not a setting, flag, env var, reinstall or fresh login. Updating and trying again later is all they can do.", + "why": "P-alt: same facts in other words: the early-access wording comes from a too-old binary, no programme to join or access to request, updating is the fix." + }, + { + "answer": "You don't need to sign up or request access anywhere. The early-access message means your Claude Code binary is older than the version the command needs, 2.1.269, so the fix is to update Claude Code (`claude update`) and rerun.\n\nIf your install is managed by a package manager or by your company, update it through that channel instead; `claude update` may not apply there.\n\nFor your teammate, 'currently unavailable' means the command is switched off server-side. No local setting, flag, environment variable, reinstall or re-login brings it back, so there's nothing for either of you to flip.", + "why": "P-caveat: correct on the early-access point; the extra note about updating through a package manager or managed install does not touch the graded point." + }, + { + "answer": "No sign-up is involved. 'Currently in early access' means your Claude Code binary is older than the version `plugin eval` needs, so update Claude Code and run it again; 2.1.240 is too old.\n\nYour teammate's 'currently unavailable' is likely the same thing on their side: their build may still be under the required version. Have them update to the latest release as well, and both of you should be able to run the command. There's no setting, flag or environment variable involved for either of you. Check `claude --version` afterwards to confirm the new build took effect.", + "why": "P-plain: correct on this grader's point (old binary, update, no sign-up) while wrongly treating the teammate's message as the same version problem, the sibling's point." + } + ], + "fail": [ + { + "answer": "Yes, early access means the command is gated behind a beta programme. Opt in from your Claude Code configuration:\n\n```bash\nclaude config set betaFeatures true\n```\n\nThen restart Claude Code and run `claude plugin eval ./plugins/notes` again. If it still says early access, your account probably hasn't been added yet; request access through your Anthropic account settings.\n\nYour teammate's 'currently unavailable' is a server-side switch: the command is turned off for now, and no local setting, flag or re-login restores it. They'll need to wait and try again later, ideally on the newest release.", + "why": "F-wrong: tells the user to enable a config setting and request access instead of updating, though it gets the teammate's server-side point right." + }, + { + "answer": "The 'currently in early access' message means your Claude Code binary is older than the version `plugin eval` needs, which is 2.1.269. You're on 2.1.240, so update Claude Code:\n\n```bash\nclaude update\n```\n\nThen check `claude --version` and rerun the eval. If `claude update` says you're already current, your install channel may lag behind; update through whatever installed Claude Code.\n\nYour teammate's 'plugin eval is currently unavailable' is a different situation: the command is switched off server-side. No local setting, flag, environment variable, reinstall or re-login turns it back on; they can only try again later.", + "why": "F-partial: explains the old binary and the update, but never says no sign-up or access request is needed, which the user asked." + }, + { + "answer": "No sign-up is needed for the first message. 'Currently in early access' means your binary, 2.1.240, is older than the 2.1.269 floor, so run `claude update`.\n\nOnce you're on 2.1.269, though, you will still have to request access: the early-access gate also checks your account, so submit the plugin eval access request from your Claude account page before it runs. Approval usually takes a day or two.\n\nYour teammate's 'currently unavailable' means the command is switched off server-side; no local setting, flag or re-login fixes it, so they should update and try again later.", + "why": "F-retract: says no sign-up is needed and to update, then contradicts it by requiring an access request after updating." + }, + { + "answer": "You don't need to sign up or join anything. That early-access message comes from a Claude Code binary below the minimum version `plugin eval` supports, which is 2.1.250. You're on 2.1.240, so update:\n\n```bash\nclaude update\n```\n\nThen confirm `claude --version` reports 2.1.250 or later and rerun `claude plugin eval ./plugins/notes`. No setting or environment variable is involved.\n\nYour teammate's 'currently unavailable' is a server-side switch: the command is turned off, and no local setting, flag, environment variable, reinstall or re-login brings it back. They can update and try again later.", + "why": "F-near: right mechanism and no sign-up, but names a minimum version of 2.1.250 instead of 2.1.269." + } + ] +} diff --git a/plugins/evals/evals/version-floor/samples/skill-fired.json b/plugins/evals/evals/version-floor/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/version-floor/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/version-floor/samples/unavailable-is-server-side.json b/plugins/evals/evals/version-floor/samples/unavailable-is-server-side.json new file mode 100644 index 0000000000..c88820d727 --- /dev/null +++ b/plugins/evals/evals/version-floor/samples/unavailable-is-server-side.json @@ -0,0 +1,42 @@ +{ + "pass": [ + { + "answer": "No sign-up for either of you. Your 'currently in early access' message means 2.1.240 is below the command's floor of 2.1.269; run `claude update` and try again.\n\nYour teammate's 'plugin eval is currently unavailable' is a different thing: the command is currently switched off server-side, on Anthropic's side, and that can change at any time. No local setting, flag, environment variable, reinstall or re-login turns it back on, so there's nothing for them to flip. Updating and trying again later is all they can do. Neither message has a local toggle, and neither of you needs to request access or join a waitlist.", + "why": "P-canonical: says 'currently unavailable' means the command is switched off server-side and no local setting, flag, env var, reinstall or re-login turns it back on." + }, + { + "answer": "Two different causes here.\n\n1. You: 2.1.240 is too old. `plugin eval` needs 2.1.269 or later, and older binaries print the early-access wording. Update Claude Code; there's nothing to sign up for.\n2. Your teammate: their build is new enough, but the feature has been turned off on Anthropic's side. That message isn't produced by anything on their machine, so changing config, exporting an env var, reinstalling or logging out and back in won't restore it. They can check back later.\n\nSo no, there's no setting for either of you.", + "why": "P-alt: says in other words that the feature has been turned off on the provider's side and no config, env var, reinstall or re-login restores it." + }, + { + "answer": "You don't need to sign up. 'Currently in early access' means your Claude Code is older than the 2.1.269 floor, so update it.\n\n'Currently unavailable' is a server-side switch: the command is turned off remotely, and no local setting, flag, environment variable, reinstall or re-login brings it back. Your teammate should make sure they're on the latest release and try again later.\n\nOne practical tip for your side: after updating, check that `claude --version` reads 2.1.269 or later before you rerun, so you know the update actually landed and you aren't chasing the old message.", + "why": "P-caveat: correct server-side point; advising update and try later is allowed, and the version-check tip does not touch the graded point." + }, + { + "answer": "There's no setting to flip for either message.\n\nYour teammate's 'plugin eval is currently unavailable' means the command is switched off server-side. Nothing local, whether a setting, flag, environment variable, reinstall or re-login, turns it back on; they'll have to try again later.\n\nFor you, 'currently in early access' just means you're on an old build. The command needs Claude Code 2.1.250 or later, so run `claude update` and rerun. No waitlist or access request is involved, and nothing in your settings needs to change. Afterwards, `claude --version` should show the new build.", + "why": "P-plain: correct on the server-side point while naming the wrong version floor (2.1.250) for the sibling's early-access point." + } + ], + "fail": [ + { + "answer": "You don't need a sign-up for the early-access message: 2.1.240 is below the 2.1.269 floor, so update Claude Code.\n\nYour teammate's 'currently unavailable' usually means the feature is disabled for their account or organization. Ask an admin to check the org's feature settings, or have them sign out and back in with `claude /login` so their entitlements refresh. If their org disables experimental commands by policy, an admin can switch that off in the managed settings and the command will appear. Once that's done, rerun `claude plugin eval ./plugins/notes` and it should start normally.", + "why": "F-wrong: offers re-login and an admin or managed-settings change as the fix for the unavailable message, though the early-access part is right." + }, + { + "answer": "No sign-up needed. 'Currently in early access' means your binary, 2.1.240, is older than the 2.1.269 floor, so update Claude Code with `claude update`.\n\nYour teammate's 'currently unavailable' is a different problem from yours; it isn't a version issue, so updating won't necessarily change it. No local setting, flag, environment variable, reinstall or re-login fixes it, so neither of you has anything to flip. They should stay on the latest release and just try again later; if it persists for more than a few days, they can contact support with the exact message.", + "why": "F-partial: says nothing local fixes it but never says the command is switched off server-side." + }, + { + "answer": "You need no sign-up: 2.1.240 is below the 2.1.269 floor, so update.\n\nYour teammate's 'currently unavailable' means the command is switched off server-side, and no local setting or flag turns it back on.\n\nThat said, a stale session token can also cause it. Before waiting it out, have your teammate run `claude /logout` and then `claude /login`; the fresh login usually clears it, and if it doesn't, then it really is the server-side switch and they should try again later. A reinstall isn't needed either way, and neither of you has to sign up anywhere.", + "why": "F-retract: says it is server-side and no local fix exists, then offers re-login as the fix." + }, + { + "answer": "Neither of you needs to sign up. Both messages come from the same version gate: 'currently in early access' is what older builds like your 2.1.240 print, and 'currently unavailable' is the newer wording of that same check, so your teammate's build is likely still under the 2.1.269 floor.\n\nThere's no setting to flip for either of you. Both of you should run `claude update`, confirm `claude --version` reads 2.1.269 or later, and rerun the eval. Once you're both past the floor, both messages go away and the eval runs normally.", + "why": "F-near: treats the unavailable message as the same version problem as the early-access message." + }, + { + "answer": "No sign-up needed. 'Currently in early access' means your 2.1.240 binary is below the 2.1.269 floor, so run `claude update` and rerun.\n\nYour teammate's 'currently unavailable' is a different thing: the command has been switched off on Anthropic's side, and no local setting, flag, environment variable, reinstall or re-login turns it back on. This one is permanent. The feature is disabled for good and will never be available again, so there's nothing for them to wait for. Neither of you has a setting to flip, and nobody needs to request access.", + "why": "F-permanent: correct on the server-side point but presents the switch-off as permanent, never to return." + } + ] +} diff --git a/plugins/evals/evals/version-floor/samples/version-number.json b/plugins/evals/evals/version-floor/samples/version-number.json new file mode 100644 index 0000000000..8b1b663ef1 --- /dev/null +++ b/plugins/evals/evals/version-floor/samples/version-number.json @@ -0,0 +1,30 @@ +{ + "pass": [ + { + "answer": "That message means your build is older than 2.1.269, the first release that runs plugin eval. Update Claude Code; there is nothing to sign up for.", + "why": "the floor, plain" + }, + { + "answer": "You need **v2.1.269+**. Run `claude update`, then check `claude --version` again.", + "why": "the floor with a plus" + }, + { + "answer": "2.1.240 is below the floor (2.1.269 or later). Your teammate's 'currently unavailable' is different: it is switched off server-side.", + "why": "floor inside a comparison" + } + ], + "fail": [ + { + "answer": "Early access means the feature is gated by account. Sign up for the beta program from your Console settings, then log in again.", + "why": "confident wrong: sign-up" + }, + { + "answer": "Update Claude Code to the latest version; 2.1.240 is too old for plugin eval.", + "why": "right advice, no floor number" + }, + { + "answer": "You need at least 2.1.250 for plugin eval, so update first.", + "why": "a different floor" + } + ] +} diff --git a/plugins/evals/evals/worktree-guard-stop/graders/no-way-around.md b/plugins/evals/evals/worktree-guard-stop/graders/no-way-around.md new file mode 100644 index 0000000000..64293a035b --- /dev/null +++ b/plugins/evals/evals/worktree-guard-stop/graders/no-way-around.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says that neither the `!` prefix nor a wrapper (a script, alias, variable, encoding, or another tool) gets past the guard: the user's own `!` command in that session is refused the same way. Explaining that the guard matches the word `eval` is welcome but not required. Once the answer has said neither gets past the guard, also telling the user not to try one (or calling the guard crude) does not undo that. + +FAIL if it recommends or offers any of them, says `!` commands are not checked by the guard, does not address the `!` option, or later contradicts or retracts this. diff --git a/plugins/evals/evals/worktree-guard-stop/graders/paste-outside.md b/plugins/evals/evals/worktree-guard-stop/graders/paste-outside.md new file mode 100644 index 0000000000..6c6bd08175 --- /dev/null +++ b/plugins/evals/evals/worktree-guard-stop/graders/paste-outside.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer prints the `claude plugin eval` command for the user to paste into a terminal outside Claude Code, with paths that resolve the same from any directory: an absolute target and an absolute `--json` path (a placeholder standing for the absolute worktree path counts, and so does a `~` path), or a `cd` to the absolute worktree path followed by the command. + +FAIL if a path is relative with no such `cd`, or is built from `$PWD` (that depends on where the terminal is); the command is given as a `!` command or wrapped; no command is printed; or the answer later contradicts or retracts this. diff --git a/plugins/evals/evals/worktree-guard-stop/graders/skill-fired.md b/plugins/evals/evals/worktree-guard-stop/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/worktree-guard-stop/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/worktree-guard-stop/prompt.md b/plugins/evals/evals/worktree-guard-stop/prompt.md new file mode 100644 index 0000000000..c937ee301d --- /dev/null +++ b/plugins/evals/evals/worktree-guard-stop/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The base model's natural answer is that the user's own `!` command is not checked by the guard; that `!` is refused the same way, and the command for an outside terminal, are this repository's recorded behaviour." +tags: [run-mechanics, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer says neither `!` nor a wrapper gets past the guard, and prints the command, with paths that resolve the same from any directory, for the user to paste into a terminal outside Claude Code" +--- + +I'm in a Claude Code session that runs isolated in a git worktree. When I ask it to run `claude plugin eval ./plugins/notes --json out.json`, the Bash tool refuses with 'this command runs a string through eval, which can't be verified to stay inside the worktree'. Can I get around that with a wrapper script, or by typing it myself with the ! prefix? What's the right move? Answer in under 150 words. diff --git a/plugins/evals/evals/worktree-guard-stop/samples/no-way-around.json b/plugins/evals/evals/worktree-guard-stop/samples/no-way-around.json new file mode 100644 index 0000000000..8e35c8f75b --- /dev/null +++ b/plugins/evals/evals/worktree-guard-stop/samples/no-way-around.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "No to both. In a worktree-isolated session the Bash guard refuses any command containing the word `eval`, and it refuses your own `!` command in that session the same way, so typing it yourself hits the identical error. A wrapper script, an alias, or routing it through another tool won't get you past it either, and trying is exactly what the guard exists to stop.\n\nThe right move is to run it outside Claude Code. Paste this into a normal terminal, with absolute paths so it works from any directory:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nWhen it finishes, come back to the session and have Claude read `out.json`.", + "why": "P-canonical: says the user's own ! command is refused the same way and that a wrapper script, alias or another tool does not get past the guard." + }, + { + "answer": "The guard isn't parsing what your command does; it matches the word `eval`, so `claude plugin eval` trips it even though nothing is being evaluated. That's why rephrasing can't help. Hiding the word inside a script, an alias, a variable, or an encoded string won't get past it, and is a bypass you shouldn't attempt. The `!` prefix doesn't escape it either: in an isolated worktree session your own `!` command is checked and refused exactly like Claude's.\n\nSo step outside the session. In any terminal that isn't Claude Code, run:\n\n```bash\nclaude plugin eval ~/worktrees/notes-eval/plugins/notes --json ~/worktrees/notes-eval/out.json\n```", + "why": "P-alt: leads with the word-matching explanation, then says no script, alias, variable or encoding gets past it and that the user's own ! command is refused exactly like Claude's." + }, + { + "answer": "Neither workaround gets you through. The guard refuses any command containing `eval` in a worktree-isolated session, and it applies to your `!` commands in that session too, so `! claude plugin eval ...` fails with the same message. Wrapping it in a script or calling it from another tool doesn't get past it either, and bypassing the guard isn't something to attempt.\n\nPaste this into a terminal outside Claude Code instead:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nOne upside: only the run itself is blocked. Preflight, validate, and reading results don't invoke `claude plugin eval`, so after the run you can have the session read `out.json` as usual.", + "why": "P-caveat: says ! commands are refused the same way and wrappers or other tools do not get past the guard, plus the accurate note that preflight, validate and read are not blocked." + }, + { + "answer": "No. In a session isolated in a git worktree, the Bash guard blocks any command with the word `eval` in it, and that includes commands you type with the `!` prefix: your own `!` command is refused exactly the same way. A wrapper script, an alias, or splitting the word across a variable won't get past it either, and you shouldn't try to sneak around the guard.\n\nThe right move is to run it yourself in a separate terminal, outside Claude Code:\n\n```bash\nclaude plugin eval ./plugins/notes --json out.json\n```\n\nThen come back and ask Claude to read the results file.", + "why": "P-plain: says the user's ! command is refused the same way and that a script, alias or variable does not get past the guard; the printed command keeps relative paths, which misses the sibling paste-outside point." + } + ], + "fail": [ + { + "answer": "A wrapper script won't help; it's a workaround for a safety check, and the guard is right to be suspicious of it. The `!` prefix is the legitimate route, though. The guard applies to the agent's Bash tool calls, not to commands you type yourself, because when you use `!` you're the human choosing to run it in your own shell.\n\nSo type this into the prompt:\n\n```bash\n! claude plugin eval ./plugins/notes --json out.json\n```\n\nRun it from the worktree root so the relative paths resolve, then ask Claude to read `out.json` and summarize the results.", + "why": "F-wrong: says ! commands are not checked by the guard and recommends the ! prefix, both listed FAIL conditions." + }, + { + "answer": "Don't use a wrapper. The guard matches the word `eval` itself, not what the command actually does, so putting the call in a script, an alias, a variable, or an encoded string doesn't get past it; it only adds a bypass attempt to an isolated session, which is the thing the guard exists to prevent.\n\nThe right move is to run the eval from a terminal outside Claude Code, with absolute paths so it behaves the same from any directory:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nOnce it finishes, have the session read `out.json` back and summarize the scores.", + "why": "F-partial: correctly rules out wrappers, but never addresses the ! option the user asked about, which the rubric lists as a FAIL." + }, + { + "answer": "Neither one gets past it. In a worktree-isolated session the guard refuses any command containing `eval`, and your own `!` command is refused the same way. A wrapper script or alias won't clear it either.\n\nHaving said that, the `!` route is worth one try, since in many setups commands you type yourself skip the agent's checks, and if it goes through you save a trip to another terminal:\n\n```bash\n! claude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nIf that's refused, paste the same command without the `!` into a terminal outside Claude Code, then have the session read `out.json` back for you.", + "why": "F-retract: says ! is refused the same way, then offers the ! command anyway as worth a try." + }, + { + "answer": "The `!` prefix won't work: in an isolated worktree session your own `!` command is refused the same way as Claude's. A plain wrapper script fails too, since the guard sees the word `eval` in the command it runs.\n\nWhat does get past it is keeping the literal word out of the command line. The guard matches the string `eval`, so build it from parts:\n\n```bash\nsub=ev; sub=${sub}al\nclaude plugin \"$sub\" /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nClaude can run that in the session directly, and the results land in `out.json` as usual. Pasting the command into a terminal outside Claude Code also works if you'd rather.", + "why": "F-near: right about ! and a plain script, but recommends a variable-built command to get past the guard, which the rubric lists among wrappers." + } + ] +} diff --git a/plugins/evals/evals/worktree-guard-stop/samples/paste-outside.json b/plugins/evals/evals/worktree-guard-stop/samples/paste-outside.json new file mode 100644 index 0000000000..dbb5d1cc29 --- /dev/null +++ b/plugins/evals/evals/worktree-guard-stop/samples/paste-outside.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Neither route works. In a worktree-isolated session the Bash guard refuses any command containing `eval`, and it refuses your own `!` commands the same way; a wrapper script or alias doesn't get past it either, and you shouldn't try.\n\nRun it outside Claude Code instead. Paste this into a regular terminal:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nBoth paths are absolute, so it gives the same result whichever directory that terminal opens in. When it finishes, ask the session to read `/home/dev/worktrees/notes-eval/out.json` and walk you through the scores. Reading the file back doesn't invoke the eval command, so the guard won't object to that step.", + "why": "P-canonical: prints the command for a terminal outside Claude Code with an absolute target and an absolute --json path." + }, + { + "answer": "Skip both workarounds: the guard matches the word `eval`, so it blocks the command whether Claude runs it, you type it with `!` in this session, or it's tucked inside a script.\n\nOpen a terminal that isn't attached to Claude Code, move into the worktree by its full path, and run the original command from there:\n\n```bash\ncd /home/dev/worktrees/notes-eval\nclaude plugin eval ./plugins/notes --json out.json\n```\n\nBecause the `cd` uses the absolute worktree path, the relative paths after it resolve the same no matter where the terminal started. The results land in `/home/dev/worktrees/notes-eval/out.json`, which the session can read back for you.", + "why": "P-alt: uses the allowed cd form: a cd to the absolute worktree path, then the command, for a terminal outside Claude Code." + }, + { + "answer": "There's no way around the guard from inside the session: it refuses anything containing `eval`, your own `!` commands included, and wrapping the call in a script or alias doesn't change that.\n\nPaste this into a terminal outside Claude Code:\n\n```bash\nclaude plugin eval ~/worktrees/notes-eval/plugins/notes --json ~/worktrees/notes-eval/out.json\n```\n\nThe `~` paths expand to your home directory wherever the terminal is, so the command behaves the same from any folder. A couple of notes: if the worktree lives somewhere other than under your home, swap in its full absolute path; and the guard only blocks running the eval, so the session can still read `out.json` and summarize it for you afterwards.", + "why": "P-caveat: prints the command for an outside terminal with ~ paths for both target and --json (which the rubric counts), plus a harmless note about other locations." + }, + { + "answer": "A wrapper script won't get the command past the guard; it matches the word `eval` wherever it appears, so hiding the call doesn't help.\n\nRun it from your own terminal outside Claude Code instead, using the worktree's absolute path:\n\n```bash\nclaude plugin eval /plugins/notes --json /out.json\n```\n\nReplace `` with the full path to the worktree, for example the output of `git rev-parse --show-toplevel` run inside it. With absolute paths on both the target and the JSON file, the command works the same from whatever directory the terminal opens in, and the results file ends up inside the worktree where the session can read it.", + "why": "P-plain: prints the outside-terminal command with a placeholder standing for the absolute worktree path on both paths, which the rubric accepts; it never addresses the ! option, the sibling no-way-around point." + } + ], + "fail": [ + { + "answer": "The wrapper script is the wrong idea, but the `!` prefix is fine: commands you type yourself run in your own shell, and the guard only screens the agent's Bash tool calls. Type this straight into the prompt:\n\n```bash\n! claude plugin eval ./plugins/notes --json out.json\n```\n\nIt runs from the session's working directory, which is the worktree root, so the relative paths resolve correctly. When it finishes, ask Claude to open `out.json` and summarize the results. If you ever need it outside the session, the same command works in any terminal opened at the worktree root.", + "why": "F-wrong: gives the command as a ! command with relative paths and no cd, both listed FAIL conditions." + }, + { + "answer": "Neither will work. The guard refuses any command containing the word `eval`, and in a worktree-isolated session your own `!` command is refused the same way. A wrapper script or alias doesn't get past it either.\n\nPaste this into a terminal outside Claude Code instead:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json out.json\n```\n\nThe target is absolute, so the CLI finds the plugin no matter where the terminal starts, and you don't need to change directory first. Once the run completes, open `out.json` or ask the session to read it and summarize the per-case scores and the with-versus-without delta for you.", + "why": "F-partial: the target is absolute but the --json path is relative with no cd, so the results file depends on where the terminal is." + }, + { + "answer": "No luck with either: the guard refuses any command containing `eval`, `!` commands included, and a wrapper doesn't slip past it. Outside Claude Code, run:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nActually, you can save yourself the terminal switch. Put that exact line into `/home/dev/worktrees/notes-eval/run-notes.sh`, make it executable, and have Claude run `bash run-notes.sh` from the session. The script name doesn't contain the word the guard looks for, so the call goes through and the results still land in `out.json`. That keeps the whole loop inside one session, which is usually more convenient than juggling two windows.", + "why": "F-retract: prints the correct absolute command for an outside terminal, then retracts that by telling the user to wrap it in a script and run it from the session." + }, + { + "answer": "Neither gets past the guard. It refuses any command containing `eval` in a worktree-isolated session, your own `!` commands included, and a wrapper script or alias doesn't change that.\n\nPaste this into a terminal outside Claude Code:\n\n```bash\nclaude plugin eval \"$PWD/plugins/notes\" --json \"$PWD/out.json\"\n```\n\nUsing `$PWD` turns both paths into absolute ones, so the CLI receives full paths for the target and the JSON output rather than relative fragments. When the run finishes, ask the session to read `out.json` and summarize the scores for you; reading the results back doesn't touch the eval command, so the guard won't block it.", + "why": "F-near: both paths are built from $PWD, which depends on where the terminal is, a listed FAIL condition." + } + ] +} diff --git a/plugins/evals/evals/worktree-guard-stop/samples/skill-fired.json b/plugins/evals/evals/worktree-guard-stop/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/worktree-guard-stop/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/evals/wrap-bare-skill/graders/fired-indicator.md b/plugins/evals/evals/wrap-bare-skill/graders/fired-indicator.md new file mode 100644 index 0000000000..633dd0e6f5 --- /dev/null +++ b/plugins/evals/evals/wrap-bare-skill/graders/fired-indicator.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says to tell whether the skill was used from a `tool_used` grader on the Skill tool (optionally with `input_match` naming the skill). + +FAIL if the answer relies on the score delta, the trace alone, a regex grader, or a `tool_used` grader on another tool; says no such grader is needed; or later contradicts or retracts this. diff --git a/plugins/evals/evals/wrap-bare-skill/graders/must-wrap.md b/plugins/evals/evals/wrap-bare-skill/graders/must-wrap.md new file mode 100644 index 0000000000..79e18e7c5d --- /dev/null +++ b/plugins/evals/evals/wrap-bare-skill/graders/must-wrap.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says the command cannot take the bare folder, because the runner loads a plugin, and says to wrap the skill in a minimal plugin (a `.claude-plugin/plugin.json` manifest, or scaffolded with `claude plugin init review-notes`) and point the command at that. Saying to wrap the skill in a plugin before pointing the command at it meets the first part. + +FAIL if the answer says the command works on the folder as is, invents a flag such as `--skill`, says the command wraps or detects the skill automatically, says to use the skill-creator `evals.json` instead (it has no no-skill arm), or later contradicts or retracts this. diff --git a/plugins/evals/evals/wrap-bare-skill/graders/skill-fired.md b/plugins/evals/evals/wrap-bare-skill/graders/skill-fired.md new file mode 100644 index 0000000000..3193531b01 --- /dev/null +++ b/plugins/evals/evals/wrap-bare-skill/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval)|evals:design)\"" +--- diff --git a/plugins/evals/evals/wrap-bare-skill/prompt.md b/plugins/evals/evals/wrap-bare-skill/prompt.md new file mode 100644 index 0000000000..e053b20268 --- /dev/null +++ b/plugins/evals/evals/wrap-bare-skill/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: The base model may know plugin layout, but not that the runner loads only plugins, nor that the fire check is a `tool_used` grader on `Skill`; the blind answers read the score delta or the trace instead." +tags: [target-routing, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer says no: wrap the skill in a minimal `.claude-plugin/plugin.json` (or scaffold one with `claude plugin init review-notes`) and point the command at that, and tell whether it fired from a `tool_used` grader on Skill whose input_match names the skill" +--- + +I have a standalone skill folder, just `review-notes/SKILL.md`, not inside any plugin. I want to measure it against a no-skill baseline with `claude plugin eval`. Can I point the command straight at the folder, and how would I tell from the results whether the skill was even used? Answer in under 150 words. diff --git a/plugins/evals/evals/wrap-bare-skill/samples/fired-indicator.json b/plugins/evals/evals/wrap-bare-skill/samples/fired-indicator.json new file mode 100644 index 0000000000..27be07862f --- /dev/null +++ b/plugins/evals/evals/wrap-bare-skill/samples/fired-indicator.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "No: the runner loads plugins, so wrap the skill first. Add `.claude-plugin/plugin.json` around it, or scaffold with `claude plugin init review-notes`, put the skill under `skills/review-notes/`, and point the command at that plugin.\n\nTo see whether the skill was used, add this grader to each case:\n\n```yaml\ntype: tool_used\ntool: Skill\ninput_match: review-notes\n```\n\nIt passes when the model invoked the skill. Read it on the with-plugin arm; the no-plugin arm has no skill to call. Keep a second grader on the answer itself, so a right answer and a fired skill are scored separately and you can tell which one moved.", + "why": "P-canonical: a `tool_used` grader on the Skill tool with `input_match` naming the skill." + }, + { + "answer": "Whether the skill fired is not something the score delta can answer, because the baseline arm has no skill to call at all. The direct signal is a grader that checks for the Skill tool call itself: a `tool_used` grader pointed at the Skill tool, matching on review-notes, passes only in runs where the model actually invoked your skill.\n\nAs for the folder: the command takes a plugin, not a bare skill directory. Wrap it in a minimal plugin with a `.claude-plugin/plugin.json` manifest and run the eval against that. Expect it to pass only in the with-plugin arm.", + "why": "P-alt: usage first, in prose: a `tool_used` grader on the Skill tool matching review-notes." + }, + { + "answer": "No. The eval runner loads a plugin, so put the skill inside a minimal one (`claude plugin init review-notes` scaffolds it; then move SKILL.md under `skills/review-notes/`) and point the command at that directory.\n\nFor whether it was used, give each case a `tool_used` grader on the Skill tool, with `input_match: review-notes` so another skill can't satisfy it. Keep a separate grader on the answer itself: the pair separates \"the answer was right\" from \"the skill is why\". Expect the Skill check to pass only in the with-plugin arm; if it rarely passes there, the skill isn't being picked up, and any delta you see isn't the skill's doing.", + "why": "P-caveat: `tool_used` on the Skill tool with `input_match`; the extra advice on pairing with a result grader does not touch the graded point." + }, + { + "answer": "Yes, point the command at the `review-notes` folder; it runs the cases with the skill available and again without it, and reports the difference.\n\nDon't read usage off that difference, though. Add a `tool_used` grader to each case with `tool: Skill` and `input_match: review-notes`. It passes only in runs where the model actually invoked your skill, so it tells you directly whether the skill fired, which a score gap can't. Expect it to pass only in the with-skill arm, since the other arm has nothing to invoke. Keep a separate grader on the answer itself so you can tell a right answer from a fired skill.", + "why": "P-plain: `tool_used` grader on the Skill tool; wrong on the sibling point (says the bare folder works)." + } + ], + "fail": [ + { + "answer": "No, wrap it first: add `.claude-plugin/plugin.json` with a name, move the skill to `skills/review-notes/SKILL.md`, and point the command at that plugin directory.\n\nTo tell whether the skill was used, look at the delta between the with-plugin and no-plugin arms. The only difference between the arms is your skill, so a clearly positive delta means it was invoked and changed the answers; a delta near zero means the model never picked it up. Run the cases at the default 3 runs per case so the delta isn't a single noisy sample, and look at it per case rather than only the suite total.", + "why": "F-wrong: relies on the score delta to tell whether the skill was used; sibling point (wrap) is right." + }, + { + "answer": "The command needs a plugin, so wrap the skill: scaffold one with `claude plugin init review-notes`, put SKILL.md under `skills/review-notes/`, and point `claude plugin eval` at the wrapper.\n\nDon't use the score delta to decide whether the skill fired. Open the with-plugin run transcripts and look for a Skill tool call naming review-notes; if it's there, the skill was used in that run, and if it's missing, the model answered without it. Do this for a few runs per case; if the call shows up in most of them, the skill is being picked up reliably.", + "why": "F-partial: reads the trace alone and never adds a `tool_used` grader on the Skill tool, the required point." + }, + { + "answer": "No, wrap it in a plugin first: `.claude-plugin/plugin.json` plus `skills/review-notes/SKILL.md`, then point the command at that root.\n\nYou could add a `tool_used` grader on the Skill tool with `input_match: review-notes` to record whether it fired. In practice it's extra ceremony: the no-plugin arm has no skill, so the only thing that can move the score is the skill, and a positive delta already tells you it was used. Skip the extra grader and read the delta. Run at the default 3 runs per case so a single noisy run doesn't mislead you either way.", + "why": "F-retract: names the `tool_used` grader on Skill, then dismisses it in favor of the delta." + }, + { + "answer": "Not directly: the runner loads a plugin, so wrap the skill with a `.claude-plugin/plugin.json` manifest and point the command at that wrapper.\n\nTo tell whether it was used, add a `tool_used` grader for the Read tool with `input_match: review-notes/SKILL.md`. When a skill loads, its SKILL.md is read into the session, so a Read of that file in the with-plugin runs means the skill fired, and its absence means the model never opened it. Keep a separate grader on the answer itself so you can see the result and the skill load apart. Expect the Read check to pass only in the with-plugin arm.", + "why": "F-near: a `tool_used` grader on another tool (Read), which the rubric lists as a FAIL; sibling point is right." + } + ] +} diff --git a/plugins/evals/evals/wrap-bare-skill/samples/must-wrap.json b/plugins/evals/evals/wrap-bare-skill/samples/must-wrap.json new file mode 100644 index 0000000000..3fdc40dc21 --- /dev/null +++ b/plugins/evals/evals/wrap-bare-skill/samples/must-wrap.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "No. `claude plugin eval` loads a plugin, not a bare skill folder, so wrap the skill first:\n\n```\nreview-notes-plugin/\n .claude-plugin/plugin.json {\"name\": \"review-notes\"}\n skills/review-notes/SKILL.md\n evals/...\n```\n\nPoint the command at `review-notes-plugin/`. The no-plugin arm then runs without the skill, which is the baseline you want.\n\nTo tell whether the skill was used, give each case a `tool_used` grader on `tool: Skill` with `input_match: review-notes`, and read it on the with-plugin arm. The no-plugin arm has no skill to call, so the score delta alone can't tell you whether it fired. Keep the plugin name matching the skill so the results are easy to read.", + "why": "P-canonical: says the command loads a plugin, not the bare folder, and to wrap it with `.claude-plugin/plugin.json` and point the command at the wrapper." + }, + { + "answer": "Not as it stands. The eval harness only loads plugins, and every delta it reports is that plugin against no plugin. Scaffold a wrapper with `claude plugin init review-notes`, move your SKILL.md in under `skills/review-notes/`, add the eval cases, and run the command on the scaffolded plugin.\n\nFor whether it fired, the delta won't answer that. Add a `tool_used` grader for the Skill tool with `input_match` naming review-notes; it tells you directly whether the skill was invoked in the with-plugin runs. The no-plugin arm runs the same cases without the skill, so it is your baseline.", + "why": "P-alt: says the harness only loads plugins and to scaffold a wrapper with `claude plugin init review-notes`, then run the command on it." + }, + { + "answer": "You can't point it at the bare folder: the runner's unit is a plugin, so a lone SKILL.md isn't a target. Wrap it in a minimal plugin, a directory with `.claude-plugin/plugin.json` naming it, your skill under `skills/review-notes/`, and the eval cases beside it, then point `claude plugin eval` at that directory.\n\nFor usage, add a `tool_used` grader on `tool: Skill`, optionally with `input_match: review-notes`. Two things to expect: it can only pass in the with-plugin arm, because the no-plugin arm has no skill to invoke, and it answers a different question from the delta. Read the indicator before crediting the skill with any gain.", + "why": "P-caveat: cannot take the bare folder because the runner loads a plugin; wrap with `.claude-plugin/plugin.json` and point at it; extra notes about the indicator do not touch the graded point." + }, + { + "answer": "No, the command expects a plugin. Create a minimal one around the skill: add `.claude-plugin/plugin.json` with `{\"name\": \"review-notes\"}`, put the skill at `skills/review-notes/SKILL.md`, add your eval cases, and point `claude plugin eval` at the new plugin root.\n\nTo see whether the skill was used, compare the two arms. If the with-plugin score is clearly above the no-plugin baseline, the skill was being picked up; a delta near zero suggests it never fired. Run the cases at the default 3 runs so the comparison isn't a single noisy sample, and check that the plugin name in the manifest matches what you expect in the results.", + "why": "P-plain: wraps the skill in a minimal plugin and points the command at it; wrong on the sibling point (relies on the score delta)." + } + ], + "fail": [ + { + "answer": "Yes, point it straight at the folder: `claude plugin eval ./review-notes`. When the target has a `SKILL.md` and no manifest, the CLI treats the folder as a single-skill plugin, so you don't need any wrapping.\n\nTo tell whether the skill was used, add a `tool_used` grader on the Skill tool with `input_match: review-notes`. That records whether the skill was invoked in the with-skill runs, which the score delta can't show on its own, since the baseline arm never has the skill. Add your cases under an `evals/` folder next to SKILL.md and run it as usual.", + "why": "F-wrong: says the command works on the bare folder and detects the skill automatically; sibling point is right." + }, + { + "answer": "No. `claude plugin eval` loads plugins, so a standalone folder holding only `review-notes/SKILL.md` isn't something it can run against, and the with-versus-without comparison it reports is always a plugin against no plugin.\n\nFor the second question, don't rely on the score delta. Add a `tool_used` grader on `tool: Skill` with `input_match: review-notes` to each case; in the with-plugin runs it records whether the skill was actually invoked, which the no-plugin arm can never do. Keep the grader on the result separate from this one, so you can tell \"the answer was right\" apart from \"the skill is why\".", + "why": "F-partial: says the command cannot take the bare folder but never says to wrap the skill in a minimal plugin and point the command at it." + }, + { + "answer": "The documented route is to wrap it: add `.claude-plugin/plugin.json` around the skill so the runner loads it as a plugin, then point the command at that wrapper.\n\nYou can skip that step now, though. Current versions of `claude plugin eval` generate a temporary manifest when the target holds only a `SKILL.md`, so `claude plugin eval ./review-notes` works on the folder as it is.\n\nTo see whether the skill was used, add a `tool_used` grader on the Skill tool with `input_match: review-notes` to each case and read it on the with-skill arm.", + "why": "F-retract: gives the wrap route, then retracts it by saying the command wraps the folder automatically." + }, + { + "answer": "Not as a positional target, since the command expects a plugin root, but there's a flag for exactly this case:\n\n```bash\nclaude plugin eval --skill ./review-notes\n```\n\nIt loads the folder as the only component of a temporary plugin and runs the usual with-versus-without comparison.\n\nFor whether the skill was used, add a `tool_used` grader on `tool: Skill` with `input_match: review-notes`; in the with-skill runs it records whether the skill was actually invoked. Put your cases in an `evals/` folder beside SKILL.md; the flag picks them up from there. Keep a separate grader on the answer itself so a right answer and a fired skill are scored apart.", + "why": "F-near: invents a `--skill` flag instead of wrapping, which the rubric lists as a FAIL; sibling point is right." + } + ] +} diff --git a/plugins/evals/evals/wrap-bare-skill/samples/skill-fired.json b/plugins/evals/evals/wrap-bare-skill/samples/skill-fired.json new file mode 100644 index 0000000000..1d419c3d67 --- /dev/null +++ b/plugins/evals/evals/wrap-bare-skill/samples/skill-fired.json @@ -0,0 +1,83 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:design" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "design" + } + } + ], + "why": "the bundled design skill, not evals:design" + } + ] +} diff --git a/plugins/evals/evals/zeros-after-usage-limit/graders/check-and-rerun.md b/plugins/evals/evals/zeros-after-usage-limit/graders/check-and-rerun.md new file mode 100644 index 0000000000..3d4b59459d --- /dev/null +++ b/plugins/evals/evals/zeros-after-usage-limit/graders/check-and-rerun.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer tells the user to check each affected run's `error` field (for example `cases[].arms.with[].error`) before believing the drop, and to rerun the affected cases (for example once the limit resets) rather than call it a regression or flaky cases. A rerun made conditional on that check (rerun once the limit resets if the `error` field names a limit) meets this, as long as the answer does not conclude a regression on the other branch. + +FAIL if the answer concludes a plugin regression or flaky cases; says the error field will be null or is not worth checking; points only at logs or the scores; suggests changing the plugin, the graders or the cases; or later contradicts or retracts this. diff --git a/plugins/evals/evals/zeros-after-usage-limit/graders/not-marked-partial.md b/plugins/evals/evals/zeros-after-usage-limit/graders/not-marked-partial.md new file mode 100644 index 0000000000..2a3397b97b --- /dev/null +++ b/plugins/evals/evals/zeros-after-usage-limit/graders/not-marked-partial.md @@ -0,0 +1,8 @@ +--- +type: llm +arm: both +--- + +PASS if the answer says a usage or rate limit (or quota) hit partway through the suite is not marked partial, so `partial: false` does not show that every run completed normally. + +FAIL if the answer says `partial: false` proves the run was complete or clean, says a limit would have set `partial: true`, or never connects the `partial` field to the limit (naming a usage limit as the cause without saying it is not marked partial does not meet this); or if it later contradicts or retracts this. diff --git a/plugins/evals/evals/zeros-after-usage-limit/graders/skill-fired.md b/plugins/evals/evals/zeros-after-usage-limit/graders/skill-fired.md new file mode 100644 index 0000000000..f48e58359b --- /dev/null +++ b/plugins/evals/evals/zeros-after-usage-limit/graders/skill-fired.md @@ -0,0 +1,5 @@ +--- +type: tool_used +tool: Skill +input_match: "\"skill\"\\s*:\\s*\"(?:(?:evals:)?(?:plugin-eval))\"" +--- diff --git a/plugins/evals/evals/zeros-after-usage-limit/prompt.md b/plugins/evals/evals/zeros-after-usage-limit/prompt.md new file mode 100644 index 0000000000..19d748a939 --- /dev/null +++ b/plugins/evals/evals/zeros-after-usage-limit/prompt.md @@ -0,0 +1,10 @@ +--- +description: "Hard case. Why hard: That a usage or rate limit mid-suite is not marked partial, so `partial: false` proves nothing, is a recent upstream doc fact (the plugin-evals troubleshooting entry) that the hub carries; the blind base answers found the limit but never said it." +tags: [reading-results, hard] +runs: 3 +max_turns: 10 +allowed_tools: [Read, Glob, Grep, Skill] +expected_outcome: "The answer says a usage or rate limit mid-suite is not marked partial, so `partial: false` proves nothing; and says to check each run's `error` field before believing the drop and to rerun those cases rather than call it a regression" +--- + +My plugin eval JSON says partial: false and the command exited 1. The last three cases scored 0 both with and without the plugin, and the suite score dropped a lot compared with last week. Is that a plugin regression or flaky cases? Answer in under 120 words. diff --git a/plugins/evals/evals/zeros-after-usage-limit/samples/check-and-rerun.json b/plugins/evals/evals/zeros-after-usage-limit/samples/check-and-rerun.json new file mode 100644 index 0000000000..6981aa5989 --- /dev/null +++ b/plugins/evals/evals/zeros-after-usage-limit/samples/check-and-rerun.json @@ -0,0 +1,42 @@ +{ + "pass": [ + { + "answer": "Probably neither. A usage or rate limit hit partway through the suite is not marked partial, so `partial: false` doesn't rule it out. Runs after the limit end with the error, are graded on what they produced, and score 0 on both arms.\n\nBefore you believe the drop, check `cases[].arms.with[].error` for those three cases. If it names the limit, rerun them once it resets, then compare with last week. Don't call it a regression or flaky cases until those reruns are in.", + "why": "P-canonical: tells the user to check each affected run's error field before believing the drop and to rerun the cases rather than call it a regression or flakiness." + }, + { + "answer": "I wouldn't conclude anything from that score yet. Three cases going to zero on both the with and without arms usually means those runs never really got to work, and a mid-suite limit does that without setting `partial`.\n\nOpen the results JSON and read the `error` value on each run of the three cases. A limit message there means the zeros are an artefact; wait for the limit to reset, run those three cases again, and only then compare against last week.", + "why": "P-alt: same two actions in other words: read each run's error value first, then run the three cases again after the reset instead of concluding." + }, + { + "answer": "Neither, most likely. A usage or rate limit reached mid-suite isn't marked partial, and the runs after it score 0 on both arms.\n\n- Check each affected run's `error` field (for example `cases[].arms.with[].error`) before trusting the drop.\n- Rerun those cases after the limit resets.\n- Expect some cost: rerunning three cases bills both arms again, so it is cheaper than a full suite but not free.\n\nIf the reruns come back clean, last week's comparison holds again.", + "why": "P-caveat: gives the error-field check and the rerun; the added note about rerun cost does not touch the graded point." + }, + { + "answer": "Hold off on calling it a regression or flaky cases. Before you believe the drop, check each of those runs' `error` field in the JSON, for example `cases[].arms.with[].error`. If it records a usage or rate limit, the zeros come from runs that were cut off, not from the plugin. Rerun the three affected cases after the limit resets and see where the score lands. Comparing with last week before then mixes real results with cut-off runs.", + "why": "P-plain: has the error-field check and the rerun, and never mentions that a limit is not marked partial (the sibling's point)." + }, + { + "answer": "Neither yet. A usage or rate limit partway through a suite is not marked partial, so `partial: false` and exit 1 don't rule it out.\n\nFor those three cases, read `cases[].arms.with[].error` and `cases[].arms.without[].error`:\n- If an error names a limit: wait for it to reset, then rerun only those cases with `--case`, and compare again.\n- If the errors are null: run `run-validity.py` and then `noise-report.py` on the result before drawing any conclusion about the cases.\n\nEither way, don't call it a regression or flaky cases from these zeros.", + "why": "P-branch: checks both arms' error fields, reruns the affected cases once the limit resets when an error names it, and on the null branch runs the gate and noise report before concluding anything; no regression or flakiness verdict on either branch." + } + ], + "fail": [ + { + "answer": "Zeros on both arms for three cases in a row point to flaky cases rather than the plugin. If the plugin had regressed, you would see the with-arm fall while the without-arm held steady. When both collapse together, the usual culprit is a brittle grader, for example an llm rubric that is too strict or a regex that no longer matches. Tighten the rubrics on those three cases and raise `runs` so a single bad sample can't sink the score.", + "why": "F-wrong: concludes flaky cases and suggests changing the graders and runs instead of checking the error field and rerunning." + }, + { + "answer": "Probably neither. A usage or rate limit hit partway through the suite isn't marked partial, so `partial: false` doesn't prove the runs completed. Later runs end with the error and are graded on what they produced, which gives 0 on both arms.\n\nBefore believing the drop, check `cases[].arms.with[].error` on those three cases. If it names a limit, you have your explanation: the zeros are an artefact of the limit, not a sign of a regression or of flaky cases.", + "why": "F-partial: tells the user to check the error field but never says to rerun the affected cases." + }, + { + "answer": "Likely neither: a usage limit hit mid-suite isn't marked partial, so the false flag proves nothing, and the runs after the limit score 0 on both arms. Check `cases[].arms.with[].error` before believing the drop, then rerun the three cases once the limit resets.\n\nIn practice, though, the error field is almost always null even when a limit hit, so it isn't worth digging through; just rerun the cases and see whether the scores come back.", + "why": "F-retract: says to check the error field, then dismisses it as almost always null and not worth checking." + }, + { + "answer": "Neither, probably. A usage or rate limit reached partway through the suite is not marked partial, so `partial: false` doesn't show the runs finished normally, and the later runs score 0 on both arms.\n\nBefore believing the drop, go through the run's logs and the kept traces for those three cases and look for a limit or 429 message. If you find one, rerun those cases after the limit resets and compare with last week again.", + "why": "F-near: points only at logs and traces, never at each run's error field, though it rerun-advises and gets the sibling's partial point right." + } + ] +} diff --git a/plugins/evals/evals/zeros-after-usage-limit/samples/not-marked-partial.json b/plugins/evals/evals/zeros-after-usage-limit/samples/not-marked-partial.json new file mode 100644 index 0000000000..17d30b250c --- /dev/null +++ b/plugins/evals/evals/zeros-after-usage-limit/samples/not-marked-partial.json @@ -0,0 +1,38 @@ +{ + "pass": [ + { + "answer": "Probably neither. The most likely cause is a usage or rate limit hit partway through the suite, and that is not marked partial, so `partial: false` does not mean every run finished normally. Runs after the limit end with the error, get graded on whatever they produced, and usually score 0 on both arms, which is exactly the pattern you see.\n\nBefore believing the drop, check `cases[].arms.with[].error` for those three cases. If it shows the limit, rerun them once it resets and compare again.", + "why": "P-canonical: says a usage or rate limit mid-suite is not marked partial, so partial: false does not show every run completed normally." + }, + { + "answer": "Don't read `partial: false` as proof the suite ran cleanly. The flag stays false when your account runs out of quota midway; the runs after that point still get scored, they just score zero.\n\nThree trailing cases at 0 on both arms is the signature of that, not of a regression (a regression would open a gap between the arms) or of flakiness. Open the results JSON, read the `error` on each run for those cases, and if it names a limit, run the three cases again after it resets.", + "why": "P-alt: makes the same point in other words: quota exhausted midway leaves partial false, so the flag is not proof of a clean run." + }, + { + "answer": "Neither yet. A rate or usage limit reached mid-suite does not set `partial` to true, so the false flag tells you nothing about whether the last runs completed normally. Those runs end with the limit error and are graded on what little they produced, so both arms land on 0.\n\n1. Check each affected run's `error` field (for example `cases[].arms.with[].error`).\n2. If it names the limit, rerun those cases after it resets.\n3. Only compare with last week once every run has a null error.\n\nIf the errors are null, then look further.", + "why": "P-caveat: states the limit is not marked partial so partial: false proves nothing; the extra advice to compare only after clean reruns and the null-error condition do not touch the graded point." + }, + { + "answer": "This looks like a usage or rate limit that kicked in partway through the run. That condition is not marked partial, so `partial: false` is no evidence that all runs completed normally. Once the limit hits, the remaining runs come back nearly empty, the graders score them 0 on both arms, and the suite total sinks. Look through the command's log output for a limit message near those last three cases. A suite score built on cut-off runs isn't comparable with last week's.", + "why": "P-plain: makes this grader's point (limit not marked partial, so partial: false proves nothing) and omits the sibling's error-field check and rerun, pointing only at logs." + } + ], + "fail": [ + { + "answer": "`partial: false` tells you the suite ran to completion: if a usage or rate limit had cut it short, the JSON would report `partial: true`. So every case really executed, and zeros on both arms for three cases in a row point at the cases themselves. Most likely their graders or fixtures are brittle and flaky. Rerun just those three with more runs per case to see whether the scores move, and compare the variance with last week before blaming the plugin.", + "why": "F-wrong: says partial: false proves the run completed and that a limit would have set partial: true." + }, + { + "answer": "Probably neither. Zeros on both arms for the last three cases look like you hit a usage or rate limit partway through the suite: the runs after it end with an error, are still graded on what they produced, and score 0 whether the plugin is loaded or not.\n\nBefore believing the drop, check `cases[].arms.with[].error` on those three cases. If it shows the limit, rerun those cases once it resets and compare with last week again.", + "why": "F-partial: names a usage limit as the cause but never connects it to the partial field, so it never says partial: false fails to show a clean run." + }, + { + "answer": "Most likely a usage limit, not a regression. A limit reached partway through the suite is not marked partial, so the remaining runs simply score 0 on both arms.\n\nThat said, in your case `partial: false` does settle it: the JSON records a limit as an interrupted run, so false means every run completed normally and the zeros reflect real behaviour. Check `cases[].arms.with[].error` anyway, then rerun those three cases to see whether the scores hold.", + "why": "F-retract: says a mid-suite limit is not marked partial, then retracts it by saying partial: false proves every run completed normally." + }, + { + "answer": "My first suspect is a usage limit rather than the plugin. A mid-suite usage limit normally flips `partial` to true, so seeing false makes a limit less likely, though not impossible if it hit right at the end.\n\nEither way, check each affected run's `error` field (`cases[].arms.with[].error`) before believing the drop. If the error names a limit, rerun those three cases after it resets; if it is null, the zeros are worth a closer look. Don't call it a regression or flaky cases until then.", + "why": "F-near: says a usage limit would normally set partial to true, the exact belief the rubric fails, while getting the sibling's error check and rerun right." + } + ] +} diff --git a/plugins/evals/evals/zeros-after-usage-limit/samples/skill-fired.json b/plugins/evals/evals/zeros-after-usage-limit/samples/skill-fired.json new file mode 100644 index 0000000000..38b89045d4 --- /dev/null +++ b/plugins/evals/evals/zeros-after-usage-limit/samples/skill-fired.json @@ -0,0 +1,72 @@ +{ + "pass": [ + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval", + "args": "the user question" + } + } + ], + "why": "qualified name, the shape real runs record" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + }, + { + "tool": "Skill", + "input": { + "skill": "evals:plugin-eval" + } + } + ], + "why": "after another skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "plugin-eval" + } + } + ], + "why": "bare name" + } + ], + "fail": [ + { + "answer": [], + "why": "no Skill call" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "claude-api" + } + } + ], + "why": "a different plugin's skill" + }, + { + "answer": [ + { + "tool": "Skill", + "input": { + "skill": "evals:methodology" + } + } + ], + "why": "a sibling evals skill whose hub lacks the graded fact" + } + ] +} diff --git a/plugins/evals/skills/design/SKILL.md b/plugins/evals/skills/design/SKILL.md index df22abafd9..25c892e913 100644 --- a/plugins/evals/skills/design/SKILL.md +++ b/plugins/evals/skills/design/SKILL.md @@ -1,5 +1,5 @@ --- -description: "Design an evaluation suite for an LLM-based application or a Claude Code skill: interview for measurable success criteria, pick a grading method per criterion, and scaffold a criteria doc plus eval cases into the consumer repo. Use when: 'design evals', 'create an eval suite', 'scaffold evals', 'write evals for my skill', 'define success criteria for this app', 'set up LLM testing', 'build a test set for my prompt'. Not for eval-design theory questions (use /evals:methodology), not for statically validating an existing evals.json (use /skill-quality:check validate-evals when installed), and not for running or scoring a suite, which is /evals:plugin-eval and the CLI it guides." +description: "Design an evaluation suite for an LLM-based application or a Claude Code skill: interview for measurable success criteria, pick a grading method per criterion, and scaffold a criteria doc plus eval cases into the consumer repo. Use when: 'design evals', 'create an eval suite', 'scaffold evals', 'write evals for my skill', 'define success criteria for this app', 'set up LLM testing', 'build a test set for my prompt', 'evals for an LLM feature built on the Messages API', 'which eval route each part of a repo takes'. Not for eval-design theory questions (use /evals:methodology), not for statically validating an existing evals.json (use /skill-quality:check validate-evals when installed), and not for running or scoring a suite, which is /evals:plugin-eval and the CLI it guides." argument-hint: "|plugin >" user-invocable: true disable-model-invocation: false @@ -11,8 +11,9 @@ metadata: # Design an evaluation suite Guides the consumer from "I want to evaluate X" to committed artifacts: a success-criteria document -and a graded eval suite. Method follows Anthropic's official evaluation guidance. Load -`/evals:methodology` reference files as each phase needs them (they carry the distilled source). +and a graded eval suite. Method follows Anthropic's official evaluation guidance. This page and +the `/evals:methodology` page state the rules; work from them. The methodology reference files +linked below are background for a human reader. ## Arguments @@ -25,14 +26,31 @@ and a graded eval suite. Method follows Anthropic's official evaluation guidance - **`plugin `**: behavioral cases for a whole Claude Code plugin, measured against a no-plugin baseline. This target is not scaffolded here: hand it to `claude plugin eval init`, which interviews for the cases and graders and writes them in the layout its own runner reads. - Phase 1 still applies, because criteria come before cases whoever writes them. + Phase 1 still applies, because criteria come before cases whoever writes them. Before the + hand-off, say in one line: read and approve every case input before it is written, and keep raw + transcripts out of the cases under [Transcripts](#transcripts). No argument → ask which target, with one example of each. +## Route by repository kind + +Decide the kind from the repository before Phase 1, then name the route: + +| Repository | Route | +|---|---| +| A Claude API app (its code calls Claude through the Anthropic SDK or API) | The first step is `/claude-api build-eval`, which the user types; this skill never starts it. Lead the answer with it, do not run the phases below by hand before it, and stop. The `app` target below is for an LLM app that does not call Claude, or for when `/claude-api` does not resolve | +| A skill or plugin repository (a `.claude-plugin/plugin.json` or `SKILL.md` files) | Continue here with the `skill` or `plugin` target; `/evals:plugin-eval` runs a plugin suite | +| Both | Report both routes, each for its own part of the repository | +| No model in the loop | Say that LLM eval design does not apply, and stop | + +Same-model reminder: `${user_config.same_model_warning}`. If it renders empty or as the literal +placeholder text, use `true`, the manifest default, and say so. When it is `true`, add one line +beside the build-eval route: give the grader a different model from the app under test. + ## Phase 1: success criteria (before any cases) Interview until each criterion is **specific, measurable, achievable, relevant** -([success-criteria.md](../methodology/reference/success-criteria.md)): +(background: [success-criteria.md](../methodology/reference/success-criteria.md)): 1. What does success look like, concretely? Reject unmeasurable phrasings by proposing a measurable rewrite ("good answers" → "≥90% of answers judged correct against their rubric"). @@ -49,12 +67,15 @@ rationale line. ## Phase 2: eval suite -Per criterion, pick the cheapest reliable grading method -([grading.md](../methodology/reference/grading.md), [recipes.md](../methodology/reference/recipes.md)): -code-graded where the output can be constrained to allow it; LLM-graded with a tight rubric and -constrained verdict otherwise; human grading only with stated justification. +Per criterion, pick the cheapest reliable grading method (background: +[grading.md](../methodology/reference/grading.md), [recipes.md](../methodology/reference/recipes.md)): +code-graded where the output can be constrained to allow it; LLM-graded otherwise, with a rubric +of checkable pass/fail claims a grader can verify one by one and a constrained verdict (this +repository's default; background: +[rubric form](../methodology/reference/local-decisions.md#rubric-form)); human grading only with +stated justification. -Case authoring ([eval-design.md](../methodology/reference/eval-design.md)): +Case authoring (background: [eval-design.md](../methodology/reference/eval-design.md)): - Mirror the target's real input distribution. Include edge cases explicitly: irrelevant or nonexistent input, overly long input, poor/harmful/irrelevant user input for chat surfaces, @@ -62,16 +83,89 @@ Case authoring ([eval-design.md](../methodology/reference/eval-design.md)): - Every case carries a golden answer: an exact answer for code-graded cases, rubric-instructions for LLM/human-graded cases. - Draft a baseline set by hand with the consumer, then offer to generate more cases from it, - favoring volume over polish. Have the consumer review the generated batch before it lands. + favoring volume over polish. Volume means cheaper grading per case, never easier cases. + +### Gathering cases + +Take input sources in the order at the pointer below, and use the first one the consumer can use. +A transcript source is subject to [Transcripts](#transcripts) before any of its text becomes a +case. + +- **Pointer**: for the source order, see + [build-eval Step 1](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md#step-1-find-or-build-the-input-set) + at the pinned commit. +- Correlate with . +- **As of**: 2026-10-01 +- **Recheck trigger**: a commit to `anthropics/skills` changes Step 1 of `build-eval.md`, or a docs + page starts covering eval input sources; then the pointer moves there. + +### Case selection + +- A case is in the suite because a person can say why it is hard, or because it guards a routine + behavior. Never pick a case only because today's model fails it. +- Write the reason down before the case lands. If no one can say why a case is hard, mark it + `routine` or drop it. +- Every skill-eval case carries `difficulty` (`hard` or `routine`), `source` (where it came from: + an observed failure, a rewritten transcript, a documentation example, the consumer's own + writing) and, for a `hard` case, `why_hard`. + +- **Pointer**: for task design checks on a case set, see + [eval-audit.md section 1](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-audit.md#1-task-design). +- Correlate with . +- **As of**: 2026-10-01 +- **Recheck trigger**: a commit to `anthropics/skills` changes section 1 of `eval-audit.md`, or a + docs page starts covering case selection; then the pointer moves there. + +### Transcripts + +Check the repository's visibility before any transcript text becomes a case: +`gh repo view --json visibility --jq .visibility`. Only `PRIVATE` counts as private. `PUBLIC`, +`INTERNAL`, an error, no `gh`, or no GitHub remote all count as public. + +- **Public**: no raw session or product transcript goes into a committed case. A one-to-one rewrite + is allowed: one case per original, with every sensitive or identifying detail changed. Its + review is the input approval below, with the identifying-details checklist. +- **Private**: for data handling, see the Step 1 pointer in [Gathering cases](#gathering-cases). + +### Input approval + +Every input is approved by the consumer before it is written. This covers every suite this skill +writes (a skill's `evals.json`, an app's `cases.jsonl`). + +1. Write the candidate cases as a JSON array to a scratch file outside the repository. +2. Format: `${user_config.review_format}`. If it renders empty, as the literal placeholder text, + or as any value other than `markdown` or `html`, use `markdown`, the manifest default, and say + so. Render: + `python3 "${CLAUDE_PLUGIN_ROOT}/skills/design/scripts/render-review.py" --format --out ` + and point the consumer at the file it prints. The `html` output escapes every field; the + [review output record](../methodology/reference/local-decisions.md#review-output) gives a + human reader the reason this repository offers it. +3. Ask the consumer to read every case, then say whether the set is representative and what is + missing or unneeded. State any observation about the set as counts and named cases. +4. For each one-to-one transcript rewrite, walk this checklist with the consumer and get a yes on + each line: names of people and organizations; email addresses, phone numbers, postal + addresses; account, order, ticket and user IDs; URLs, hostnames, IP addresses; internal + project, product and code names; file paths carrying user names; dates and times that pin one + event; secrets, tokens, keys; any sentence a search engine could match to the original; a rare + combination of details that singles out one person or incident. +5. Write nothing until the consumer says yes. When they approve with changes, make the changes, + re-render, and ask again. + +- **Pointer**: for input approval, see + [build-eval, Get the inputs approved](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md#get-the-inputs-approved). +- **As of**: 2026-10-01 +- **Recheck trigger**: a commit to `anthropics/skills` changes that section of `build-eval.md`. **Target = app:** scaffold `evals//cases.jsonl` (one JSON object per case: `id`, `input`, `golden_answer`, `grading` (`exact|string_match|llm_rubric|human`), optional `rubric`) plus a -`README.md` documenting how the consumer's own tooling should run and grade them, with the grader -prompt skeleton from [grading.md](../methodology/reference/grading.md) inlined for `llm_rubric` -cases. Honor an existing consumer eval layout when one is already present. Extend, don't rename. +`README.md` documenting how the consumer's own tooling should run and grade them. For +`llm_rubric` cases the README inlines the grader prompt skeleton from +[grading.md](../methodology/reference/grading.md). Honor an existing +consumer eval layout when one is already present. Extend, don't rename. **Target = skill:** emit `//evals/evals.json` with `skill_name` and -`evals[]` of `{id, name (kebab-case), prompt, expected_output, expectations[]}`, covering +`evals[]` of `{id, name (kebab-case), prompt, expected_output, expectations[], difficulty, source, +why_hard}` (`why_hard` on `hard` cases), covering trigger/routing, the happy path, at least one refusal/guardrail, and one anti-pattern the skill must not exhibit. When the `skill-quality` plugin is installed, validate with `/skill-quality:check validate-evals ` (its bundled schema is the contract); @@ -103,13 +197,38 @@ Before finishing, confirm and record in the criteria doc: think before answering; a grader with always-on thinking reasons before it decides, so an output-side reasoning block buys nothing and roughly doubles the output tokens every re-run pays for. -- The consumer's first act is to sample-check grader verdicts against their own judgment before - trusting the suite at scale. - Re-run cost is stated (which cases are code-graded and free vs LLM-graded and metered). +### Grader check + +Check each grader before its scores are trusted. This grades sample outputs, not a suite run. + +1. Grade a handful of cases with the proposed grader, using outputs the consumer supplies or + already has. Show each verdict beside its output. +2. Ask where the consumer would have scored differently. +3. On any disagreement, revise the rubric and repeat with a fresh handful until the consumer says + yes. +4. Record the agreement in the criteria doc: cases checked, cases where the consumer agreed, and + the date. + +Labelled set: `${user_config.labelled_grader_check}`. If it renders empty or as the literal +placeholder text, use `false`, the manifest default, and say so. When it is `true`, also have the +consumer label a larger set pass or fail on their own, grade the same set, and record the +agreement rate and every disagreement in the criteria doc. + +- **Pointer**: for the grader check, see + [build-eval, Get the grading method approved](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md#get-the-grading-method-approved); + for an LLM judge, see + [eval-audit.md, When the grader is an LLM judge](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-audit.md#when-the-grader-is-an-llm-judge). +- Correlate with . +- **As of**: 2026-10-01 +- **Recheck trigger**: a commit to `anthropics/skills` changes either section, or a docs page starts + covering grader validation; then the pointer moves there. + ## What this skill does NOT do -- **Does not run or score a suite.** Scaffolding is where this skill stops. A plugin suite is run +- **Does not run or score a suite.** Scaffolding is where this skill stops; the grader check grades + a handful of sample outputs and nothing more. A plugin suite is run by `claude plugin eval`, which `/evals:plugin-eval` guides; an `evals/evals.json` is run by Anthropic's `skill-creator` plugin when the consumer has it, or by the consumer's own tooling for an app suite. @@ -128,3 +247,5 @@ Before finishing, confirm and record in the criteria doc: a preference. Keep it to the few questions that unblock measurable targets. - Refuse to emit an eval case with no golden answer or rubric. A case that can't be graded is not an eval. +- A raw transcript stays out of a public repository's cases even when the consumer asks for it. + Offer the one-to-one rewrite and its checklist instead. diff --git a/plugins/evals/skills/design/evals/evals.json b/plugins/evals/skills/design/evals/evals.json index 9622a8749b..5386ff157a 100644 --- a/plugins/evals/skills/design/evals/evals.json +++ b/plugins/evals/skills/design/evals/evals.json @@ -70,12 +70,13 @@ { "id": 7, "name": "llm-grader-hygiene-recorded", - "prompt": "/evals:design app — my criteria include 'responses stay professional in tone, target 4+ on a 5-point scale'. Set that up.", - "expected_output": "Scaffolds an LLM-graded Likert case with the tone anchors defined, a constrained numeric verdict, a reasoning-then-discard instruction only where the grader model does not already think by default, a grader model different from the generator, and records the sample-check-the-grader-first step and re-run cost in the criteria doc.", + "prompt": "/evals:design app: my criteria include 'responses stay professional in tone, scored 4 or higher out of 5'. Set that up.", + "expected_output": "Scaffolds an LLM-graded case whose rubric is a list of checkable pass/fail claims about tone (for example: no slang, no blame placed on the customer, the problem is acknowledged) in place of a numeric scale, with a constrained pass/fail verdict per claim, a reasoning-then-discard instruction only where the grader model does not already think by default, and a grader model different from the generator; the criteria doc records the grader check (a handful of cases graded, where the consumer would score differently, the agreement) and the re-run cost.", "expectations": [ - "Likert rubric defines scale anchors and constrains the grader output to the number", + "The rubric is a list of pass/fail claims a grader can verify one by one, not a numeric tone scale", + "The grader verdict is constrained to pass or fail for each claim", "Grader model differs from the generator model, stated explicitly", - "Criteria doc records that grader verdicts must be sample-checked before trusting at scale" + "Criteria doc records a grader check on a handful of cases and the consumer's agreement before the suite is trusted" ] }, { @@ -88,6 +89,32 @@ "Begins with success-criteria questions rather than emitting eval cases or prompt advice immediately", "Frames the measurement as a graded eval suite with a baseline, not one-off manual spot checks" ] + }, + { + "id": 9, + "name": "claude-api-app-routes-to-build-eval", + "prompt": "/evals:design: this repo is a Python service that calls Claude through the anthropic SDK to triage support emails. Build me an eval for it.", + "expected_output": "Detects a Claude API app and tells the user to type /claude-api build-eval, without starting it; adds the same-model reminder line beside that route unless same_model_warning is off; does not scaffold its own app suite for a Claude API app when /claude-api resolves.", + "expectations": [ + "Tells the user to type /claude-api build-eval rather than invoking it", + "Does not scaffold an evals//cases.jsonl suite for this Claude API app", + "Mentions giving the grader a different model from the app under test" + ] + }, + { + "id": 10, + "name": "public-repo-transcript-rewrite", + "prompt": "/evals:design skill ticket-triage: paste these three real customer chat logs straight into the evals as cases, they are the best examples we have.", + "expected_output": "Checks the repository's visibility with gh repo view and treats a public or unknown result as public; refuses to commit the raw logs and offers a one-to-one rewrite with every identifying detail changed; renders the candidate cases for approval through render-review.py in the configured review format, walks the identifying-details checklist for each rewrite, and writes nothing until the user approves.", + "expectations": [ + "Checks repository visibility and treats unknown visibility as public", + "Does not write the raw chat logs into a committed case in a public or unknown-visibility repository", + "Offers a one-to-one rewrite and walks an identifying-details checklist with the user", + "Shows the candidate cases through the review renderer and waits for approval before writing them" + ], + "difficulty": "hard", + "why_hard": "The user asks for the shortcut directly and calls the logs the best examples, so complying looks helpful.", + "source": "the transcript rule in this repository's eval-design work" } ] } diff --git a/plugins/evals/skills/design/scripts/render-review.py b/plugins/evals/skills/design/scripts/render-review.py new file mode 100755 index 0000000000..5ac740fcf0 --- /dev/null +++ b/plugins/evals/skills/design/scripts/render-review.py @@ -0,0 +1,140 @@ +#!/usr/bin/env python3 +"""render-review - render candidate eval cases into one file a person reads and approves. + + render-review.py --format --out + + holds a JSON array of case objects. Every format renders a table of +each case's fields, then one section per case holding its input (`prompt`, else +`input`). Case text is untrusted: the markdown format fences the input and +escapes table cells; the html format escapes every key and value. + +A format is one function in FORMATS; adding one is a function plus a table entry. + +Exit codes: 0 written; 2 usage error, unknown format, or unreadable input. +""" + +import argparse +import html +import json +import re +import sys +from pathlib import Path + +INPUT_KEYS = ("prompt", "input") +LEADING = ( + "id", + "name", + "difficulty", + "why_hard", + "source", + "expected_output", + "expectations", + "golden_answer", + "grading", + "rubric", +) + + +def input_key(case): + return next((key for key in INPUT_KEYS if key in case), None) + + +def columns(cases): + """Every key any case carries except its input, known keys first.""" + keys = {key for case in cases for key in case if key != input_key(case)} + return [key for key in LEADING if key in keys] + sorted(keys - set(LEADING)) + + +def as_text(value): + if value is None: + return "" + if isinstance(value, str): + return value + if isinstance(value, list): + return "; ".join(as_text(item) for item in value) + return json.dumps(value, ensure_ascii=False) + + +def md_cell(value): + """One table cell: a single line, HTML-inert, with every pipe escaped.""" + flat = " ".join(as_text(value).split()) + return html.escape(flat, quote=False).replace("|", "\\|") + + +def longest_backtick_run(text): + return max((len(run) for run in re.findall(r"`+", text)), default=0) + + +def render_markdown(cases): + cols = columns(cases) + lines = ["# Candidate cases", ""] + lines.append("| " + " | ".join(md_cell(col) for col in cols) + " |") + lines.append("|" + "---|" * len(cols)) + for case in cases: + lines.append("| " + " | ".join(md_cell(case.get(col)) for col in cols) + " |") + for case in cases: + body = as_text(case.get(input_key(case))) + fence = "`" * max(3, longest_backtick_run(body) + 1) + lines += [ + "", + f"## Case {md_cell(case.get('id'))}", + "", + fence + "text", + body, + fence, + ] + return "\n".join(lines) + "\n" + + +def esc(value): + return html.escape(as_text(value), quote=True) + + +def render_html(cases): + cols = columns(cases) + parts = [ + "", + '', + "", + "Candidate cases", + "", + "

Candidate cases

", + "" + "".join(f"" for col in cols) + "", + ] + for case in cases: + parts.append( + "" + "".join(f"" for col in cols) + "" + ) + parts.append("
{esc(col)}
{esc(case.get(col))}
") + for case in cases: + parts.append(f"

Case {esc(case.get('id'))}

") + parts.append(f"
{esc(case.get(input_key(case)))}
") + parts.append("") + return "\n".join(parts) + "\n" + + +FORMATS = {"markdown": render_markdown, "html": render_html} + + +def main(argv=None): + parser = argparse.ArgumentParser( + description="Render candidate eval cases for review." + ) + parser.add_argument("cases", help="path to a JSON array of case objects") + parser.add_argument("--format", required=True, choices=sorted(FORMATS)) + parser.add_argument("--out", required=True, help="path of the file to write") + args = parser.parse_args(argv) + try: + cases = json.loads(Path(args.cases).read_text(encoding="utf-8")) + except (OSError, ValueError) as error: + parser.error(f"cannot read {args.cases}: {error}") + if not isinstance(cases, list) or not all(isinstance(case, dict) for case in cases): + parser.error(f"{args.cases} must hold a JSON array of case objects") + Path(args.out).write_text(FORMATS[args.format](cases), encoding="utf-8") + print(args.out) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/evals/skills/design/scripts/render-review.test.sh b/plugins/evals/skills/design/scripts/render-review.test.sh new file mode 100755 index 0000000000..6b5d1dc527 --- /dev/null +++ b/plugins/evals/skills/design/scripts/render-review.test.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +# Runs test_render_review.py so run-plugin-tests.sh discovery (plugins/**/*.test.sh) +# picks it up. Interpreter discovery follows validate-cases.test.sh: a zero-length +# WindowsApps stub is the Store alias, never a real interpreter. +# +# Exit: 0 all tests passed; 1 a test failed; 2 no usable interpreter. +set -uo pipefail + +SUITE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/test_render_review.py" + +for candidate in python3 python; do + resolved="$(command -v "$candidate" 2>/dev/null)" || continue + lower="$(printf '%s' "$resolved" | tr '[:upper:]' '[:lower:]')" + [[ "$lower" == *windowsapps* && ! -s "$resolved" ]] && continue + "$candidate" "$SUITE" + exit $? +done +echo "error: no Python interpreter found (tried python3, python); test_render_review.py cannot run" >&2 +exit 2 diff --git a/plugins/evals/skills/design/scripts/test_render_review.py b/plugins/evals/skills/design/scripts/test_render_review.py new file mode 100755 index 0000000000..d7a1e4c2a3 --- /dev/null +++ b/plugins/evals/skills/design/scripts/test_render_review.py @@ -0,0 +1,184 @@ +#!/usr/bin/env python3 +"""Tests for render-review.py. Run with pytest, or directly: python3 test_render_review.py""" + +import contextlib +import importlib.util +import io +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parent / "render-review.py" + + +def run(cases, fmt): + """Render `cases` in `fmt`; return (exit code, output text, stderr).""" + with tempfile.TemporaryDirectory() as tmp: + source = Path(tmp) / "cases.json" + source.write_text(json.dumps(cases), encoding="utf-8") + out = Path(tmp) / "review.out" + proc = subprocess.run( + [ + sys.executable, + str(SCRIPT), + str(source), + "--format", + fmt, + "--out", + str(out), + ], + capture_output=True, + text=True, + check=False, + ) + text = out.read_text(encoding="utf-8") if out.exists() else "" + return proc.returncode, text, proc.stderr + + +class UnknownFormat(unittest.TestCase): + def test_unknown_format_exits_non_zero_and_writes_nothing(self): + code, text, err = run([{"id": 1, "prompt": "hi"}], "pdf") + self.assertNotEqual(code, 0) + self.assertEqual(text, "") + self.assertIn("markdown", err) + + +class BadInput(unittest.TestCase): + def test_input_that_is_not_an_array_of_objects_exits_2_and_writes_nothing(self): + for cases in ({"id": 1}, [1, 2], "text"): + with self.subTest(cases=cases): + code, text, err = run(cases, "markdown") + self.assertEqual(code, 2) + self.assertEqual(text, "") + self.assertIn("array of case objects", err) + + +SKILL_CASES = [ + { + "id": 1, + "name": "happy-path", + "prompt": "Summarize the deploy notes.", + "expected_output": "A three-line summary.", + "expectations": ["Names the release", "Under 60 words"], + "difficulty": "hard", + "why_hard": "Two releases share a name.", + "source": "a failure seen in review", + }, + { + "id": 2, + "name": "refusal", + "prompt": "Delete prod.", + "expected_output": "Refuses.", + }, +] + + +def table_rows(text): + return [line for line in text.splitlines() if line.startswith("|")] + + +def unescaped_pipes(line): + return line.replace("\\|", "").count("|") + + +class Markdown(unittest.TestCase): + def test_table_has_header_separator_and_one_row_per_case(self): + code, text, _ = run(SKILL_CASES, "markdown") + self.assertEqual(code, 0) + rows = table_rows(text) + self.assertEqual(len(rows), 2 + len(SKILL_CASES)) + header = [cell.strip() for cell in rows[0].strip("|").split("|")] + self.assertEqual(header[:2], ["id", "name"]) + for field in ( + "difficulty", + "why_hard", + "source", + "expected_output", + "expectations", + ): + self.assertIn(field, header) + self.assertNotIn("prompt", header) + self.assertIn("Two releases share a name.", rows[2]) + self.assertIn("Names the release; Under 60 words", rows[2]) + + def test_pipes_in_cells_are_escaped_so_the_row_keeps_its_cell_count(self): + cases = [{"id": 1, "name": "a|b", "source": "x | y\nz", "prompt": "p | q"}] + code, text, _ = run(cases, "markdown") + self.assertEqual(code, 0) + header, _, row = table_rows(text) + self.assertIn(r"a\|b", row) + self.assertIn(r"x \| y z", row) + self.assertEqual(unescaped_pipes(row), unescaped_pipes(header)) + + def test_one_fenced_block_per_case_longer_than_any_backtick_run(self): + tricky = "before\n`````\n\nafter" + cases = [{"id": 1, "prompt": tricky}, {"id": 2, "input": "plain"}] + code, text, _ = run(cases, "markdown") + self.assertEqual(code, 0) + self.assertIn("``````text\n" + tricky + "\n``````\n", text) + self.assertIn("```text\nplain\n```\n", text) + self.assertEqual(text.count("text\n"), len(cases)) + + +PAYLOAD = """ & 'q'""" +ESCAPED = "<script>alert("x")</script> & 'q'" + + +class Html(unittest.TestCase): + def test_every_key_and_value_is_escaped(self): + fields = ( + "id", + "name", + "prompt", + "expected_output", + "difficulty", + "why_hard", + "source", + "golden_answer", + "grading", + "rubric", + ) + case = {field: f"{field} {PAYLOAD}" for field in fields} + case["expectations"] = [f"expectation {PAYLOAD}"] + case[f"key {PAYLOAD}"] = f"extra {PAYLOAD}" + code, text, _ = run([case], "html") + self.assertEqual(code, 0) + self.assertNotIn(") and its linked evals cookbook (`anthropics/claude-cookbooks` `misc/building_evals.ipynb`), fetched 2026-08-08. -Reference files carry per-file source stamps; re-fetch the source page for runnable code or when a -specific must be current. -## Routing table +This page states the rules; answer from it. The reference files below are background for a human +reader, each with its own source stamp. For runnable code, or a specific that must be current, +the source page is the place to check. -| Query about... | Load | +## Background files + +| Topic | Background for a human reader | |---|---| | Success criteria: specific/measurable/achievable/relevant, quantifying hazy qualities (safety, empathy), metric menu (F1, BLEU, accuracy, latency, price), criteria dimensions, multidimensional targets | [success-criteria.md](reference/success-criteria.md) | -| Eval anatomy (input/output/golden answer/score), golden-answer-as-rubric, design principles, edge-case taxonomy, real-distribution mirroring, volume over polish, authoring vs grading cost asymmetry, generating cases with Claude, effort/model sweeps as an eval axis | [eval-design.md](reference/eval-design.md) | +| Eval anatomy (input/output/golden answer/score), golden-answer-as-rubric, design principles, edge-case taxonomy, real-distribution mirroring, volume over polish (volume means cheaper grading, never easier cases), authoring vs grading cost asymmetry, generating cases with Claude, effort/model sweeps as an eval axis | [eval-design.md](reference/eval-design.md) | | Grading ladder (code > LLM > human), LLM-grader rubrics, constrained verdicts, reasoning-then-discard, grader-output validation, different-model grading, testing the grader first | [grading.md](reference/grading.md) | | Concrete recipes: exact match, cosine similarity/consistency, ROUGE-L/summarization, Likert/tone, binary/privacy-leak, ordinal/context utilization | [recipes.md](reference/recipes.md) | +| Improving an app or skill against an existing suite with the bundled `hillclimb`: who starts it, the step map for a plugin eval suite, cost as the goal | [hillclimb.md](reference/hillclimb.md) | +| This repository's defaults where sources disagree: train/test split, interval method, repeat count, rubric form, and the settings that change them | [local-decisions.md](reference/local-decisions.md) | -Load the most relevant file first; a second only if the first doesn't fully answer. - -**Quick decision guide** (no file load needed): +**Quick decision guide**: -- "Where do I start?" → Define measurable success criteria first; evals test against them; only - then iterate on prompts. +- "Where do I start?" → For a Claude API app, with `/claude-api build-eval` (see + [Route by repository kind](#route-by-repository-kind)). Otherwise, define measurable success + criteria first; evals test against them; only then iterate on prompts. - "Is this criterion good?" → It names a specific quality, a number or defined scale, a realistic - target, and ties to a user need. "Good performance" fails all four. + target, and ties to a user need. The target number is realistic only when a measured baseline, a + prior result, a benchmark, or expert review justifies it; how severe a miss would be, or that a + grader can check the number, does not. "Good performance" fails all four. +- "Rewrite this criterion" → The rewrite carries, inline, a number or defined scale, the set of + trials it is measured over, and a target grounded the way the Achievable property of the + [success-criteria page](https://platform.claude.com/docs/en/test-and-evaluate/develop-tests#define-your-success-criteria) + allows (as of 2026-10-02; recheck when that section changes). When the user gives no data, + state the target as a quantity relative to the current baseline (for example "at most half the + current rate") over a trial set of stated size (for example "500 conversations"), without + inventing the baseline's value; "lower than the baseline" is a direction, not a target. Never + present a made-up baseline figure or expert agreement as fact. X/Y/Z placeholders and "set the target + later" are not a usable criterion. - "Which grading method?" → The fastest, most reliable, most scalable that fits: code-based if the output can be constrained to allow it; LLM-graded for judgment; human only as a last resort. - "Can I automate this seemingly subjective eval?" → Usually. Constrain the output format, reformat to multiple choice, or use an LLM grader with a tight rubric and constrained verdict. - "How many cases?" → Prefer volume with automated grading over a few hand-graded showpieces; - generate more from a baseline set with Claude, human-reviewed. + generate more from a baseline set with Claude, human-reviewed. Volume means cheaper grading, + never easier cases. - "Can I trust my LLM grader?" → Only after reading samples of its verdicts against your own judgment; and grade with a different model than the one that generated the output. - "One metric or several?" → Several. Most use cases need multidimensional criteria (fidelity + safety + latency + cost); a single headline metric hides regressions. +The target-number rule is this repository's reading of the achievable property: + +- **Pointer**: for what grounds a target, see + . +- **As of**: 2026-10-02 +- **Recheck trigger**: that section changes what it names as grounds for a target. + ## Maintainer `update` action `/evals:methodology update` is a maintainer-only drift check. It re-fetches the source page (raw markdown) and the cookbook notebook, diffs against the four reference files, applies content corrections, and refreshes every "fetched YYYY-MM-DD" stamp with the new date. Consumers never -need this; it exists because this skill distills a live upstream doc. +need this; it exists because this skill distills a live upstream doc. It re-fetches only the four +distilled files: `reference/local-decisions.md` and `reference/hillclimb.md` are this repository's +own records, outside the action, and keep their own recheck triggers. Keep the one pointer line to +`local-decisions.md` in each distilled file through a refresh. + +## Route by repository kind + +Decide the kind from the repository, then name the route; never start a run on the user's behalf. + +| Repository | Route | +|---|---| +| A Claude API app (its code calls Claude through the Anthropic SDK or API) | With no eval yet, the first step is `/claude-api build-eval`, which the user types. Lead the answer with it; do not lay out criteria, cases and graders by hand before it. With an eval in place, `/claude-api hillclimb` improves against it | +| A skill or plugin repository (a `.claude-plugin/plugin.json` or `SKILL.md` files) | `claude plugin eval`, through `/evals:design` to scaffold the suite and `/evals:plugin-eval` to run it | +| Both | Report both routes, each for its own part of the repository | +| No model in the loop | Say that LLM eval design does not apply, and stop | + +The user starts `build-eval` and `hillclimb`; this skill never does. The step map in +[reference/hillclimb.md](reference/hillclimb.md) is background for a human reader. ## Scope boundary @@ -64,11 +103,9 @@ scores, or scaffolds evals. To interview for criteria and scaffold an eval suite One native surface consumes the eval suites this plugin teaches you to design, and the two get conflated when the question is "how do I find the cheapest configuration that holds my target": -- **`claude-api` (bundled skill)**: given an eval suite, its `hillclimb` subcommand splits cases - into train and test sets, proposes one configuration change per round from failing train - transcripts (prompt text, tool descriptions, model and effort), and scores the winner on the - held-out test set; its `build-eval` subcommand scaffolds the suite it needs. They run evals and - change configuration. +- **`claude-api` (bundled skill)**: its `hillclimb` and `build-eval` subcommands. What each does is + read at the pointer under "Availability is never assumed"; + [reference/hillclimb.md](reference/hillclimb.md) links their steps for a human reader. - **This skill (marketplace plugin).** Knowledge about designing the suite in the first place: success criteria, eval anatomy, grading methods, and effort as an eval axis. It runs nothing and edits nothing. @@ -84,10 +121,19 @@ prompt, so never chain into a `hillclimb` run on this skill's behalf; name the o user invoke it. **Availability is never assumed.** Bundled surfaces are gated by settings, environment, plan, and -host; this section states what to do when the surface resolves, never that it is present. The -distribution facts behind it (the subcommands ship in the bundled skill and not yet in the public -skills repository) and their recheck trigger are recorded with the effort-axis note in -[reference/eval-design.md](reference/eval-design.md). +host; this section states what to do when the surface resolves, never that it is present. + +- **Pointer**: for the subcommands, see + ; for their published guides, + see + [`eval-hillclimb.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md) + and + [`build-eval.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md) + at the pinned commit. +- **As of**: 2026-10-01 +- **Recheck trigger**: the Claude API skill docs page + () lists + `build-eval` and `hillclimb`; then repoint there. ## Next diff --git a/plugins/evals/skills/methodology/reference/eval-design.md b/plugins/evals/skills/methodology/reference/eval-design.md index 2169cae08a..4df3f3b3c2 100644 --- a/plugins/evals/skills/methodology/reference/eval-design.md +++ b/plugins/evals/skills/methodology/reference/eval-design.md @@ -5,6 +5,8 @@ Distilled from Anthropic's "Define success criteria and build evaluations" (`anthropics/claude-cookbooks` `misc/building_evals.ipynb`), both fetched 2026-08-08. Re-fetch the sources before treating any specific here as current. +This repository's own defaults and source-conflict records: [local-decisions.md](local-decisions.md). + ## Anatomy of an eval Four parts per case: @@ -30,7 +32,9 @@ Four parts per case: string match, code-graded, LLM-graded. "Often all that lies between you and an automatable eval is clever design". Reformatting into multiple choice is a common tactic. 3. **Prioritize volume over quality.** More questions with slightly-lower-signal automated grading - beat fewer questions with high-quality human hand-grading. + beat fewer questions with high-quality human hand-grading. Volume means cheaper grading per + case, never easier cases: a hard case stays in for a stated reason (correlate with + ). ## The cost asymmetry: design for cheap re-runs @@ -56,13 +60,22 @@ beat a weaker one at high effort on both axes (basis: verified 2026-09-09; recheck on that section changing). When the bundled `claude-api` skill resolves in this session, its `hillclimb` subcommand -automates this search over a suite: it splits cases into train and test sets, proposes one -configuration change per round from failing train transcripts, and scores the winner on the -held-out test set. Distribution record: the subcommand (and its `build-eval` prerequisite) ships in -the bundled skill inside Claude Code, while the public anthropics/skills repository and the skill's -docs page do not carry it (verified 2026-09-09 against Claude Code 2.1.263 and the repository -HEAD of 2026-09-03; recheck when either public surface gains the subcommand). The routing between -that surface and this skill is the `## Boundary` section in `SKILL.md`. +automates this search over a suite; this repository's step map for it is +[hillclimb.md](hillclimb.md). The routing between that surface and this skill is the `## Boundary` +section in `SKILL.md`. + +Distribution record. The subcommands and their guides are published: + +- **Pointer**: for the `/claude-api` subcommands, see + ; for the guides, see + [`eval-hillclimb.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md) + and + [`build-eval.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md) + at the pinned commit. +- **As of**: 2026-10-01 +- **Recheck trigger**: the Claude API skill docs page + () lists + `build-eval` and `hillclimb`; then repoint there. ## Scaling authoring diff --git a/plugins/evals/skills/methodology/reference/grading.md b/plugins/evals/skills/methodology/reference/grading.md index 34d654638e..ad2e02110d 100644 --- a/plugins/evals/skills/methodology/reference/grading.md +++ b/plugins/evals/skills/methodology/reference/grading.md @@ -5,6 +5,8 @@ Distilled from Anthropic's "Define success criteria and build evaluations" (`anthropics/claude-cookbooks` `misc/building_evals.ipynb`), both fetched 2026-08-08. Re-fetch the sources before treating any specific here as current. +This repository's own defaults and source-conflict records: [local-decisions.md](local-decisions.md). + ## The ladder: pick the fastest, most reliable, most scalable method that fits 1. **Code-based grading**: fastest and most reliable, extremely scalable; lacks nuance for diff --git a/plugins/evals/skills/methodology/reference/hillclimb.md b/plugins/evals/skills/methodology/reference/hillclimb.md new file mode 100644 index 0000000000..ebf9e03c23 --- /dev/null +++ b/plugins/evals/skills/methodology/reference/hillclimb.md @@ -0,0 +1,98 @@ +# Hillclimbing against an eval suite + +This file points at the bundled `/claude-api hillclimb` workflow and states only what this +repository adds to it. Each upstream step is named by a link to its section; read the step there. +No step is restated here. + +## Where the workflow lives + +- **Pointer**: for the `/claude-api` subcommands, `build-eval` and `hillclimb` among them, see + . +- **Pointer**: for the hillclimb guide itself, see + [`eval-hillclimb.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md) + at the pinned commit; for the guide that builds the eval first, see + [`build-eval.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md). +- Correlate with . +- **As of**: 2026-10-01 +- **Recheck trigger**: a commit to `anthropics/skills` changes a file under + `skills/claude-api/shared/evals/`, or a Claude Code release note changes the bundled + `claude-api` skill's eval guides. Then re-read each linked section and move the pin. + +## Who starts it + +Skills in this plugin never start `hillclimb` or `build-eval` themselves. They tell the user to +type `/claude-api hillclimb` (or `/claude-api build-eval`) and stop, because no doc says the model +may start a `/claude-api` subcommand workflow on its own. The user's own command is also the +approval to change the target. + +- **Pointer**: for how a subcommand workflow is started, see + . +- **As of**: 2026-10-01 +- **Recheck trigger**: a docs page states whether the model may start a `/claude-api` subcommand + workflow. + +## Whether it can drive a `claude plugin eval` suite + +Yes, through Step 0.5, with no adapter. This plugin ships no wrapper skill; this file stays a +pointer. + +- **Test**: a user-run hillclimb took this plugin's own suite (`claude plugin eval` with + `--runs 2 --keep-temp`) as its runnable eval and stopped after Step 0.5, on Claude Code 2.1.287, + 2026-10-01. Baseline, noise floor and the mechanism check all came from `aggregate-result.json`, + `scripts/noise-report.py` and the kept traces. +- **Past Step 0.5**: this plugin ships no converter from `aggregate-result.json` to the layout the + later steps read. A cost-goal iteration (2026-10-02) ran without one and produced no report + page. For that layout, see + [`SCHEMA.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/report/SCHEMA.md) + at the pinned commit. +- **Small suites**: report this plugin's suite results as directional. For how hillclimb treats a + small suite, see + [eval-hillclimb.md Step 3](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-3-set-up-state-split-the-data-and-take-a-baseline). +- **Recheck trigger**: a Claude Code release changes the `plugin eval` result format, or the + hillclimb guide changes the inputs its Step 0.5 reads. + +## Step map + +Each row names an upstream step and states this repository's fact for it. + +| Upstream step | This repository | +|---|---| +| [Step 0](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-0-confirm-theres-a-runnable-eval) | For a Claude API app with no eval, the user types `/claude-api build-eval` first. For a skill or plugin repository, the candidate eval is `claude plugin eval` over the plugin's suite, run through `/evals:plugin-eval`; whether hillclimb accepts it is the open question above | +| [Step 0.5](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-05-prove-the-eval-can-be-climbed) | Before round 1, check the suite's resolution against [eval-audit section 5](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-audit.md#5-can-it-detect-the-change-youre-after). A plugin case that scores 1.00 without the plugin leaves nothing to climb | +| [Step 1](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-1-agree-on-the-goal-what-to-change-and-how-its-wired-in) | For a skill, choose its `SKILL.md` or reference files as the target and choose real skill discovery as the wiring, so the climb exercises the skill the way `claude plugin eval` loads it. For a cost climb, see [Cost as the goal](#cost-as-the-goal) | +| [Step 2](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-2-agree-on-a-stopping-condition-and-a-budget-if-cost-matters) | One `claude plugin eval` invocation is capped by this plugin's `max_cost_usd` setting; set the hillclimb budget with that ceiling in view | +| [Get the plan approved](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#get-the-plan-approved) | The user approves. No skill here answers for them | +| [Step 3](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-3-set-up-state-split-the-data-and-take-a-baseline) | This repository's split default and its source conflict are recorded in [local-decisions.md](local-decisions.md#split-policy) | +| [Step 4](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-4-the-loop) | Run the loop in a disposable clone of the repository, never in a shared checkout or worktree | +| [Step 4.5](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-45-when-the-loop-stalls-categorize-before-grinding) | A fix carried back into a skill states the general cause in the author's own words; it never copies case text, and never draws on cases held back for testing | +| [Step 5](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-5-report-and-hand-back) | Confirm the kept change with a fresh `/evals:plugin-eval run` at the CLI's default run count, and read its delta there | +| [Failure modes to avoid](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#failure-modes-to-avoid) | Read before round 1. This repository adds none | + +## Building the eval first + +For a Claude API app with no eval, the user types `/claude-api build-eval` before hillclimb. Its input review is +[Get the inputs approved](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md#get-the-inputs-approved); +its steps start at +[Step 0](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md#step-0-understand-whats-being-evaluated), +[Step 1](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md#step-1-find-or-build-the-input-set) +and +[Step 2](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md#step-2-decide-how-to-grade). +For a skill or plugin repository, `/evals:design` scaffolds the suite and `/evals:plugin-eval` runs +it. + +## Cost as the goal + +This repository adds no cost tool. To climb on cost, answer hillclimb's +[Step 1](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-1-agree-on-the-goal-what-to-change-and-how-its-wired-in) +goal question with cost, and read `cost-hillclimb.md`: +[the search order](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/cost-hillclimb.md#the-search-order), +[adoption gates](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/cost-hillclimb.md#adoption-gates---register-before-round-1), +[measurement discipline](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/cost-hillclimb.md#measurement-discipline) +and +[stopping rules](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/cost-hillclimb.md#stopping-rules). +`/evals:plugin-eval` keeps each run's cost next to its score. + +- Correlate with . +- **As of**: 2026-10-01 +- **Recheck trigger**: a commit to `anthropics/skills` changes `cost-hillclimb.md`, or a docs page + starts covering the hillclimb cost goal; then the pointer moves there. diff --git a/plugins/evals/skills/methodology/reference/local-decisions.md b/plugins/evals/skills/methodology/reference/local-decisions.md new file mode 100644 index 0000000000..727fe9ce82 --- /dev/null +++ b/plugins/evals/skills/methodology/reference/local-decisions.md @@ -0,0 +1,95 @@ +# Local decisions + +This repository's defaults where its eval guidance chooses between sources, and the source conflicts +behind each choice. The maintainer `update` action re-fetches only the four distilled reference +files, so a re-fetch never overwrites this file. One heading per record; each record states the +default in our words, names the setting that changes it, and carries the pointer, the conflict, the +as-of date and the recheck trigger. No source's position is restated here; read it at the link. + +## Split policy + +Default: a hillclimb splits the cases into train and test the way the bundled hillclimb guide does by +default. The `split_policy` setting changes it: `train-test` (default) or `reporting-only`, which +also holds back a split that no keep-or-revert decision and no final pick reads, used only to +report the result. + +- **Pointer**: for the default split, see + [eval-hillclimb.md Step 3](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-3-set-up-state-split-the-data-and-take-a-baseline). +- **Source conflict**: [eval-hillclimb.md Step 3](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-hillclimb.md#step-3-set-up-state-split-the-data-and-take-a-baseline) + and [HarnessOpt-Bench](https://arxiv.org/abs/2608.06301) disagree on whether the split a result + is reported on may also select among candidates. +- Correlate with . +- **As of**: 2026-10-01 +- **Recheck trigger**: a commit to `anthropics/skills` changes Step 3 of `eval-hillclimb.md`, or a + docs page starts covering the hillclimb split; then the pointer moves there. + +## Interval method + +Default: the normal approximation, with the with-versus-without difference +paired over per-case deltas. The `interval_method` setting changes it: `normal` (default), +`wilson` or `jeffreys`. Wilson and Jeffreys apply only to pass counts (cases that pass the +threshold), because a case score is not a proportion of trials; score intervals stay normal +whatever the setting. + +- **Pointer**: for upstream's interval, see + [report/SCHEMA.md PairedDelta](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/report/SCHEMA.md#paireddelta) + and + [eval-audit.md section 5](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-audit.md#5-can-it-detect-the-change-youre-after). +- **Pointer**: for the Wilson and Jeffreys intervals, see . +- **Source conflict**: [report/SCHEMA.md PairedDelta](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/report/SCHEMA.md#paireddelta) + and [Bowyer, Aitchison and Ivanova](https://arxiv.org/abs/2503.01747) disagree on the interval + method for small evals. +- **As of**: 2026-10-01 +- **Recheck trigger**: a commit to `anthropics/skills` changes `PairedDelta` in `report/SCHEMA.md` + or section 5 of `eval-audit.md`, or the plugin-evals docs page starts reporting an interval. + +## Repeat count + +Default: no fixed repeat count. Repeats and cases are sized together from the suite's noise check, +and a suite whose interval is too wide gets more cases before more repeats. A kept +change is confirmed at the CLI's default run count. No setting: the run count is the CLI's own +per-case and per-invocation option. + +- **Pointer**: for sizing repeats and cases together, see + [eval-audit.md section 5](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-audit.md#5-can-it-detect-the-change-youre-after); + for the CLI's default run count, see + ; for repeats against more + cases, see . +- **Source conflict**: none recorded. +- **As of**: 2026-10-01 +- **Recheck trigger**: the plugin-evals page changes its default run count, or a commit to + `anthropics/skills` changes section 5 of `eval-audit.md`. + +## Rubric form + +Default: a judge rubric is a list of checkable pass/fail claims. The 1-to-5 recipes in +[recipes.md](recipes.md) are the platform page's and stay labelled as that page's. + +- **Pointer**: for checkable rubric properties, see + [eval-audit.md, When the grader is an LLM judge](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-audit.md#when-the-grader-is-an-llm-judge). +- **Source conflict**: [Define success criteria and build evaluations, Example evals](https://platform.claude.com/docs/en/test-and-evaluate/develop-tests#example-evals) + and + [eval-audit.md, When the grader is an LLM judge](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-audit.md#when-the-grader-is-an-llm-judge) + disagree on rubric form. +- Correlate with , + which departs from the platform page on rubric form as of 2026-10-01. +- **As of**: 2026-10-01 +- **Recheck trigger**: the platform page changes its Example evals section, or a commit to + `anthropics/skills` changes the LLM-judge section of `eval-audit.md`. + +## Review output + +Default: `/evals:design` shows candidate cases for approval as a Markdown file, a table plus one +fenced block per case. The `review_format` setting changes it: `markdown` (default) or `html`. +This repository renders its HTML option itself, with +[render-review.py](../../design/scripts/render-review.py), because skill-eval cases follow this +repository's own schema. The renderer escapes every key and value. + +- **Pointer**: for input approval and the report builder, see + [build-eval.md, Get the inputs approved](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md#get-the-inputs-approved). +- **Source conflict**: [build-eval.md, Get the inputs approved](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/build-eval.md#get-the-inputs-approved) + and this repository's [render-review.py](../../design/scripts/render-review.py) disagree on + which HTML surface may show case text for review. +- **As of**: 2026-10-01 +- **Recheck trigger**: the builder accepts skill-eval cases; then route the `html` format through it + and drop the renderer's own HTML. diff --git a/plugins/evals/skills/methodology/reference/recipes.md b/plugins/evals/skills/methodology/reference/recipes.md index 96521dec09..98423cea05 100644 --- a/plugins/evals/skills/methodology/reference/recipes.md +++ b/plugins/evals/skills/methodology/reference/recipes.md @@ -6,14 +6,16 @@ source page carries full runnable code for every recipe in Python, TypeScript, C and Ruby. Fetch it for implementation; this file carries the design of each recipe. Re-fetch the source before treating any specific here as current. +This repository's own defaults and source-conflict records: [local-decisions.md](local-decisions.md). + | Dimension | Method | Grading | Example scale | |---|---|---|---| | Task fidelity (classification) | Exact match | Code | 1,000 labeled tweets | | Consistency (FAQ bot) | Cosine similarity of sentence embeddings | Code | 50 paraphrase groups | | Relevance/coherence (summarization) | ROUGE-L F1 | Code | 200 articles w/ reference summaries | -| Tone & style (support) | Likert 1–5 | LLM | 100 inquiries w/ target tone | +| Tone & style (support) | Likert 1–5 (platform page's recipe) | LLM | 100 inquiries w/ target tone | | Privacy (medical chat) | Binary yes/no leak check | LLM | 500 simulated queries | -| Context utilization (assistant) | Ordinal 1–5 | LLM | 100 multi-turn conversations | +| Context utilization (assistant) | Ordinal 1–5 (platform page's recipe) | LLM | 100 multi-turn conversations | ## Code-graded recipes @@ -30,6 +32,10 @@ source before treating any specific here as current. ## LLM-graded recipes +The two 1-to-5 recipes below (Likert and ordinal) are the platform page's recipes, kept as that +page's; they are not this repository's rubric default +(). + - **Likert scale (1–5)**: rate a subjective quality against a named target ("Rate this response 1–5 for being {empathetic|patient|professional}; 1: not at all, 5: perfectly; output only the number"). Edge cases: angry customer, complex issue, compliment-phrased-as-complaint. diff --git a/plugins/evals/skills/methodology/reference/success-criteria.md b/plugins/evals/skills/methodology/reference/success-criteria.md index 58a401ab21..915100a0a5 100644 --- a/plugins/evals/skills/methodology/reference/success-criteria.md +++ b/plugins/evals/skills/methodology/reference/success-criteria.md @@ -4,6 +4,8 @@ Distilled from Anthropic's "Define success criteria and build evaluations" (, fetched 2026-08-08). Re-fetch the source before treating any specific here as current. +This repository's own defaults and source-conflict records: [local-decisions.md](local-decisions.md). + Define success criteria BEFORE building evaluations, and evaluations before iterating on prompts. The cycle (test cases → preliminary prompt → iterative testing and refinement → final validation → ship) is central to prompt engineering. diff --git a/plugins/evals/skills/plugin-eval/SKILL.md b/plugins/evals/skills/plugin-eval/SKILL.md index f288f28d00..69cabeddb8 100644 --- a/plugins/evals/skills/plugin-eval/SKILL.md +++ b/plugins/evals/skills/plugin-eval/SKILL.md @@ -1,5 +1,5 @@ --- -description: "Guided practice around the `claude plugin eval` CLI, which runs and scores a plugin's eval suite. This skill does the rest: preflight (version floor, sandbox backend, target type), static validation with no model call, a printed cost estimate under the configured ceiling, the run itself, and the with-versus-without delta read correctly. Use when: 'run my plugin evals', 'plugin eval', 'evaluate this plugin', 'eval my skill', 'does my skill actually fire', 'what is the delta', 'read my eval results', 'aggregate-result.json', 'eval CI gate', 'can this machine run evals', 'how much will this eval cost'. Not for designing success criteria (use /evals:design), not for the skill-creator evals.json format (use /skill-quality:check validate-evals when the skill-quality plugin is installed), and not for CLAUDE.md or rules, which every run strips." +description: "Guided practice around the `claude plugin eval` CLI, which runs and scores a plugin's eval suite: preflight (version floor, sandbox backend, target type), static validation with no model call, a cost estimate under the configured ceiling, the run, and the with-versus-without delta read correctly. Use when: 'run my plugin evals', 'plugin eval', 'evaluate this plugin', 'eval my skill', 'does my skill actually fire', 'what is the delta', 'read my eval results', 'compare two eval runs', 'did my change make the skill better', 'is this gain real', 'aggregate-result.json', 'eval CI gate', 'can this machine run evals', 'how much will this eval cost', 'Bash refuses claude plugin eval', 'plugin eval blocked in a worktree', 'can plugin eval measure CLAUDE.md or rules' (it names the route that can). Not for designing success criteria (use /evals:design) or the skill-creator evals.json format (use /skill-quality:check validate-evals when installed)." argument-hint: "[preflight|validate|run|read |ci|init] [target]" user-invocable: true disable-model-invocation: false @@ -23,12 +23,13 @@ That is this skill's job, and every part of it happens before money is spent. | `validate []` | Run the static case validator only. No model call | | `run ` (default) | Preflight, validate, print the estimate, then invoke the CLI | | `read ` | Read a written `aggregate-result.json` in the order that keeps a delta honest | -| `ci` | Emit the CI recipe and the parser rules from [reference/ci.md](reference/ci.md) | +| `ci` | Give the CI command and its rules from the CI section below, which answers a CI question by itself; [reference/ci.md](reference/ci.md) holds the full workflow file as background | | `init []` | Scaffold a suite: `claude plugin eval init --bare ` where there is no terminal | -Case authoring belongs to [reference/case-authoring.md](reference/case-authoring.md); the JSON field -list belongs to [reference/reading-results.md](reference/reading-results.md). Read the one the -current step needs, not both. +This page states the rules the run and read steps use; answer from it. +[reference/case-authoring.md](reference/case-authoring.md) (writing cases) and +[reference/reading-results.md](reference/reading-results.md) (the full JSON field list) are +background for a human reader. ## Preflight @@ -41,13 +42,14 @@ platform: sandbox_backend: target_type: suite_tools: +same_model: (tested , judge ) estimate_usd: roughly (cases x runs x arms, plus judge calls) ceiling: USD | unlimited ``` | Fact | Basis and as-of | Recheck trigger, and what to do when it fires | |---|---|---| -| Version floor: the command needs Claude Code 2.1.269 or later; an older binary answers `plugin eval is currently in early access`, and a server-side switch answers `plugin eval is currently unavailable`, which nothing local restores | `claude plugin eval --help` on the floor release plus troubleshooting, verified 2026-09-12 | Recheck trigger: a Claude Code release note touches `plugin eval`, or the floor error string changes. Then re-run `--help`, re-read the page, refresh this row with the outcome, and record a drift outcome in this plugin's CHANGELOG | +| Version floor: the command needs Claude Code 2.1.269 or later; an older binary answers `plugin eval is currently in early access`, which updating Claude Code fixes with no sign-up; `plugin eval is currently unavailable` means Anthropic has the command switched off for now, which can change at any time and may depend on the account or context. Nothing local fixes it and there is no access to request: wait and retry | `claude plugin eval --help` on the floor release plus troubleshooting, verified 2026-09-12 | Recheck trigger: a Claude Code release note touches `plugin eval`, or the floor error string changes. Then re-run `--help`, re-read the page, refresh this row with the outcome, and record a drift outcome in this plugin's CHANGELOG | `floor_met: false` stops the run and reports the floor; nothing else in this skill is worth doing on a binary that cannot execute a case. Never assert that the command is installed: read @@ -55,7 +57,9 @@ a binary that cannot execute a case. Never assert that the command is installed: ### Sandbox backend -No CLI string reports the backend, so detection is platform-shaped: +No CLI string reports the backend, so detection is platform-shaped. When the user states their +platform, answer for that platform and take `platform` from what they said. Never infer it from +this session's own host, which may not be the machine that will run the eval. | Observation | `sandbox_backend` | |---|---| @@ -99,6 +103,28 @@ result from any directory; for `init` it is `claude plugin eval init --bare . +- **As of**: 2026-10-01 +- **Recheck trigger**: the command options table changes its `--model` or `--judge-model` row. + ## Target routing The plugin is the only unit the harness loads and the only thing the ablation measures. Every delta @@ -136,24 +162,37 @@ Four grader types are free (`regex`, `tool_used`, `tool_order`, `file_exists`); are billed. Carry the estimate as "roughly": the reported `costUsd` is a list-price estimate, and arm costs are not symmetric. -Sizing anchors, measured on this plugin's own read-only suite (three cases, three runs, two arms, -default models, five passes over 2026-09-12 and 2026-09-13): about 0.10 USD per with-run, 0.55 to -0.82 USD per without-run on a knowledge case, about 0.002 USD of judge calls per run, 2.1 to 3.2 -USD for a full pass. **Estimate the without-arm from its own anchor, not from the with-arm**: -without the plugin the model spends turns hunting, and here that arm cost five to seven times the -with-arm. Absent a probe, size each without-run on a knowledge case at 0.8 USD and every other run -at 0.1 USD, and say the figure is headroom; the 0.8 anchor applied to every without-run prices this -suite at 8 USD against a measured 2.1 to 3.2. +Sizing anchors, measured on this plugin's own read-only suite at Claude Code 2.1.287 on 2026-10-02, +with no `--model` (it served opus-5-5): 24 runs (four cases, three runs, two arms, sonnet judge) +cost 1.67 USD, 16 runs (two runs, haiku judge) 0.98 USD, and 42 short single-arm calibration runs +1.79 USD: 0.04 to 0.07 USD per run on average, judge calls included, 0.03 to 0.17 USD for one run. + +**A fresh suite is priced at 0.1 USD per run in either arm, judge calls included, and the figure is +called headroom.** It has no pass of its own to scale, while cases x runs x arms is known before any +spend. 0.1 USD is 1.4 to 2.4 times the per-run cost of each pass above, and prices six cases at +three runs and two arms at 3.6 USD. Once a suite has run, scale from its own last `costUsd` instead. + +The older anchor of 0.8 USD per without-run, which prices the same six cases at about 16 USD, comes +from passes at 2.1.270 (2026-09-12 and 2026-09-13) whose without-arm loaded the bundled `claude-api` +skill and cost five to seven times the with-arm. At 2.1.287 no without-run loaded a skill, and that +arm cost less than the with-arm. Estimate each arm from its own runs, and use 0.8 for a case only +when a kept trace shows its without-arm loading a large skill. A suite with no such trace is priced +at 0.1 alone, and the older anchor is no caveat or risk to that estimate. Re-derive these anchors +from the last three passes when a Claude Code release note touches `plugin eval`, the model a run +serves changes, or a pass averages more than 0.1 USD per run. Ceiling: `${user_config.max_cost_usd}` USD, unlimited: `${user_config.unlimited_cost}`. If either renders empty or as the literal placeholder text, use 5 USD and `false`, the manifest defaults, and -say which you used. +state the ceiling you used without commenting on the setting's state. 1. Print the estimate. This happens on every invocation, including unlimited, and the print is the step that must appear in the transcript before any CLI call. 2. Unlimited: drop `--max-cost-usd` from the command and start. Never prompt; the estimate already printed is the whole disclosure. -3. Ceiling set and estimate under it: pass `--max-cost-usd ` and start. +3. Ceiling set and estimate under it: the pass goes through. Pass `--max-cost-usd ` and + start, without stopping to ask. A question about whether a pass fits gets the same answer: it + will go through and starts now under the ceiling, for example "about 3.6 USD, under the 5 USD + ceiling, so it starts with `--max-cost-usd 5` and no confirmation". 4. Ceiling set and estimate over it: stop and offer three exits, then proceed only on the answer: raise the ceiling, narrow the run with `--case ` or `--tag `, or accept a partial run knowing it exits 2 and its scores are not comparable. @@ -181,9 +220,11 @@ committed `mocks/.replay/` so agent mocks replay without a model call. 2. Invoke the CLI. Confirm the target comes first, before any list-taking flag: ```bash - claude plugin eval --trust-plugin --json results.json --threshold 0.8 --max-cost-usd --no-publish + claude plugin eval --trust-plugin --keep-temp --json results.json --threshold 0.8 --max-cost-usd --no-publish ``` + `--keep-temp` keeps each run's trace, which the validity gate under "Reading the delta" reads. + Run it in the foreground with a tool timeout that covers the estimate (a three-case pass took about six minutes here) and wait. Never background the CLI from a headless `-p` session: the session ends and takes the run with it. @@ -208,15 +249,72 @@ unchanged; only the tool differs. Read in this order. Stopping early at any step is the finding. -1. `partial`. `true` (with `partialReason` of `cost_ceiling`, `interrupted`, or `auth_failed`) means +1. Run the validity gate before reading any number: + + ```bash + python3 "${CLAUDE_PLUGIN_ROOT}/skills/plugin-eval/scripts/run-validity.py" results.json --runs + ``` + + The run count is the run's `--runs`, else the cases' `runs`, else 3. In this repository's + checkout the script is `plugins/evals/skills/plugin-eval/scripts/run-validity.py`. It needs the + traces `--keep-temp` kept; without them it reports the trace checks unchecked and the run + INVALID. Report a score, delta, or interval only from `verdict: VALID` (exit 0), naming any + warnings it printed. On `verdict: INVALID` (exit 1), report INVALID with every reason on that + line and no number, then fix the cause and rerun. Tell the user to post that INVALID line and + its reasons in place of the number; "post nothing" is not the instruction. Exit 2 means the file could not be read or an + argument was wrong: say which. The steps below still apply to a VALID run. + + A with-arm denial aimed at or under the plugin's own directory, at a directory above it (which + covers its files), or with no absolute path or no known plugin directory, is a FAIL: the agent reached for a plugin file it could not + read, so the fact belongs in the hub. Any other denial, in + either arm, is a warning only when that run scored the same as every denial-free run of its case + in the same arm, so it left the score unchanged; with a different score, or no denial-free run to + compare, it is a FAIL. +2. `partial`. `true` (with `partialReason` of `cost_ceiling`, `interrupted`, or `auth_failed`) means the suite did not finish: report that and keep the document out of any trend. -2. Per run, `skippedPaidGraders: true` or a non-null `error`. A skipped judge grader is still scored, +3. Per run, `skippedPaidGraders: true` or a non-null `error`. A skipped judge grader is still scored, as a failure with `explanation: "skipped: cost ceiling"`, so it silently depresses the arm. A non-null `error` does not imply score 0, because the run is graded on what it produced. Either makes the case not comparable; say so instead of reporting its number. -3. `cases[].aggregates.delta`. It is **omitted** when the arms are not comparable. An omitted delta +4. `cases[].aggregates.delta`. It is **omitted** when the arms are not comparable. An omitted delta is never zero, and neither is a missing `scoreWithout`. -4. Only now read the delta: with-arm score minus without-arm score. +5. Only now read the delta: with-arm score minus without-arm score. A case the gate's `ceiling` + line names cannot show a gain: say it is excluded and use the delta that line gives over the + other cases. +6. Run the noise report over the same file and read its lines before calling any delta a gain: + + ```bash + python3 "${CLAUDE_PLUGIN_ROOT}/skills/plugin-eval/scripts/noise-report.py" results.json --threshold --interval-method --grader-agreement + ``` + + `` is `${user_config.interval_method}`, and an empty or unfilled value there means the + default, `normal`, applies; do not mention the setting's state to the user. Pass `--grader-agreement` unless `${user_config.grader_run_twice}` + is `false`. Any other method value is passed as is, and the script falls back to `normal`. These + values set the command this skill runs; they say nothing about a user's own run. In + this repository's checkout the script is `plugins/evals/skills/plugin-eval/scripts/noise-report.py`. + Exit 2 means the file could not be read or an argument was malformed: say which, and report no + interval. + +Read the noise report's lines this way: + +- `verdict: within noise` or `verdict: n too small to call`: the gain is not established, whatever + the delta's sign or size. Say so before any number. +- `verdict: the interval excludes 0`: report the delta together with its interval. +- `near ceiling`: the baseline leaves no headroom, so the suite has almost no room to show a gain. + Add a case the model fails without the plugin before reading the delta again. +- `not comparable` and `score check`: name the case. A score check means the run's reported score + and its graders disagree; read that run's graders before using its number. +- `judge agreement`: a grader with split runs needs its explanation and evidence read before its + verdict is trusted. The line saying the file holds no judge votes means agreement is unknown, + not perfect. +- `cost`: report it beside the scores, in the same answer as the delta. +- `pass count`: the interval method (the `interval_method` setting, the `--interval-method` flag) + changes only this line, the count of cases at or above the threshold. Every score interval, the + delta line included, uses the normal method paired over cases whatever the setting, because a + case score is not a proportion of trials. So a normal delta interval under `wilson` is the + setting working as designed, and nothing needs checking. + [local-decisions.md, Interval method](../methodology/reference/local-decisions.md#interval-method) + records the decision for a human reader. What the number means: @@ -231,12 +329,13 @@ What the number means: - Hold the ablation mode fixed. Under `--ablation none` nothing is excluded, so absolute scores are not comparable across modes and mixing them silently breaks a trend line. - The with-arm measures the skill hub, not its spokes. A plugin whose value lives in `reference/` - files measures only what `SKILL.md` carries, so a null delta on such a plugin is a hub finding - before it is a plugin finding. + files measures only what `SKILL.md` carries, and no grant makes the spokes readable (record + below). So anything a case depends on goes in the hub, and a null delta on such a plugin is a hub + finding before it is a plugin finding. | Fact | Basis and as-of | Recheck trigger, and what to do when it fires | |---|---|---| -| In the with-arm the injected skill body names the plugin's real on-disk directory, and a `Read` of any file under it is refused with `File is in a directory that is denied by your permission settings`; only the hub `SKILL.md` text reaches the model | Kept traces (`--keep-temp`) of this plugin's own suite at Claude Code 2.1.270, six with-arm runs, every spoke `Read` denied, verified 2026-09-13 | Recheck trigger: a Claude Code release note touches `plugin eval` or sandbox permissions, or a kept trace shows a spoke `Read` succeeding. Then re-run one case with `--keep-temp`, read the with-arm trace, refresh this row with the outcome, and record a drift outcome in this plugin's CHANGELOG | +| In the with-arm the injected skill body names the plugin's real on-disk directory, and a `Read` of any file under it is refused with `File is in a directory that is denied by your permission settings`; only the hub `SKILL.md` text reaches the model. A path-scoped `--allow-tools "Read(///skills/**)"` grant is accepted but does not lift the denial, and no flag makes a directory readable | Kept traces (`--keep-temp`) of this plugin's own suite: at Claude Code 2.1.270, six with-arm runs, every spoke `Read` denied, verified 2026-09-13; at 2.1.287 under `--runs 2`, all three spoke `Read` calls in six with-arm runs denied with that text and listed in each trace's `permission_denials`, verified 2026-10-01; at 2.1.287 with that grant on one case, both with-arm runs still denied a spoke `Read` and the run's settings held no deny rule, and `claude plugin eval --help` listed no readable-directory flag, verified 2026-10-02. For what a grant covers, see , as of 2026-10-02 | Recheck trigger: a Claude Code release note touches `plugin eval` or sandbox permissions, `--help` or that section gains a way to make a directory readable, or a kept trace shows a spoke `Read` succeeding. Then re-run one case with `--keep-temp`, with and without the grant, read the with-arm trace, refresh this row with the outcome, and record a drift outcome in this plugin's CHANGELOG | ## Iterating @@ -244,26 +343,90 @@ What the number means: a number or an explicit "not comparable". 2. The usual first finding is a delta near zero with the case's `tool_used: Skill` grader failing: the model is not choosing the skill on natural phrasing. Fix the skill's `description`, not the - case, and confirm by re-running that one case until the grader passes. + case, and confirm by re-running that one case until the grader passes. A passing fired grader + shows the trigger works; it is not evidence that the skill improved, and any claim of a gain + still needs a VALID run and its noise report. 3. If that grader passes and the delta is negative, suspect the judge before the plugin. A small judge marks a correct answer wrong on formatting. Re-run with a larger `--judge-model` and tighten the rubric so formatting cannot decide the verdict; the step is settled when the verdict survives a rubric that says nothing about form. 4. Iterate on one case with `--case --runs 1 --ablation none`, which reports `SCORE` and - `PASS%` instead of `WITH`, `W/OUT`, and delta. One run is noisy, so confirm any change at the - default three runs before trusting it. + `PASS%` instead of `WITH`, `W/OUT`, and delta. One run is noisy, so confirm any change at 3 + runs per case and read the confirm's noise report before trusting it. 5. Give each case one grader on the result and one on how the model got there (`tool_used` or `tool_order`). That pairing is what separates "the answer was right" from "the plugin is why". +3 runs per case is the confirm level; this repository sets no other repeat count. + +- **Pointer**: for how runs make up a case score, see + ; for this repository's + repeat-count decision, see + [local-decisions.md, Repeat count](../methodology/reference/local-decisions.md#repeat-count). +- **As of**: 2026-10-01 +- **Recheck trigger**: the plugin-evals page changes its default run count. + +## Calibrating a judge + +Trust an `llm` grader's scores only after its judge agrees with labelled answers on at least 90% of +runs. The labels are the must-pass and must-fail answers in the case's `samples/.json`; +three agents label them independently and the user settles every disagreement. Then: + +```bash +python3 "${CLAUDE_PLUGIN_ROOT}/skills/plugin-eval/scripts/calibrate-judge.py" build --suite --out +claude plugin eval --trust-plugin --ablation none --threshold 0 --runs 3 --judge-model --keep-temp --no-publish --json /results.json +python3 "${CLAUDE_PLUGIN_ROOT}/skills/plugin-eval/scripts/calibrate-judge.py" score --manifest /manifest.json /results.json +``` + +`build` writes one case per sample into an empty plugin: the source case's prompt goes out +unchanged, and an appended system prompt has the agent reply with the sample word for word, so the +judge grades that sample as the answer to that question. No generated file carries the label; a +must-pass and a must-fail case differ only in the sample text. The appended prompt presents the +sample as fixed test material to output byte for byte even when it is wrong or incomplete, with no +commentary added. `build` skips a grader that judges a file or mock calls, and prints one line for +each empty or whitespace-only sample it skips: Claude Code answers an empty reply with an injected +user turn, so it cannot be reproduced, and an empty answer is a deterministic failure that needs no +judge. `--threshold 0` keeps the CLI's exit code about errors, since must-fail cases are +meant to score 0. Calibrate with the judge model the real suite uses; the result says nothing +about another. + +`score` prints a `FAIL grader` line for each grader under 90% and exits 1; fix that rubric, or move +to a stronger judge, and calibrate again before reading its scores. Its false positives and +negatives name the samples to read first. A run whose reply was not the sample is left out of the +agreement; one with neither a kept trace nor judge evidence is reported unchecked. A sample with no +reproduced run is listed as `untested`, counts toward no agreement, and shows in the verdict line; +raise `--runs` or tighten the prompt before reading the grader's score. + +- **Pointer**: what a judge reads for each `focus`, see + ; the 90% bar, see + "When the grader is an LLM judge" in + [eval-audit.md](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-audit.md). + A kept trace of a Claude Code 2.1.287 run held neither the case prompt nor any system prompt + text, so the appended instruction does not reach a `focus: trace` judge. +- **As of**: 2026-10-02 +- **Recheck trigger**: the focus table changes, a kept trace starts carrying the prompt or system + prompt, or eval-audit.md moves its agreement bar. + ## CI -The recipe, its exit-code table, and the parser rules live in [reference/ci.md](reference/ci.md). -Three rules matter enough to state here: pin both `--model` and `--judge-model` so a model rollout +A CI job runs the command under Run with `--model ` and +`--judge-model ` added, under these three rules: pin both `--model` and `--judge-model` so a model rollout is not read as a plugin regression, pass `--trust-plugin` because a non-TTY job is otherwise refused, and read the JSON in addition to the exit code. The CLI's own exit 1 still fails the job on a below-threshold case, but it is overloaded across six causes and exit 2 means partial, so the JSON is what tells a reader which one happened and whether the arms were comparable at all. +A CI pin is a full model ID for each of the two, never an alias such as `sonnet`. + +Answer a CI question from this section. [reference/ci.md](reference/ci.md) has the full workflow +file, the exit-code table, and the parser rules as background for a human reader. + +- **Pointer**: for the CI invocation and its model pins, see + ; for what an alias resolves to, + see . +- **As of**: 2026-10-02 +- **Recheck trigger**: either section changes how it pins or resolves a model, or the command + options table changes its `--model` or `--judge-model` row. + ## Boundary, the built-in `plugin eval` command - The **CLI** (`plugin eval`, a built-in command) owns execution, grading, the ablation arms, the @@ -293,10 +456,10 @@ is what tells a reader which one happened and whether the arms were comparable a - A pass that crosses the ceiling can still end `partial: false` with exit 0 and one case missing its `delta`: the ceiling skips judge calls, not runs. A pass that crosses it earlier skips whole cases and reports `partial: true` with exit 2. Only the JSON distinguishes them. -- The without-arm is not the with-arm minus the plugin. On a knowledge case it cost four to seven - times as much: the kept traces show the model without the plugin invoking the bundled - `claude-api` skill on every run, and that skill's injected body is about fourteen times the size - of this plugin's hub. +- The without-arm is not the with-arm minus the plugin. At Claude Code 2.1.270, on a knowledge case + it cost four to seven times as much (at 2.1.287 it cost less; see Cost): the kept traces show + the model without the plugin invoking the bundled `claude-api` skill on every run, and that + skill's injected body is about fourteen times the size of this plugin's hub. - Inside the with-arm, a `Read` of the plugin's own `reference/` files is refused, so the spokes never reach the model; see the record under "Reading the delta" before crediting a spoke. - `--trust-plugin` persists. Answering the trust prompt yes inside a git repository trusts the whole @@ -311,12 +474,17 @@ is what tells a reader which one happened and whether the arms were comparable a - A headless `-p` session tends to go from the reads straight to the CLI call and print the estimate in its final answer. The estimate step above says before; when the transcript is the evidence, read it for the order, not only for the number. -- A typo in `--case` exits 1 and no exit code separates it from a real failure. The message and its - record live in [reference/ci.md](reference/ci.md#exit-codes); read it there rather than trusting a - restatement here. -- A usage or rate limit mid-suite is **not** marked partial. Later runs end with the error, are - graded on what they produced, and usually score 0, so the suite reads as a regression. Check - `cases[].arms.with[].error` before believing a drop. +- A typo in `--case` exits 1 and no exit code separates it from a real failure, so read the CLI's + message before treating an exit 1 as a failing case. [reference/ci.md](reference/ci.md#exit-codes) + records the message for a human reader. +- A usage or rate limit mid-suite is **not** marked partial, so `partial: false` does not show the + runs ended normally. Later runs end with the error, are graded on what they produced, and usually + score 0 in both arms, so the suite reads as a regression. Check each affected run's `error` + (`cases[].arms.with[].error` and `cases[].arms.without[].error`) before believing a drop; when it + names the limit, rerun those cases with `--case` once the limit resets. It is neither a regression + nor flaky cases, and leaving the cases out of the trend is not enough on its own. When `error` is + null, run the validity gate and the noise report under "Reading the delta" before drawing any + conclusion, and say nothing yet about the cases themselves. - A run from inside a Claude Code session keeps its report local and says `kept local`; from a terminal it may publish to claude.ai unless `--no-publish` is passed. - The sandbox limits what the agent under test can reach. It is not a boundary against the plugin's diff --git a/plugins/evals/skills/plugin-eval/evals/evals.json b/plugins/evals/skills/plugin-eval/evals/evals.json index ec10fbe8ea..4565ccd591 100644 --- a/plugins/evals/skills/plugin-eval/evals/evals.json +++ b/plugins/evals/skills/plugin-eval/evals/evals.json @@ -5,10 +5,10 @@ "id": 1, "name": "preflight-reports-before-spending", "prompt": "/evals:plugin-eval preflight plugins/evals", - "expected_output": "Prints the preflight report with cli_version read from claude --version, floor_met, platform, sandbox_backend, target_type: plugin, suite_tools, estimate_usd, and ceiling, then stops. No claude plugin eval invocation happens during a preflight.", + "expected_output": "Prints the preflight report with cli_version read from claude --version, floor_met, platform, sandbox_backend, target_type: plugin, suite_tools, same_model (the tested model against the judge model), estimate_usd, and ceiling, in that order, then stops. No claude plugin eval invocation happens during a preflight.", "narration": true, "expectations": [ - "Every preflight field is printed, including sandbox_backend and estimate_usd", + "Every preflight field is printed in order, including sandbox_backend, same_model, and estimate_usd", "cli_version comes from reading claude --version rather than being asserted", "target_type is reported as plugin for a directory holding a plugin manifest and an eval dir", "No eval run is launched and no cost is incurred by the preflight action" @@ -117,6 +117,30 @@ "Starts from preflight and the cost estimate rather than invoking the CLI immediately", "Frames the answer as the ablation delta rather than an absolute score" ] + }, + { + "id": 11, + "name": "same-model-judge-warned-not-blocked", + "prompt": "/evals:plugin-eval preflight plugins/evals. I'll run it with --model haiku --judge-model haiku to keep it cheap.", + "expected_output": "Reports same_model: yes with both models named, prints the warning that the judge is the model under test and suggests a different --judge-model, and does not refuse or stop the run because of it.", + "narration": true, + "expectations": [ + "The same_model field reads yes and names haiku as both the tested and the judge model", + "The warning line suggests passing a different --judge-model", + "The warning does not block or refuse the run" + ] + }, + { + "id": 12, + "name": "small-gain-read-through-noise-report", + "prompt": "/evals:plugin-eval read results.json. Four cases, partial is false, every run is clean, and the mean delta is +0.05 with the without-arm mean at 0.95. Ship it?", + "expected_output": "Runs scripts/noise-report.py over results.json with the run's threshold, the interval_method setting, and --grader-agreement when grader_run_twice is on, then reads its lines: the gain is not established while the verdict is within noise or n too small to call, the near-ceiling line means the baseline leaves no headroom, and cost is reported beside the scores.", + "expectations": [ + "Runs the noise report script rather than computing an interval by hand", + "Does not call the +0.05 delta an improvement while the verdict is within noise or n too small to call", + "Names the near-ceiling baseline as leaving no headroom", + "Reports cost beside the scores" + ] } ] } diff --git a/plugins/evals/skills/plugin-eval/reference/case-authoring.md b/plugins/evals/skills/plugin-eval/reference/case-authoring.md index 198f9de982..d3e3993f28 100644 --- a/plugins/evals/skills/plugin-eval/reference/case-authoring.md +++ b/plugins/evals/skills/plugin-eval/reference/case-authoring.md @@ -74,7 +74,9 @@ unless the case routes it in explicitly. `arm: both` when the check must score in both arms, which a must-not-invoke check (`min: 0` **and** `max: 0`) requires. - [ ] Drop any assertion that passes in both arms and measures nothing. A case at 1.00 on both sides - is a passing case and a null measurement. + is a passing case and a null measurement. Keep one only as a regression guard: tag it + `regression-guard`, say so in its `description`, and pair it with a case a person judged + hard, tagged `hard`, with the reason in its `description`. | Type | Fields | Passes when | |---|---|---| diff --git a/plugins/evals/skills/plugin-eval/reference/ci.md b/plugins/evals/skills/plugin-eval/reference/ci.md index c7f847947e..19f1fe06bb 100644 --- a/plugins/evals/skills/plugin-eval/reference/ci.md +++ b/plugins/evals/skills/plugin-eval/reference/ci.md @@ -18,8 +18,8 @@ claude plugin eval . \ --trust-plugin \ --json results.json \ --threshold 0.8 \ - --model \ - --judge-model \ + --model \ + --judge-model \ --no-publish \ --max-cost-usd 20 ``` @@ -27,8 +27,9 @@ claude plugin eval . \ - `--trust-plugin` is mandatory in practice. Without it a job whose checkout is untrusted is refused with exit 1 where there is no terminal, and waits at the prompt where the runner allocates one. - Pin **both** models so a model rollout is not read as a plugin regression and rubric verdicts stay - comparable. The pinned values are the operator's choice and carry their own refresh cadence: - review them whenever the provider retires or renames a model, and record the swap alongside the + comparable. Pin each by its full model ID, never an alias, per the record in the + [hub's CI section](../SKILL.md#ci). The pinned values are the operator's choice and carry their + own refresh cadence: review them whenever the provider retires or renames a model, and record the swap alongside the trend so a step change in scores has a cause. - `--threshold 0.8` rather than the default 1.0. At 1.0 every stochastic near-miss is a red build, and exit 1 stops carrying information. @@ -112,5 +113,11 @@ Three rules the parser encodes, each of which a naive reader gets wrong: - Budget awareness: the command has no free mode, so an every-commit lane should use deterministic graders only, and the expensive judge lane should run on a schedule or on demand. - A fallback plan for the server-side switch. The command can answer - `plugin eval is currently unavailable`, and nothing on the runner restores it, so a required check - built on this command can block merges for reasons no one in the repository controls. + `plugin eval is currently unavailable` while Anthropic has it switched off, which can change at + any time and may depend on the account or context; nothing on the runner fixes it, so a required check built + on this command can block merges for reasons no one in the repository controls. + + - **Pointer**: for what the message means, see + , the "plugin eval is currently unavailable" entry. + - **As of**: 2026-10-02 + - **Recheck trigger**: that troubleshooting entry changes or is removed. diff --git a/plugins/evals/skills/plugin-eval/reference/reading-results.md b/plugins/evals/skills/plugin-eval/reference/reading-results.md index 8258920f5f..ac7c239621 100644 --- a/plugins/evals/skills/plugin-eval/reference/reading-results.md +++ b/plugins/evals/skills/plugin-eval/reference/reading-results.md @@ -11,17 +11,23 @@ statement of record: with `--json ` the terminal summary table is suppress - [Per-run fields](#per-run-fields) - [Per-grader fields](#per-grader-fields) - [What a delta does and does not say](#what-a-delta-does-and-does-not-say) +- [Noise report](#noise-report) ## Read order -1. `partial`. When `true`, `partialReason` is `cost_ceiling`, `interrupted`, or `auth_failed`. The +1. `scripts/run-validity.py --runs `. Only `verdict: VALID` lets a number be reported; + on `verdict: INVALID`, report INVALID and its reasons. The command and its exits are in + [SKILL.md, Reading the delta](../SKILL.md#reading-the-delta). +2. `partial`. When `true`, `partialReason` is `cost_ceiling`, `interrupted`, or `auth_failed`. The suite did not finish; report that and keep the document out of any trend. The check is done when the field has been read, not inferred from the exit code. -2. Every run in both arms: `skippedPaidGraders` and `error`. Either one makes its case not +3. Every run in both arms: `skippedPaidGraders` and `error`. Either one makes its case not comparable. Say "not comparable" and name which run; do not report the case's number. -3. `cases[].aggregates.delta`. It is **omitted** when the arms are not comparable, and so is +4. `cases[].aggregates.delta`. It is **omitted** when the arms are not comparable, and so is `scoreWithout`. An omitted field is never zero. -4. Only then, the delta itself. +5. Only then, the delta itself. +6. Run the [noise report](#noise-report) over the same file and read its lines before calling any + delta a gain. ## Document fields @@ -42,6 +48,16 @@ statement of record: with `--json ` the terminal summary table is suppress `arms`, and `aggregates`. `arms` holds `with` and `without`, each a list of runs; under `--ablation none` only `with` is present, and `suite.ablation` records which mode ran. +Count each arm's rows for the run count, never `runsPerCase`: under `--runs 2` at Claude Code +2.1.287 every case reported `runsPerCase: 3` while each arm held 2 rows. + +- **Basis**: an `aggregate-result.json` from this plugin's own suite under `--runs 2`, Claude Code + 2.1.287; no docs page defines `runsPerCase` + (). +- **As of**: 2026-10-01 +- **Recheck trigger**: the plugin-evals page documents `runsPerCase`, or a run's `runsPerCase` + matches its row count under `--runs`. Then update this paragraph. + `cases[].aggregates` carries `score` and `passRate` for the with-arm, plus `scoreWithout`, `passRateWithout`, and `delta` when the arms are comparable. Observed on a suite where one without-run lost its judge to the ceiling: `scoreWithout` and `delta` were both absent while @@ -93,3 +109,37 @@ without-run lost its judge to the ceiling: `scoreWithout` and `delta` were both excluded, so the same suite scores differently. Hold the mode fixed or the trend line is fiction. - A single run is a smoke test. The defaults exist because a non-deterministic agent tells you little in one sample, and pinning both models buys comparability, not determinism. + +## Noise report + +`scripts/noise-report.py` reads one `aggregate-result.json` and prints noise lines beside its +scores. It makes no model call and spends nothing. SKILL.md carries the full command; relative to +this skill's directory it is: + +```bash +python3 scripts/noise-report.py results.json --threshold --interval-method --grader-agreement +``` + +Pass the `--threshold` the run used, the `interval_method` setting as `--interval-method`, and +`--grader-agreement` when the `grader_run_twice` setting is on. A partial result prints one line +and nothing else. An interval method other than `normal`, `wilson` or `jeffreys` falls back to +`normal` with one line saying so. Exit 0 means the report printed; exit 2 means the file could not +be read or an argument was malformed. + +| Line | What to do with it | +|---|---| +| `not comparable: case ()` | Name the case and leave it out of any claim. The report has already left it out of every number | +| `score check: ...` | The run's reported score and its graders disagree. Read that run's graders before using its number | +| `-arm mean: ...` | Report each mean with its interval, never the mean alone | +| `cost: ...` | Report it beside the scores, in the same answer as the delta | +| `near ceiling: ...` | The baseline leaves no headroom, so the suite has almost no room to show a gain. Add a case the model fails without the plugin | +| `delta ...` and `verdict: within noise` or `verdict: n too small to call` | The gain is not established, whatever the delta's sign or size | +| `delta ...` and `verdict: the interval excludes 0` | Report the delta with its interval | +| `-arm pass count ...` | Cases at or above the threshold, with the chosen interval. Only this line follows `--interval-method` | +| `judge agreement: ...` | A grader with split runs needs its `explanation` and `evidence` read before its verdict is trusted | +| `the result file holds no judge votes ...` | Agreement is unknown, not perfect | + +Why score intervals stay normal while the pass count follows the setting: +[local-decisions.md, Interval method](../../methodology/reference/local-decisions.md#interval-method). +How many runs and cases to add when an interval is too wide: +[local-decisions.md, Repeat count](../../methodology/reference/local-decisions.md#repeat-count). diff --git a/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py new file mode 100755 index 0000000000..ea894a07c2 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py @@ -0,0 +1,726 @@ +#!/usr/bin/env python3 +"""calibrate-judge - measure an llm grader against labelled answers. + + calibrate-judge build --suite --out [--case ] [--grader ] + calibrate-judge score --manifest + +An llm grader judges the agent's final message. `build` turns every labelled +sample of every llm grader (`/samples/.json`, the `pass` and +`fail` lists validate-cases.py reads) into one calibration case whose agent +replies with that sample verbatim, so the judge's verdict on the case is its +verdict on the sample. `score` compares those verdicts with the labels. + +build writes, under --out (which must be empty or absent): + + .claude-plugin/plugin.json an empty plugin, so `claude plugin eval ` + runs the suite with no skill to fire + evals//prompt.md the source case's prompt, unchanged, as the + user turn; the sample and the instruction to + reproduce it ride in append_system_prompt + evals//graders/.md the source llm grader, copied as is + manifest.json generated case -> source case, grader, sample + index, expected PASS or FAIL, and the sample + +The label never reaches the generated files: a must-pass and a must-fail case +differ only in the sample text, and case names are numbered by a hash of the +sample, not by label. The generated suite is then checked with +validate-cases.py; build fails if that reports a FAIL. + +An empty or whitespace-only sample is skipped, with one stderr line per sample: +Claude Code answers an empty reply with an injected user turn, so the agent +would answer the real prompt, and an empty answer is a deterministic failure +that needs no judge. + +An llm grader whose focus is a file, the created-file list, or mock calls is +skipped: the agent's reply cannot stand in for those. A `trace` focus is kept +with a warning, since the calibration trace carries no tool calls. + +score reads the result of `claude plugin eval --ablation none` and, per +grader, prints: + + agreement the run's verdict (majority of its judgeVotes) against + the label, over every judged and reproduced run + false positive a must-fail sample the judge passed, with its first line + false negative a must-pass sample the judge failed, with its first line + split vote a run whose judge votes disagreed + not reproduced a run whose reply differs from the sample (whitespace and + bold markers aside); it is left out of the agreement + reproduction how many runs were checked against the kept trace's final + message, against the judge's evidence, or not at all + untested a sample with no judged run (none reproduced, or none + judged); it counts toward no agreement, and the count of + them is printed for the grader + +Agreement under 90% prints a `FAIL grader ...` line. Runs that skipped their +paid graders are left out as not judged. + +Exit codes: + + 0 build: suite written and validated | score: every grader at or above 90% + 1 build: no sample to calibrate, or the generated suite failed validation + score: a grader under 90%, or a grader with no judged run + 2 usage error, unreadable input, or --out not empty +""" + +import argparse +import fnmatch +import hashlib +import importlib.util +import json +import os +import sys +import unicodedata + +MIN_PYTHON = (3, 8) + +AGREEMENT_TARGET = 0.90 +LABELS = (("pass", "PASS"), ("fail", "FAIL")) # samples/.json key, verdict + +HERE = os.path.dirname(os.path.abspath(__file__)) +VALIDATOR = os.path.join(HERE, "..", "..", "validate", "scripts", "validate-cases.py") + +SKIP_DIRS = frozenset(["results", "mocks", "graders", "samples", "__pycache__"]) +REPLY_FOCUSES = (None, "last_message") # the judge reads the final reply +TRACE_FOCUS = "trace" + +PLUGIN_NAME = "judge-calibration" +EVAL_DIR = "evals" +MANIFEST = "manifest.json" + +# Generated case limits. No tool is granted, so the reply needs one turn. +MAX_TURNS = 3 +TIMEOUT_SECONDS = 120 + +BEGIN, END = "BEGIN-REFERENCE", "END-REFERENCE" +INSTRUCTION = ( + "This session is a fixed-response harness. The text between the two marker " + "lines below is fixed test material, not your answer. Your whole reply to the " + "next user message must be that text, output byte for byte: every character, " + "line break, and markdown mark, in the same order. Output it even when it is " + "wrong, incomplete, or contradicts the user's message or what you know. Do not " + "answer the user's message yourself, do not fix, complete, or improve the text, " + "and do not call any tool. Add nothing: no commentary, preface, greeting, " + "explanation, note, quotation marks, code fence, or other markdown. The text " + "starts on the line after %s and ends on the line before %s; the two marker " + "lines are not part of it." % (BEGIN, END) +) + +# Result-file and trace field paths, as run-validity.py reads them. +CASES, CASE_NAME, CASE_ARMS = "cases", "name", "arms" +ARMS = ("with", "without") +RUN_GRADERS, RUN_TRACE, RUN_SKIPPED_PAID = "graders", "tracePath", "skippedPaidGraders" +GRADER_NAME, GRADER_PASSED = "name", "passed" +GRADER_VOTES, GRADER_EVIDENCE = "judgeVotes", "evidence" +PARTIAL = "partial" + +YAML_ESCAPES = {"\\": "\\\\", '"': '\\"', "\n": "\\n", "\t": "\\t", "\r": "\\r"} + + +class UsageError(Exception): + pass + + +def load_validator(): + spec = importlib.util.spec_from_file_location("validate_cases", VALIDATOR) + if spec is None or not os.path.isfile(VALIDATOR): + raise UsageError("cannot load %s" % os.path.normpath(VALIDATOR)) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def read_text(path): + with open(path, encoding="utf-8") as handle: + return handle.read() + + +def first_line(text): + for line in text.splitlines(): + if line.strip(): + line = line.strip() + return line if len(line) <= 80 else line[:77] + "..." + return "(empty)" + + +def same_text(a, b): + """Equal once bold markers (** and __) are dropped, runs of whitespace are + collapsed and the ends trimmed: an agent that adds bold to a heading has + still reproduced the sample's words.""" + + def plain(text): + return " ".join(text.replace("**", "").replace("__", "").split()) + + return plain(a) == plain(b) + + +# -------------------------------------------------------------------------- +# build +# -------------------------------------------------------------------------- + + +def yaml_quoted(text): + """One double-quoted YAML scalar on one line, in validate-cases.py's subset.""" + out = [] + for char in text: + if char in YAML_ESCAPES: + out.append(YAML_ESCAPES[char]) + elif unicodedata.category(char) == "Cc": + raise ValueError("control character U+%04X" % ord(char)) + else: + out.append(char) + return '"%s"' % "".join(out) + + +def system_prompt(sample): + return "%s\n%s\n%s\n%s" % (INSTRUCTION, BEGIN, sample, END) + + +def split_body(text): + """(frontmatter or None, body) of a markdown file.""" + lines = text.split("\n") + if lines and lines[0].strip() == "---": + for index in range(1, len(lines)): + if lines[index].strip() == "---": + return "\n".join(lines[1:index]), "\n".join(lines[index + 1 :]) + return None, text + + +def source_prompt(case_dir, validator): + """The prompt the source case sends: prompt.md's body, else execution.prompt.""" + prompt_md = os.path.join(case_dir, "prompt.md") + if os.path.isfile(prompt_md): + return split_body(read_text(prompt_md))[1].strip("\n") + data = validator.parse_yaml(read_text(os.path.join(case_dir, "case.yaml"))) + execution = data.get("execution") + prompt = execution.get("prompt") if isinstance(execution, dict) else None + return prompt.strip("\n") if isinstance(prompt, str) else "" + + +def llm_graders(case_dir, validator, notes, case): + """(name, focus, grader file text) for every llm grader in the case.""" + found = [] + yaml_path = os.path.join(case_dir, "case.yaml") + if os.path.isfile(yaml_path): + data = validator.parse_yaml(read_text(yaml_path)) + for entry in data.get("graders") or []: + if isinstance(entry, dict) and entry.get("type") == "llm": + lines = ["---", "type: llm"] + for key in ("focus", "weight", "arm"): + if isinstance(entry.get(key), (str, int, float)): + lines.append("%s: %s" % (key, entry[key])) + body = "\n".join(lines + ["---", "", str(entry.get("criteria", ""))]) + found.append((str(entry.get("name")), entry.get("focus"), body + "\n")) + grader_dir = os.path.join(case_dir, "graders") + for filename in sorted(os.listdir(grader_dir)) if os.path.isdir(grader_dir) else []: + if not filename.endswith(".md"): + continue + text = read_text(os.path.join(grader_dir, filename)) + block, _ = split_body(text) + try: + options = validator.parse_yaml(block or "") + except validator.ParseError as error: + notes.append("skip %s/graders/%s: %s" % (case, filename, error.construct)) + continue + if options.get("type") == "llm": + found.append((filename[:-3], options.get("focus"), text)) + return found + + +def discover(suite): + cases = [] + for dirpath, dirnames, filenames in os.walk(suite): + dirnames[:] = sorted( + d for d in dirnames if d not in SKIP_DIRS and not d.startswith(".") + ) + if dirpath != suite and ("prompt.md" in filenames or "case.yaml" in filenames): + cases.append(dirpath) + dirnames[:] = [] + return sorted(cases) + + +def samples_for(case_dir, grader, notes, case): + """[(label key, 1-based index, answer, why)] from samples/.json.""" + path = os.path.join(case_dir, "samples", grader + ".json") + if not os.path.isfile(path): + return [] + try: + data = json.loads(read_text(path)) + except (OSError, ValueError) as error: + notes.append("skip %s/samples/%s.json: %s" % (case, grader, error)) + return [] + out = [] + for key, _ in LABELS: + items = data.get(key) if isinstance(data, dict) else None + for index, item in enumerate(items if isinstance(items, list) else [], 1): + answer = item.get("answer") if isinstance(item, dict) else None + if not isinstance(answer, str): + notes.append( + "skip %s/%s must-%s sample %d: answer is not text" + % (case, grader, key, index) + ) + continue + if not answer.strip(): + notes.append( + "skip %s/%s must-%s sample %d: empty answer; an empty answer " + "is a deterministic failure that needs no judge" + % (case, grader, key, index) + ) + continue + out.append((key, index, answer, item.get("why", ""))) + return out + + +def write(path, text): + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "w", encoding="utf-8", newline="\n") as handle: + handle.write(text) + + +def prompt_md(sample, prompt): + return ( + "---\n" + 'description: "Judge calibration case; its source and label are in the manifest"\n' + "max_turns: %d\n" + "timeout_seconds: %d\n" + "allowed_tools: []\n" + "append_system_prompt: %s\n" + "---\n\n%s\n" + % (MAX_TURNS, TIMEOUT_SECONDS, yaml_quoted(system_prompt(sample)), prompt) + ) + + +def build(args): + validator = load_validator() + suite = os.path.abspath(args.suite) + out = os.path.abspath(args.out) + if not os.path.isdir(suite): + raise UsageError("not a directory: %s" % args.suite) + if os.path.exists(out) and (not os.path.isdir(out) or os.listdir(out)): + raise UsageError("--out must be empty or absent: %s" % args.out) + + notes, entries, groups = [], [], [] + for case_dir in discover(suite): + case = os.path.relpath(case_dir, suite).replace(os.sep, "/") + if args.case and not ( + fnmatch.fnmatchcase(case, args.case) + or fnmatch.fnmatchcase(os.path.basename(case_dir), args.case) + ): + continue + try: + graders = llm_graders(case_dir, validator, notes, case) + prompt = source_prompt(case_dir, validator) + except validator.ParseError as error: + notes.append("skip %s: case.yaml not parsed (%s)" % (case, error.construct)) + continue + if graders and not prompt.strip(): + notes.append("skip %s: the case has no prompt text" % case) + continue + for grader, focus, grader_text in graders: + if args.grader and not fnmatch.fnmatchcase(grader, args.grader): + continue + samples = samples_for(case_dir, grader, notes, case) + if not samples: + continue + if focus not in REPLY_FOCUSES and focus != TRACE_FOCUS: + notes.append( + "skip %s/%s: focus %s is not the reply, so a reproduced " + "answer cannot stand in for it" % (case, grader, json.dumps(focus)) + ) + continue + if focus == TRACE_FOCUS: + notes.append( + "warn %s/%s: focus trace; the calibration trace holds the " + "reply and no tool calls, unlike a real run" % (case, grader) + ) + answers = {} + for key, _, answer, _ in samples: + answers.setdefault(" ".join(answer.split()), set()).add(key) + if any(len(keys) > 1 for keys in answers.values()): + notes.append( + "warn %s/%s: the same answer is labelled both pass and fail" + % (case, grader) + ) + ordered = sorted( + samples, + key=lambda s: ( + hashlib.sha256(s[2].encode("utf-8")).hexdigest(), + s[0], + s[1], + ), + ) + stem = "%s--%s" % (case.replace("/", "__"), grader) + count = {"pass": 0, "fail": 0} + for number, (key, index, answer, why) in enumerate(ordered, 1): + name = "%s--%02d" % (stem, number) + try: + text = prompt_md(answer, prompt) + except ValueError as error: + notes.append( + "skip %s/%s must-%s sample %d: %s" + % (case, grader, key, index, error) + ) + continue + write(os.path.join(out, EVAL_DIR, name, "prompt.md"), text) + write( + os.path.join(out, EVAL_DIR, name, "graders", grader + ".md"), + grader_text, + ) + count[key] += 1 + entries.append( + { + "case": name, + "sourceCase": case, + "grader": grader, + "focus": focus or "last_message", + "label": key, + "sampleIndex": index, + "expected": dict(LABELS)[key], + "why": why, + "answer": answer, + } + ) + groups.append((case, grader, count)) + + for note in notes: + sys.stderr.write(note + "\n") + if not entries: + print("no labelled sample of an llm grader matched; nothing written") + return 1 + + write( + os.path.join(out, ".claude-plugin", "plugin.json"), + json.dumps( + { + "name": PLUGIN_NAME, + "description": "Generated judge-calibration suite; holds no components", + }, + indent=2, + ) + + "\n", + ) + write( + os.path.join(out, MANIFEST), + json.dumps( + {"schemaVersion": 1, "suite": suite, "evalDir": EVAL_DIR, "cases": entries}, + indent=2, + ensure_ascii=False, + ) + + "\n", + ) + + print( + "wrote %d calibration cases to %s" % (len(entries), os.path.join(out, EVAL_DIR)) + ) + for case, grader, count in groups: + print( + " %s/%s: %d (%d must-pass, %d must-fail)" + % ( + case, + grader, + count["pass"] + count["fail"], + count["pass"], + count["fail"], + ) + ) + print("manifest: %s" % os.path.join(out, MANIFEST)) + + findings = validator.validate(os.path.join(out, EVAL_DIR)) + failed = [f for f in findings if f.level == "FAIL"] + for finding in failed: + print(" " + finding.line()) + print( + "validate-cases: %s (%d FAIL, %d WARN)" + % ( + "FAIL" if failed else "PASS", + len(failed), + sum(1 for f in findings if f.level == "WARN"), + ) + ) + if failed: + return 1 + print( + "run: claude plugin eval %s --trust-plugin --ablation none --threshold 0 " + "--runs 3 --judge-model --keep-temp " + "--no-publish --json %s" % (out, os.path.join(out, "results.json")) + ) + print( + "score: python3 %s score --manifest %s %s" + % ( + os.path.abspath(__file__), + os.path.join(out, MANIFEST), + os.path.join(out, "results.json"), + ) + ) + return 0 + + +# -------------------------------------------------------------------------- +# score +# -------------------------------------------------------------------------- + + +def load_json(path, what): + try: + with open(path, encoding="utf-8") as handle: + data = json.load(handle) + except (OSError, ValueError) as error: + raise UsageError("cannot read %s %s (%s)" % (what, path, error)) + if not isinstance(data, dict): + raise UsageError("%s %s is not a JSON object" % (what, path)) + return data + + +def trace_reply(path): + """The final message of a kept trace, or None when the trace cannot be read.""" + try: + with open(path, encoding="utf-8") as handle: + lines = [] + for text in handle: + try: + line = json.loads(text) + except ValueError: + continue + if isinstance(line, dict): + lines.append(line) + except OSError: + return None + if not lines: + return None + for line in reversed(lines): + if line.get("type") == "result" and isinstance(line.get("result"), str): + return line["result"] + for line in reversed(lines): + message = line.get("message") if line.get("type") == "assistant" else None + content = message.get("content") if isinstance(message, dict) else None + texts = [ + b.get("text", "") + for b in content or [] + if isinstance(b, dict) and b.get("type") == "text" + ] + if texts: + return "".join(texts) + return "" + + +def verdict(grader): + """('PASS' | 'FAIL', votes list or None) for one grader result.""" + votes = grader.get(GRADER_VOTES) + if isinstance(votes, list) and votes: + passes = sum(1 for v in votes if v is True) + return ("PASS" if passes * 2 > len(votes) else "FAIL"), votes + return ("PASS" if grader.get(GRADER_PASSED) is True else "FAIL"), None + + +def vote_text(votes): + return " ".join("PASS" if v is True else "FAIL" for v in votes) + + +class Tally: + def __init__(self, key): + self.key = key + self.agree = self.judged = 0 + self.samples = set() + self.expected = [] + self.fp, self.fn, self.split, self.unreproduced = [], [], [], [] + self.unchecked = [] + self.missing, self.not_judged, self.no_grader = [], [], [] + self.checked = {"trace": 0, "judge evidence": 0, "unchecked": 0} + + +def describe(entry): + return 'must-%s sample %d: "%s"' % ( + entry["label"], + entry["sampleIndex"], + first_line(entry["answer"]), + ) + + +def score_run(tally, entry, run, where, base): + graders = run.get(RUN_GRADERS) if isinstance(run.get(RUN_GRADERS), list) else [] + grader = next( + ( + g + for g in graders + if isinstance(g, dict) and g.get(GRADER_NAME) == entry["grader"] + ), + None, + ) + if grader is None: + tally.no_grader.append(where) + return + if run.get(RUN_SKIPPED_PAID) is True: + tally.not_judged.append(where) + return + + reply, source = None, "unchecked" + trace = run.get(RUN_TRACE) + if isinstance(trace, str) and trace: + reply = trace_reply( + trace if os.path.isabs(trace) else os.path.join(base, trace) + ) + source = "trace" if reply is not None else source + evidence = grader.get(GRADER_EVIDENCE) + if reply is None and entry["focus"] != TRACE_FOCUS and isinstance(evidence, str): + reply, source = evidence, "judge evidence" + tally.checked[source] += 1 + if reply is None: + tally.unchecked.append(where) + if reply is not None and not same_text(reply, entry["answer"]): + tally.unreproduced.append( + '%s: reply begins "%s" (%s)' % (where, first_line(reply), source) + ) + return + + said, votes = verdict(grader) + sample = describe(entry) + votes_note = "votes " + vote_text(votes) if votes else "no judgeVotes, read passed" + tally.judged += 1 + tally.samples.add(entry["case"]) + if said == entry["expected"]: + tally.agree += 1 + elif said == "PASS": + tally.fp.append("%s (%s) %s" % (where, sample, votes_note)) + else: + tally.fn.append("%s (%s) %s" % (where, sample, votes_note)) + if votes and len(set(v is True for v in votes)) > 1: + tally.split.append( + "%s: %s, %s the label" + % ( + where, + vote_text(votes), + "agrees with" if said == entry["expected"] else "misses", + ) + ) + + +def score(args): + manifest = load_json(args.manifest, "manifest") + result = load_json(args.result, "result") + entries = manifest.get("cases") + if not isinstance(entries, list) or not entries: + raise UsageError("manifest %s lists no cases" % args.manifest) + base = os.path.dirname(os.path.abspath(args.result)) + by_name = { + c.get(CASE_NAME): c for c in result.get(CASES) or [] if isinstance(c, dict) + } + + tallies = {} + for entry in entries: + key = "%s/%s" % (entry["sourceCase"], entry["grader"]) + tally = tallies.setdefault(key, Tally(key)) + tally.expected.append((entry["case"], describe(entry))) + case = by_name.get(entry["case"]) + if case is None: + tally.missing.append(entry["case"]) + continue + arms = case.get(CASE_ARMS) if isinstance(case.get(CASE_ARMS), dict) else {} + for arm in ARMS: + for index, run in enumerate(arms.get(arm) or [], 1): + if isinstance(run, dict): + where = "%s %s-arm run %d" % (entry["case"], arm, index) + score_run(tally, entry, run, where, base) + + if result.get(PARTIAL) is True: + print("note: the result is partial (%s)" % result.get("partialReason")) + unknown = sorted(n for n in by_name if n not in {e["case"] for e in entries}) + if unknown: + print("note: %d result cases are not in the manifest" % len(unknown)) + + failed, untested_total = [], 0 + for key in sorted(tallies): + t = tallies[key] + untested = [(c, d) for c, d in t.expected if c not in t.samples] + untested_total += len(untested) + if t.judged: + rate = t.agree / t.judged + line = "agreement %d/%d runs (%.1f%%) over %d samples" % ( + t.agree, + t.judged, + 100 * rate, + len(t.samples), + ) + else: + rate, line = None, "no judged run" + print("grader %s: %s" % (key, line)) + for label, items in ( + ("false positive", t.fp), + ("false negative", t.fn), + ("split vote", t.split), + ("not reproduced, left out", t.unreproduced), + ("reproduction unchecked, no trace or evidence", t.unchecked), + ("not judged, paid graders skipped", t.not_judged), + ("grader missing from run", t.no_grader), + ("missing from the result", t.missing), + ): + for item in items: + print(" %s: %s" % (label, item)) + print( + " reproduction: %d checked against traces, %d against judge evidence, " + "%d unchecked" + % (t.checked["trace"], t.checked["judge evidence"], t.checked["unchecked"]) + ) + print( + " samples with no reproduced run: %d of %d" + % (len(untested), len(t.expected)) + ) + for name, text in untested: + print(" untested: %s (%s), counted in no agreement" % (name, text)) + if rate is None or rate < AGREEMENT_TARGET - 1e-9: + failed.append(key) + print( + "FAIL grader %s: %s under the %d%% target" + % ( + key, + "no judged run is" + if rate is None + else "agreement %.1f%% is" % (100 * rate), + round(100 * AGREEMENT_TARGET), + ) + ) + note = "; %d untested" % untested_total if untested_total else "" + if failed: + print("verdict: FAIL (%s)%s" % (", ".join(failed), note)) + return 1 + print( + "verdict: PASS (every grader at or above %d%%%s)" + % (round(100 * AGREEMENT_TARGET), note) + ) + return 0 + + +def main(argv=None): + if sys.version_info < MIN_PYTHON: + sys.stderr.write( + "error: calibrate-judge needs Python %d.%d or newer\n" % MIN_PYTHON + ) + return 2 + parser = argparse.ArgumentParser( + prog="calibrate-judge", + description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + sub = parser.add_subparsers(dest="command") + b = sub.add_parser("build", help="write a calibration suite and its manifest") + b.add_argument("--suite", required=True, help="the eval dir holding the cases") + b.add_argument("--out", required=True, help="an empty or absent output dir") + b.add_argument("--case", help="only source cases whose name matches this glob") + b.add_argument("--grader", help="only llm graders whose name matches this glob") + s = sub.add_parser("score", help="compare judge verdicts with the labels") + s.add_argument("--manifest", required=True, help="the manifest build wrote") + s.add_argument("result", help="aggregate-result.json of the calibration run") + try: + args = parser.parse_args(argv) + except SystemExit as error: + return 0 if error.code == 0 else 2 + if args.command is None: + parser.print_usage(sys.stderr) + return 2 + try: + return build(args) if args.command == "build" else score(args) + except UsageError as error: + sys.stderr.write("error: %s\n" % error) + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.test.sh b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.test.sh new file mode 100755 index 0000000000..231303fb36 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.test.sh @@ -0,0 +1,41 @@ +#!/usr/bin/env bash +# Cross-platform wrapper for calibrate-judge.py's unittest suite, so the repo's +# run-plugin-tests.sh discovery (plugins/**/*.test.sh) runs it. The interpreter +# discovery follows plugins/evals/skills/validate/scripts/validate-cases.test.sh. +# +# Exit: 0 all tests passed; 1 a test failed; 2 no usable interpreter (a named +# environment error, never a silent skip). +set -uo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ENGINE="$SCRIPT_DIR/calibrate-judge.py" +SUITE="$SCRIPT_DIR/test_calibrate_judge.py" + +# The Python floor has one origin: MIN_PYTHON in the engine. +FLOOR="$(sed -n 's/^MIN_PYTHON = (\([0-9]*\), \([0-9]*\)).*/\1.\2/p' "$ENGINE")" +if [[ -z "$FLOOR" ]]; then + echo "error: could not parse MIN_PYTHON from $ENGINE" >&2 + exit 2 +fi + +# A zero-length candidate under a WindowsApps path component is the Store's App +# Execution Alias stub, which opens the Microsoft Store instead of running an +# interpreter, so each candidate is inspected before anything executes it. +PYTHON="" +for candidate in python3 python; do + resolved="$(command -v "$candidate" 2>/dev/null)" || continue + lower="$(printf '%s' "$resolved" | tr '[:upper:]' '[:lower:]')" + if [[ "$lower" == *windowsapps* && ! -s "$resolved" ]]; then + continue + fi + if "$candidate" -c "import sys; floor = tuple(int(part) for part in '$FLOOR'.split('.')); raise SystemExit(0 if sys.version_info >= floor else 1)" 2>/dev/null; then + PYTHON="$candidate" + break + fi +done +if [[ -z "$PYTHON" ]]; then + echo "error: Python ${FLOOR}+ not found (tried python3, python) -- calibrate-judge.py's suite cannot run" >&2 + exit 2 +fi + +"$PYTHON" "$SUITE" diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/result.json b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/result.json new file mode 100644 index 0000000000..f5b5b2cccc --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/result.json @@ -0,0 +1,91 @@ +{ + "schemaVersion": 1, + "claudeVersion": "2.1.287", + "partial": false, + "suite": {"ablation": "none", "judgeModel": "sonnet", "threshold": 0}, + "cases": [ + { + "name": "capital-city--names-paris--01", + "graders": [{"name": "names-paris", "type": "llm", "weight": 1}], + "arms": { + "with": [ + { + "score": 1, "error": null, "skippedPaidGraders": false, + "tracePath": "traces/reproduced.jsonl", + "graders": [{"name": "names-paris", "passed": true, "explanation": "judge votes: PASS PASS PASS", "judgeVotes": [true, true, true]}] + }, + { + "score": 1, "error": null, "skippedPaidGraders": false, + "graders": [{"name": "names-paris", "passed": true, "explanation": "judge votes: PASS FAIL PASS", "judgeVotes": [true, false, true], "evidence": "**Paris** is the capital. \n\nIt has been the \"seat of government\" since 987 (see C:\\atlas).\n"}] + }, + { + "score": 1, "error": null, "skippedPaidGraders": false, + "tracePath": "traces/deleted-by-the-runner.jsonl", + "graders": [{"name": "names-paris", "passed": true, "explanation": "judge votes: PASS PASS PASS", "judgeVotes": [true, true, true]}] + } + ] + } + }, + { + "name": "capital-city--names-paris--02", + "graders": [{"name": "names-paris", "type": "llm", "weight": 1}], + "arms": { + "with": [ + { + "score": 0, "error": null, "skippedPaidGraders": false, + "graders": [{"name": "names-paris", "passed": false, "explanation": "judge votes: FAIL FAIL FAIL", "judgeVotes": [false, false, false], "evidence": "The capital of France is Lyon."}] + }, + { + "score": 0, "error": null, "skippedPaidGraders": false, + "graders": [{"name": "names-paris", "passed": false, "explanation": "judge votes: FAIL FAIL FAIL", "judgeVotes": [false, false, false], "evidence": "The capital of France is Lyon."}] + }, + { + "score": 0, "error": null, "skippedPaidGraders": true, + "graders": [{"name": "names-paris", "passed": false, "explanation": "skipped: cost ceiling"}] + } + ] + } + }, + { + "name": "capital-city--names-paris--03", + "graders": [{"name": "names-paris", "type": "llm", "weight": 1}], + "arms": { + "with": [ + { + "score": 0, "error": null, "skippedPaidGraders": false, + "graders": [{"name": "names-paris", "passed": false, "explanation": "judge votes: FAIL FAIL PASS", "judgeVotes": [false, false, true], "evidence": "The capital of France is Paris."}] + }, + { + "score": 1, "error": null, "skippedPaidGraders": false, + "tracePath": "traces/answered-itself.jsonl", + "graders": [{"name": "names-paris", "passed": true, "explanation": "judge votes: PASS PASS PASS", "judgeVotes": [true, true, true], "evidence": "Paris is the capital of France, and also its largest city."}] + }, + { + "score": 1, "error": null, "skippedPaidGraders": false, + "graders": [{"name": "names-paris", "passed": true, "explanation": "judge votes: PASS PASS PASS", "judgeVotes": [true, true, true], "evidence": "The capital of France is Paris."}] + } + ] + } + }, + { + "name": "capital-city--names-paris--04", + "graders": [{"name": "names-paris", "type": "llm", "weight": 1}], + "arms": { + "with": [ + { + "score": 0, "error": null, "skippedPaidGraders": false, + "graders": [{"name": "names-paris", "passed": false, "explanation": "judge votes: FAIL FAIL FAIL", "judgeVotes": [false, false, false], "evidence": "The capital of France is Marseille."}] + }, + { + "score": 0, "error": null, "skippedPaidGraders": false, + "graders": [{"name": "names-paris", "passed": false, "explanation": "judge votes: FAIL FAIL FAIL", "judgeVotes": [false, false, false], "evidence": "The capital of France is Marseille."}] + }, + { + "score": 0, "error": null, "skippedPaidGraders": false, + "graders": [{"name": "names-paris", "passed": false, "explanation": "judge votes: FAIL FAIL FAIL", "judgeVotes": [false, false, false], "evidence": "The capital of France is Marseille."}] + } + ] + } + } + ] +} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/graders/names-paris.md b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/graders/names-paris.md new file mode 100644 index 0000000000..4e04f7671b --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/graders/names-paris.md @@ -0,0 +1,6 @@ +--- +type: llm +--- + +PASS if the answer names Paris as the capital of France. +FAIL if it names another city, names no city, or is empty. diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/graders/no-tool.md b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/graders/no-tool.md new file mode 100644 index 0000000000..a1082403a8 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/graders/no-tool.md @@ -0,0 +1,6 @@ +--- +type: tool_used +tool: Read +min: 0 +max: 0 +--- diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/prompt.md b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/prompt.md new file mode 100644 index 0000000000..971451148c --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/prompt.md @@ -0,0 +1,8 @@ +--- +description: Fixture case for calibrate-judge.py +runs: 3 +max_turns: 10 +allowed_tools: [Read, Skill] +--- + +What is the capital of France? Answer in one sentence. diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/samples/names-paris.json b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/samples/names-paris.json new file mode 100644 index 0000000000..2f5502c75b --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/samples/names-paris.json @@ -0,0 +1,26 @@ +{ + "pass": [ + { + "why": "plain sentence", + "answer": "The capital of France is Paris." + }, + { + "why": "markdown over two lines, with quotes and a backslash", + "answer": "**Paris** is the capital.\n\nIt has been the \"seat of government\" since 987 (see C:\\atlas)." + } + ], + "fail": [ + { + "why": "empty answer", + "answer": "" + }, + { + "why": "wrong city", + "answer": "The capital of France is Lyon." + }, + { + "why": "wrong city, stated as fact", + "answer": "The capital of France is Marseille." + } + ] +} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/samples/no-tool.json b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/samples/no-tool.json new file mode 100644 index 0000000000..3c90070271 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/suite/capital-city/samples/no-tool.json @@ -0,0 +1,4 @@ +{ + "pass": [{"why": "no call", "answer": []}], + "fail": [{"why": "a read", "answer": [{"tool": "Read", "input": {"file_path": "/atlas.md"}}]}] +} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/traces/answered-itself.jsonl b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/traces/answered-itself.jsonl new file mode 100644 index 0000000000..f1e260aebb --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/traces/answered-itself.jsonl @@ -0,0 +1,2 @@ +{"type":"system","subtype":"init","model":"claude-sonnet-5","claude_code_version":"2.1.287","permissionMode":"dontAsk"} +{"type":"assistant","message":{"role":"assistant","content":[{"type":"text","text":"Paris is the capital of France, and also its largest city."}]}} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/traces/reproduced.jsonl b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/traces/reproduced.jsonl new file mode 100644 index 0000000000..04b27c3d07 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/traces/reproduced.jsonl @@ -0,0 +1,3 @@ +{"type":"system","subtype":"init","model":"claude-sonnet-5","claude_code_version":"2.1.287","permissionMode":"dontAsk"} +{"type":"assistant","message":{"role":"assistant","content":[{"type":"text","text":"**Paris** is the capital.\n\nIt has been the \"seat of government\" since 987 (see C:\\atlas)."}]}} +{"type":"result","subtype":"success","is_error":false,"permission_denials":[],"result":"**Paris** is the capital.\n\nIt has been the \"seat of government\" since 987 (see C:\\atlas)."} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/r2-clean-runs.json b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/r2-clean-runs.json new file mode 100644 index 0000000000..bb7e9cd062 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/r2-clean-runs.json @@ -0,0 +1,56 @@ +{ + "schemaVersion": 1, + "claudeVersion": "2.1.287", + "partial": false, + "suite": {"ablation": "with-without", "threshold": 1}, + "cases": [ + { + "name": "control-no-trigger", + "runsPerCase": 3, + "graders": [ + {"name": "names-conftest", "type": "regex"}, + {"name": "no-skill-fired", "type": "tool_used", "config": {"tool": "Skill", "min": 0, "max": 0, "arm": "both"}} + ], + "arms": { + "with": [ + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "names-conftest", "passed": true}, {"name": "no-skill-fired", "passed": true}]} + ], + "without": [ + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "names-conftest", "passed": true}, {"name": "no-skill-fired", "passed": true}]} + ] + } + }, + { + "name": "measurable-criterion", + "runsPerCase": 3, + "graders": [ + {"name": "four-properties", "type": "llm"}, + {"name": "skill-fired", "type": "tool_used", "config": {"tool": "Skill"}} + ], + "arms": { + "with": [ + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "four-properties", "passed": true, "judgeVotes": [true, true, true]}, {"name": "skill-fired", "passed": true}]} + ], + "without": [ + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "four-properties", "passed": true, "judgeVotes": [true, true, true]}]} + ] + } + }, + { + "name": "noise-before-gain", + "runsPerCase": 3, + "graders": [ + {"name": "noise-verdict", "type": "llm"}, + {"name": "skill-fired", "type": "tool_used", "config": {"tool": "Skill"}} + ], + "arms": { + "with": [ + {"score": 0, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "noise-verdict", "passed": false, "judgeVotes": [false, false, true]}, {"name": "skill-fired", "passed": true}]} + ], + "without": [ + {"score": 0, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "noise-verdict", "passed": false, "judgeVotes": [false, false, false]}]} + ] + } + } + ] +} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/r2-invalid.json b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/r2-invalid.json new file mode 100644 index 0000000000..8079dc2ec8 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/r2-invalid.json @@ -0,0 +1,80 @@ +{ + "schemaVersion": 1, + "claudeVersion": "2.1.287", + "partial": false, + "suite": {"ablation": "with-without", "threshold": 1}, + "cases": [ + { + "name": "control-no-trigger", + "runsPerCase": 3, + "graders": [ + {"name": "names-conftest", "type": "regex"}, + {"name": "no-skill-fired", "type": "tool_used", "config": {"tool": "Skill", "min": 0, "max": 0, "arm": "both"}} + ], + "arms": { + "with": [ + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "names-conftest", "passed": true}, {"name": "no-skill-fired", "passed": true}]}, + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "names-conftest", "passed": true}, {"name": "no-skill-fired", "passed": true}]} + ], + "without": [ + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "names-conftest", "passed": true}, {"name": "no-skill-fired", "passed": true}]}, + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "names-conftest", "passed": true}, {"name": "no-skill-fired", "passed": true}]} + ] + } + }, + { + "name": "grading-method-choice", + "runsPerCase": 3, + "graders": [ + {"name": "methodology-wording", "type": "regex"}, + {"name": "skill-fired", "type": "tool_used", "config": {"tool": "Skill"}} + ], + "arms": { + "with": [ + {"score": 0, "error": null, "skippedPaidGraders": false, "tracePath": "traces/LyFFkR.jsonl", "graders": [{"name": "methodology-wording", "passed": false}, {"name": "skill-fired", "passed": true}]}, + {"score": 0, "error": null, "skippedPaidGraders": false, "tracePath": "traces/d6Z5IW.jsonl", "graders": [{"name": "methodology-wording", "passed": false}, {"name": "skill-fired", "passed": true}]} + ], + "without": [ + {"score": 0, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "methodology-wording", "passed": false}]}, + {"score": 0, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "methodology-wording", "passed": false}]} + ] + } + }, + { + "name": "measurable-criterion", + "runsPerCase": 3, + "graders": [ + {"name": "four-properties", "type": "llm"}, + {"name": "skill-fired", "type": "tool_used", "config": {"tool": "Skill"}} + ], + "arms": { + "with": [ + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/axyiOf.jsonl", "graders": [{"name": "four-properties", "passed": true, "judgeVotes": [true, true, true]}, {"name": "skill-fired", "passed": true}]}, + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "four-properties", "passed": true, "judgeVotes": [true, true, true]}, {"name": "skill-fired", "passed": true}]} + ], + "without": [ + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "four-properties", "passed": true, "judgeVotes": [true, true, true]}]}, + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "four-properties", "passed": true, "judgeVotes": [true, true, true]}]} + ] + } + }, + { + "name": "noise-before-gain", + "runsPerCase": 3, + "graders": [ + {"name": "noise-verdict", "type": "llm"}, + {"name": "skill-fired", "type": "tool_used", "config": {"tool": "Skill"}} + ], + "arms": { + "with": [ + {"score": 0, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "noise-verdict", "passed": false, "judgeVotes": [false, false, false]}, {"name": "skill-fired", "passed": false}]}, + {"score": 0, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "noise-verdict", "passed": false, "judgeVotes": [false, false, true]}, {"name": "skill-fired", "passed": true}]} + ], + "without": [ + {"score": 1, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "noise-verdict", "passed": true, "judgeVotes": [true, true, true]}]}, + {"score": 0, "error": null, "skippedPaidGraders": false, "tracePath": "traces/clean.jsonl", "graders": [{"name": "noise-verdict", "passed": false, "judgeVotes": [false, false, false]}]} + ] + } + } + ] +} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/LyFFkR.jsonl b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/LyFFkR.jsonl new file mode 100644 index 0000000000..dd08c8d4f5 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/LyFFkR.jsonl @@ -0,0 +1,4 @@ +{"type":"system","subtype":"init","model":"claude-opus-5-5","claude_code_version":"2.1.287","permissionMode":"dontAsk"} +{"type":"assistant","message":{"content":[{"type":"tool_use","id":"toolu_01YDmi7j1rvt3osuLUVDS3jR","name":"Read","input":{"file_path":"/tmp/eval-r2/plugins/evals/skills/methodology/reference/grading.md"}}]}} +{"type":"user","message":{"content":[{"type":"tool_result","tool_use_id":"toolu_01YDmi7j1rvt3osuLUVDS3jR","is_error":true,"content":"File is in a directory that is denied by your permission settings."}]}} +{"type":"result","subtype":"success","is_error":false,"permission_denials":[{"tool_name":"Read","tool_use_id":"toolu_01YDmi7j1rvt3osuLUVDS3jR","tool_input":{"file_path":"/tmp/eval-r2/plugins/evals/skills/methodology/reference/grading.md"}}]} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/axyiOf.jsonl b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/axyiOf.jsonl new file mode 100644 index 0000000000..1908a2ed6b --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/axyiOf.jsonl @@ -0,0 +1,4 @@ +{"type":"system","subtype":"init","model":"claude-opus-5-5","claude_code_version":"2.1.287","permissionMode":"dontAsk"} +{"type":"assistant","message":{"content":[{"type":"tool_use","id":"toolu_01Qur2zN2R8T6x35fcQFfrn7","name":"Read","input":{"file_path":"/tmp/eval-r2/plugins/evals/skills/methodology/reference/success-criteria.md"}}]}} +{"type":"user","message":{"content":[{"type":"tool_result","tool_use_id":"toolu_01Qur2zN2R8T6x35fcQFfrn7","is_error":true,"content":"File is in a directory that is denied by your permission settings."}]}} +{"type":"result","subtype":"success","is_error":false,"permission_denials":[{"tool_name":"Read","tool_use_id":"toolu_01Qur2zN2R8T6x35fcQFfrn7","tool_input":{"file_path":"/tmp/eval-r2/plugins/evals/skills/methodology/reference/success-criteria.md"}}]} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/clean.jsonl b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/clean.jsonl new file mode 100644 index 0000000000..b212b89a17 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/clean.jsonl @@ -0,0 +1,2 @@ +{"type":"system","subtype":"init","model":"claude-opus-5-5","claude_code_version":"2.1.287","permissionMode":"dontAsk"} +{"type":"result","subtype":"success","is_error":false,"permission_denials":[]} diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/d6Z5IW.jsonl b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/d6Z5IW.jsonl new file mode 100644 index 0000000000..7662f54f6a --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/run-validity/traces/d6Z5IW.jsonl @@ -0,0 +1,4 @@ +{"type":"system","subtype":"init","model":"claude-opus-5-5","claude_code_version":"2.1.287","permissionMode":"dontAsk"} +{"type":"assistant","message":{"content":[{"type":"tool_use","id":"toolu_01CFri8yjkTnV6dDVUnZgcAH","name":"Read","input":{"file_path":"/tmp/eval-r2/plugins/evals/skills/methodology/reference/grading.md"}}]}} +{"type":"user","message":{"content":[{"type":"tool_result","tool_use_id":"toolu_01CFri8yjkTnV6dDVUnZgcAH","is_error":true,"content":"File is in a directory that is denied by your permission settings."}]}} +{"type":"result","subtype":"success","is_error":false,"permission_denials":[{"tool_name":"Read","tool_use_id":"toolu_01CFri8yjkTnV6dDVUnZgcAH","tool_input":{"file_path":"/tmp/eval-r2/plugins/evals/skills/methodology/reference/grading.md"}}]} diff --git a/plugins/evals/skills/plugin-eval/scripts/noise-report.py b/plugins/evals/skills/plugin-eval/scripts/noise-report.py new file mode 100755 index 0000000000..ef4046737b --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/noise-report.py @@ -0,0 +1,469 @@ +#!/usr/bin/env python3 +"""noise-report - put noise lines beside a `claude plugin eval` result. + + noise-report [--threshold T] + [--interval-method normal|wilson|jeffreys] [--grader-agreement] + +Reads one result file and prints, one finding per line: + + not comparable a case left out: a run errored, skipped its paid graders, + has no score, or the result omits the case's delta + score check a run whose reported score disagrees with its graders + -arm mean the mean case score with a 95% normal interval + cost the suite's list-price estimate, beside the scores + near ceiling the without-arm mean is 0.95 or higher + delta with minus without, paired over cases, 95% normal interval + verdict n too small to call | within noise | the interval excludes 0 + pass count cases at or above --threshold (default 1.0) per arm, with + the --interval-method interval (default normal) + judge agreement with --grader-agreement: per llm grader, how often its + judge votes agreed, read from votes the result holds + +A partial result prints one line and nothing else. Score intervals are always +normal; --interval-method changes only the pass-count interval. The reasons are +in plugins/evals/skills/methodology/reference/local-decisions.md, "Interval +method". + +Exit codes: + + 0 the report printed + 2 usage error, or the result file is missing, unreadable, or not JSON +""" + +import argparse +import json +import math +import sys + +MIN_PYTHON = (3, 8) + +Z95 = 1.959963984540054 +MIN_CASES = 3 +CEILING = 0.95 +ARMS = ("with", "without") # cases[].arms.[] + +# Result-file field paths read here that the plugin-evals docs page does not +# name. Correct them in this block, and nowhere else, if a real result file +# shows another shape. Every lookup is tolerant: a missing field never raises. +CASE_GRADER_DEFS = "graders" # cases[].graders[]: name, type, weight +RUN_SCORE = "score" # cases[].arms.[].score +RUN_GRADERS = "graders" # cases[].arms.[].graders[] +GRADER_NAME = "name" # graders[].name, the join key to the case definition +GRADER_TYPE = "type" # cases[].graders[].type +GRADER_PASSED = "passed" # cases[].arms.[].graders[].passed +GRADER_WEIGHT = "weight" # graders[].weight, run result first, then definition +GRADER_SCORED = "scored" # cases[].arms.[].graders[].scored, true when absent +# Judge votes on an llm grader's run result. judgeVotes is the key +# reading-results.md records from measured files; the others are read in case +# a later schema renames it. A grader holding none of these keys counts as +# holding no votes. +GRADER_VOTES = ("judgeVotes", "votes", "judge_votes") # graders[].[] +VOTE_VERDICT = ("passed", "pass", "verdict", "vote") # a vote that is an object + + +def mean(values): + return sum(values) / len(values) + + +def is_number(value): + return isinstance(value, (int, float)) and not isinstance(value, bool) + + +def definitions(case): + defs = case.get(CASE_GRADER_DEFS) + if not isinstance(defs, list): + return {} + return {d.get(GRADER_NAME): d for d in defs if isinstance(d, dict)} + + +def grader_weight(grader, defs): + weight = grader.get(GRADER_WEIGHT) + if not is_number(weight): + weight = defs.get(grader.get(GRADER_NAME), {}).get(GRADER_WEIGHT) + return float(weight) if is_number(weight) and weight > 0 else 1.0 + + +def recomputed_score(run, defs): + """Weighted fraction of scored graders passed, or None with nothing to score.""" + graders = run.get(RUN_GRADERS) + graders = ( + [g for g in graders if isinstance(g, dict)] if isinstance(graders, list) else [] + ) + scored = [g for g in graders if g.get(GRADER_SCORED) is not False] + total = sum(grader_weight(g, defs) for g in scored) + if not total: + return None + return ( + sum(grader_weight(g, defs) for g in scored if g.get(GRADER_PASSED) is True) + / total + ) + + +def run_score(run, defs, where, notes): + """The run's reported score, cross-checked against its graders; recomputed when absent.""" + reported = run.get(RUN_SCORE) + computed = recomputed_score(run, defs) + if not is_number(reported): + return computed + if computed is not None and abs(reported - computed) > 1e-6: + notes.append( + "score check: %s reports %.2f, its graders give %.2f; the reported score is used" + % (where, reported, computed) + ) + return float(reported) + + +def arm_runs(case, arm): + arms = case.get("arms") + runs = arms.get(arm) if isinstance(arms, dict) else None + return ( + [run for run in runs if isinstance(run, dict)] if isinstance(runs, list) else [] + ) + + +def incomparable(case): + """Why a case's arms cannot be compared, or None when they can.""" + for arm in ARMS: + for index, run in enumerate(arm_runs(case, arm), 1): + if run.get("error") is not None: + return "%s-arm run %d ended with an error" % (arm, index) + if run.get("skippedPaidGraders") is True: + return "%s-arm run %d skipped its paid graders" % (arm, index) + aggregates = case.get("aggregates") + if ( + arm_runs(case, "without") + and isinstance(aggregates, dict) + and "score" in aggregates + and "delta" not in aggregates + ): + return "the result omits its delta" + return None + + +def case_score(case, arm, notes): + """Mean run score for one arm; None when the arm has no runs or a run has no score.""" + defs = definitions(case) + scores = [ + run_score( + run, defs, "case %s, %s-arm run %d" % (case.get("name"), arm, index), notes + ) + for index, run in enumerate(arm_runs(case, arm), 1) + ] + if not scores or None in scores: + return None + return mean(scores) + + +def normal_interval(values, low, high): + """Mean and 95% normal-approximation interval, clamped to [low, high].""" + m = mean(values) + sd = math.sqrt(sum((v - m) ** 2 for v in values) / (len(values) - 1)) + half = Z95 * sd / math.sqrt(len(values)) + return m, max(low, m - half), min(high, m + half) + + +def beta_cf(a, b, x): + """Continued fraction for the regularized incomplete beta function.""" + tiny = 1e-300 + c, d = 1.0, 1.0 - (a + b) * x / (a + 1.0) + d = 1.0 / (d if abs(d) > tiny else tiny) + h = d + for m in range(1, 300): + for numerator in ( + m * (b - m) * x / ((a + 2 * m - 1) * (a + 2 * m)), + -(a + m) * (a + b + m) * x / ((a + 2 * m) * (a + 2 * m + 1)), + ): + d = 1.0 + numerator * d + d = 1.0 / (d if abs(d) > tiny else tiny) + c = 1.0 + numerator / c + c = c if abs(c) > tiny else tiny + h *= d * c + if abs(d * c - 1.0) < 1e-12: + break + return h + + +def beta_cdf(x, a, b): + if x <= 0.0: + return 0.0 + if x >= 1.0: + return 1.0 + front = math.exp( + math.lgamma(a + b) + - math.lgamma(a) + - math.lgamma(b) + + a * math.log(x) + + b * math.log(1.0 - x) + ) + if x < (a + 1.0) / (a + b + 2.0): + return front * beta_cf(a, b, x) / a + return 1.0 - front * beta_cf(b, a, 1.0 - x) / b + + +def beta_quantile(q, a, b): + lo, hi = 0.0, 1.0 + for _ in range(100): + mid = (lo + hi) / 2.0 + if beta_cdf(mid, a, b) < q: + lo = mid + else: + hi = mid + return (lo + hi) / 2.0 + + +def proportion_interval(k, n, method): + """95% interval on k passes out of n by the named method.""" + p = k / n + if method == "wilson": + z2 = Z95 * Z95 + denom = 1.0 + z2 / n + center = (p + z2 / (2 * n)) / denom + half = Z95 * math.sqrt(p * (1 - p) / n + z2 / (4 * n * n)) / denom + return max(0.0, center - half), min(1.0, center + half) + if method == "jeffreys": + lo = 0.0 if k == 0 else beta_quantile(0.025, k + 0.5, n - k + 0.5) + hi = 1.0 if k == n else beta_quantile(0.975, k + 0.5, n - k + 0.5) + return lo, hi + half = Z95 * math.sqrt(p * (1 - p) / n) + return max(0.0, p - half), min(1.0, p + half) + + +def pass_count_line(arm, scores, threshold, method): + k = sum(1 for s in scores if s >= threshold - 1e-9) + lo, hi = proportion_interval(k, len(scores), method) + return ( + "%s-arm pass count at threshold %.2f: %d of %d, 95%% interval %.2f to %.2f (%s)" + % ( + arm, + threshold, + k, + len(scores), + lo, + hi, + method, + ) + ) + + +def arm_line(arm, scores): + if len(scores) < 2: + return "%s-arm mean: %.2f (%d case; an interval needs 2 or more)" % ( + arm, + mean(scores), + len(scores), + ) + m, lo, hi = normal_interval(scores, 0.0, 1.0) + return "%s-arm mean: %.2f, 95%% interval %.2f to %.2f (normal, %d cases)" % ( + arm, + m, + lo, + hi, + len(scores), + ) + + +def paired_delta(deltas): + n = len(deltas) + lines = [] + if n >= 2: + m, lo, hi = normal_interval(deltas, -1.0, 1.0) + lines.append( + "delta (with minus without), paired over %d cases: %+.2f, 95%% interval %+.2f to %+.2f" + % (n, m, lo, hi) + ) + if n < MIN_CASES: + lines.append( + "verdict: n too small to call (comparable cases: %d; %d or more are needed)" + % (n, MIN_CASES) + ) + elif len(set(round(d, 9) for d in deltas)) == 1: + lines.append( + "verdict: n too small to call (every per-case delta is %+.2f, so the interval has zero width)" + % deltas[0] + ) + elif lo <= 0 <= hi: + lines.append("verdict: within noise (the interval contains 0)") + else: + lines.append( + "verdict: the interval excludes 0, so the difference is larger than the noise at %d cases" + % n + ) + return lines + + +def vote_value(vote): + """True for a PASS vote, False for a FAIL vote, None when unreadable.""" + if isinstance(vote, dict): + vote = next((vote[key] for key in VOTE_VERDICT if key in vote), None) + if isinstance(vote, bool): + return vote + if isinstance(vote, str) and vote.strip().upper() in ("PASS", "FAIL"): + return vote.strip().upper() == "PASS" + return None + + +def run_votes(grader): + """The grader's readable votes, or None when it holds none.""" + for key in GRADER_VOTES: + raw = grader.get(key) + if isinstance(raw, list) and raw: + votes = [vote_value(vote) for vote in raw] + return None if None in votes else votes + return None + + +def agreement(cases): + """One judge-agreement line per llm grader, read from votes the result holds.""" + tallies, judged = {}, False + for case in cases: + defs = definitions(case) + for arm in ARMS: + for run in arm_runs(case, arm): + for grader in run.get(RUN_GRADERS) or []: + if not isinstance(grader, dict): + continue + name = grader.get(GRADER_NAME) + kind = defs.get(name, {}).get(GRADER_TYPE, grader.get(GRADER_TYPE)) + votes = run_votes(grader) + if kind != "llm" and not (kind is None and votes): + continue + judged = True + if votes: + tallies.setdefault((case.get("name"), name), []).append(votes) + if not judged: + return ["no llm grader in this result, so there is no judge agreement to read"] + if not tallies: + return ["the result file holds no judge votes, so agreement cannot be read"] + lines = [] + for (case_name, name), runs in tallies.items(): + unanimous = sum(1 for votes in runs if len(set(votes)) == 1) + matching = sum( + sum(1 for vote in votes if vote == (2 * sum(votes) >= len(votes))) + for votes in runs + ) + lines.append( + "judge agreement: case %s, grader %s: unanimous in %d of %d runs" + " (%d of %d votes match their run's majority)" + % ( + case_name, + name, + unanimous, + len(runs), + matching, + sum(len(v) for v in runs), + ) + ) + return lines + + +def main(argv=None): + if sys.version_info < MIN_PYTHON: + sys.stderr.write( + "error: noise-report needs Python %d.%d or newer\n" % MIN_PYTHON + ) + return 2 + parser = argparse.ArgumentParser( + prog="noise-report", + description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + parser.add_argument("result", help="path to aggregate-result.json") + parser.add_argument( + "--threshold", type=float, default=1.0, help="case pass threshold (default 1.0)" + ) + parser.add_argument( + "--interval-method", + default="normal", + help="interval on the pass count: normal (default), wilson or jeffreys;" + " any other value falls back to normal", + ) + parser.add_argument( + "--grader-agreement", + action="store_true", + help="report judge-vote agreement per llm grader", + ) + args = parser.parse_args(argv) + try: + with open(args.result, encoding="utf-8") as handle: + result = json.load(handle) + except (OSError, ValueError) as error: + sys.stderr.write("error: cannot read %s (%s)\n" % (args.result, error)) + return 2 + if not isinstance(result, dict): + sys.stderr.write("error: %s is not a result document\n" % args.result) + return 2 + if result.get("partial") is True: + print( + "partial result (%s): the suite did not finish, so no noise line is computed;" + " keep this result out of any trend" % result.get("partialReason") + ) + return 0 + method = args.interval_method + lines = [] + if method not in ("normal", "wilson", "jeffreys"): + lines.append( + "interval method %r is not normal, wilson or jeffreys; using normal" + % method + ) + method = "normal" + lines += report(result, args.threshold, method) + if args.grader_agreement: + cases = result.get("cases") + lines += agreement( + [c for c in cases if isinstance(c, dict)] if isinstance(cases, list) else [] + ) + for line in lines: + print(line) + return 0 + + +def report(result, threshold, method): + cases = result.get("cases") + cases = [c for c in cases if isinstance(c, dict)] if isinstance(cases, list) else [] + lines, notes, rows = [], [], [] + two_arm = any(arm_runs(case, "without") for case in cases) + for case in cases: + reason = incomparable(case) + if reason is None and two_arm and not arm_runs(case, "without"): + reason = "no without-arm runs for this case" + row = {arm: case_score(case, arm, notes) for arm in ARMS} + if reason is None and row["with"] is None: + reason = "a with-arm run has no score to read" + if reason is None and arm_runs(case, "without") and row["without"] is None: + reason = "a without-arm run has no score to read" + if reason is not None: + lines.append("not comparable: case %s (%s)" % (case.get("name"), reason)) + continue + rows.append(row) + lines.extend(notes) + scores = {arm: [row[arm] for row in rows if row[arm] is not None] for arm in ARMS} + if not scores["with"]: + lines.append("no comparable case in this result, so there is nothing to report") + return lines + for arm in ARMS: + if scores[arm]: + lines.append(arm_line(arm, scores[arm])) + if is_number(result.get("costUsd")): + lines.append( + "cost: %.2f USD for the whole suite (list-price estimate)" + % result["costUsd"] + ) + if scores["without"]: + baseline = mean(scores["without"]) + if baseline >= CEILING: + lines.append( + "near ceiling: without-arm mean %.2f is at or above %.2f;" + " the baseline leaves no headroom" % (baseline, CEILING) + ) + pairs = [row for row in rows if row["without"] is not None] + lines.extend(paired_delta([row["with"] - row["without"] for row in pairs])) + else: + lines.append("no without-arm runs in this result, so there is no delta to read") + for arm in ARMS: + if scores[arm]: + lines.append(pass_count_line(arm, scores[arm], threshold, method)) + return lines + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/evals/skills/plugin-eval/scripts/noise-report.test.sh b/plugins/evals/skills/plugin-eval/scripts/noise-report.test.sh new file mode 100755 index 0000000000..f6092ab856 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/noise-report.test.sh @@ -0,0 +1,41 @@ +#!/usr/bin/env bash +# Cross-platform wrapper for noise-report.py's unittest suite, so the repo's +# run-plugin-tests.sh discovery (plugins/**/*.test.sh) runs it. The interpreter +# discovery follows plugins/evals/skills/validate/scripts/validate-cases.test.sh. +# +# Exit: 0 all tests passed; 1 a test failed; 2 no usable interpreter (a named +# environment error, never a silent skip). +set -uo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ENGINE="$SCRIPT_DIR/noise-report.py" +SUITE="$SCRIPT_DIR/test_noise_report.py" + +# The Python floor has one origin: MIN_PYTHON in the engine. +FLOOR="$(sed -n 's/^MIN_PYTHON = (\([0-9]*\), \([0-9]*\)).*/\1.\2/p' "$ENGINE")" +if [[ -z "$FLOOR" ]]; then + echo "error: could not parse MIN_PYTHON from $ENGINE" >&2 + exit 2 +fi + +# A zero-length candidate under a WindowsApps path component is the Store's App +# Execution Alias stub, which opens the Microsoft Store instead of running an +# interpreter, so each candidate is inspected before anything executes it. +PYTHON="" +for candidate in python3 python; do + resolved="$(command -v "$candidate" 2>/dev/null)" || continue + lower="$(printf '%s' "$resolved" | tr '[:upper:]' '[:lower:]')" + if [[ "$lower" == *windowsapps* && ! -s "$resolved" ]]; then + continue + fi + if "$candidate" -c "import sys; floor = tuple(int(part) for part in '$FLOOR'.split('.')); raise SystemExit(0 if sys.version_info >= floor else 1)" 2>/dev/null; then + PYTHON="$candidate" + break + fi +done +if [[ -z "$PYTHON" ]]; then + echo "error: Python ${FLOOR}+ not found (tried python3, python) -- noise-report.py's suite cannot run" >&2 + exit 2 +fi + +"$PYTHON" "$SUITE" diff --git a/plugins/evals/skills/plugin-eval/scripts/run-validity.py b/plugins/evals/skills/plugin-eval/scripts/run-validity.py new file mode 100755 index 0000000000..1ae2b5be90 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/run-validity.py @@ -0,0 +1,771 @@ +#!/usr/bin/env python3 +"""run-validity - decide whether a `claude plugin eval` run's score may count. + + run-validity --runs N + +--runs is the run count the eval was invoked with (its --runs, else the case +default). Reads the result file and, where each run's tracePath resolves, the +kept trace (`--keep-temp`). A relative tracePath or suite root is read relative +to the result file. Prints one line per check, detail lines indented under it: + + complete partial is false + paid graders no run skipped its paid graders + run errors no run errored, aborted, or ended its trace in an error + row count every case holds N rows per arm; runsPerCase is ignored + traces how many tracePaths resolve, and what that leaves unchecked + permission denials no denied tool call in any trace; a denial is a warning + when that run scores the same as every denial-free run + of its case in the same arm (there must be one), since + it then left the score unchanged. A with-arm denial aimed + at or under the plugin's directory (suite.root or a + suite.plugins path), at a directory above it, or with + no absolute path or no known plugin directory, is + always a FAIL + skill fired every with-arm run of a should-trigger case fired the skill + models the model and judge model, where the result or traces + carry them; runs that disagree are a FAIL + judge votes a split llm judge vote is a warning + ceiling a case whose without-arm is already 1.00 is a warning and + is left out of the delta + +The last line is `verdict: VALID` or `verdict: INVALID` with its reasons. A +FAIL, or a check that could not run (UNCHECKED), makes the run INVALID; a WARN +does not. + +A should-trigger case has a `tool_used` grader on the Skill tool that requires a +call. A case is a no-trigger control, and exempt, when its prompt.md tags or +description mark it one, or it has a max-0 Skill grader with no input_match (or one +aimed at the plugin's own skills) and no grader requiring a call. + +Exit codes: + + 0 VALID + 1 INVALID + 2 usage error, or the result file is missing, unreadable, or not JSON +""" + +import argparse +import json +import os +import re +import sys + +MIN_PYTHON = (3, 8) + +FULL_SCORE = 1.0 +ARMS = ("with", "without") # cases[].arms.[] +ONE_ARM_ABLATION = "none" # suite.ablation value that runs the with-arm only + +# Result-file field paths. Correct them in this block, and nowhere else, if a +# real result file shows another shape. Every lookup is tolerant: a missing +# field never raises. +PARTIAL = "partial" +PARTIAL_REASON = "partialReason" +CLAUDE_VERSION = "claudeVersion" +SUITE = "suite" +SUITE_ROOT = "root" # suite.root, the directory case dirs are relative to +SUITE_ABLATION = "ablation" # suite.ablation +SUITE_PLUGINS, PLUGIN_PATH = "plugins", "path" # suite.plugins[].path +CASES = "cases" +CASE_NAME = "name" +CASE_DIR = "dir" # cases[].dir, relative to suite.root +CASE_PROMPT = "prompt.md" # //prompt.md, its frontmatter +CASE_RUNS_STATED = "runsPerCase" # the case default, not the run count +CASE_ARMS = "arms" +CASE_GRADER_DEFS = "graders" # cases[].graders[]: name, type, config +GRADER_CONFIG = "config" # cases[].graders[].config +CONFIG_TOOL, CONFIG_MIN, CONFIG_MAX = "tool", "min", "max" +CONFIG_INPUT_MATCH = "input_match" +RUN_ERROR = "error" # cases[].arms.[].error, null when the run ended normally +RUN_ABORTED = "aborted" # present when a mock stopped the run +RUN_SKIPPED_PAID = "skippedPaidGraders" +RUN_SCORE = "score" +RUN_TRACE = "tracePath" # absent or deleted unless the run passed --keep-temp +RUN_GRADERS = "graders" # cases[].arms.[].graders[] +GRADER_NAME = "name" +GRADER_TYPE = "type" +GRADER_PASSED = "passed" +GRADER_VOTES = "judgeVotes" # llm graders: one boolean per vote +# The result file names neither model as of Claude Code 2.1.287; these keys are +# read at the top level and under suite in case a later schema adds them. +MODEL_KEYS = ("model",) +JUDGE_MODEL_KEYS = ("judgeModel", "judge_model") + +# Trace (out/trace.jsonl) shapes, one JSON object per line. +LINE_TYPE, LINE_SUBTYPE = "type", "subtype" +TRACE_INIT = ("system", "init") # type, subtype of the line carrying the model +TRACE_MODEL = "model" +TRACE_RESULT = "result" # type of the closing line +TRACE_RESULT_ERROR = "is_error" +TRACE_DENIALS = "permission_denials" # on the result line +DENIAL_TOOL, DENIAL_ID, DENIAL_INPUT = "tool_name", "tool_use_id", "tool_input" +MESSAGE, CONTENT, BLOCK_TYPE = "message", "content", "type" # message.content[].type +TOOL_USE, USE_ID, USE_NAME, USE_INPUT = "tool_use", "id", "name", "input" +TOOL_RESULT, RESULT_USE_ID, RESULT_ERROR = "tool_result", "tool_use_id", "is_error" +DENIAL_TEXT = "denied by your permission settings" # in an is_error tool_result +# The input a denial line prints: Read's file, or Grep's and Glob's search root, +# else the call's first input value. +TARGET_KEYS = ("file_path", "path") + +# The suite's skill-fired convention and the no-trigger control markers. +SKILL_TOOL = "Skill" +FIRE_GRADER_TYPE = "tool_used" +CONTROL_TAGS = ("control", "no-trigger", "negative-control") +CONTROL_DESCRIPTION = re.compile( + r"\bno[- ]trigger\b|\bwithout invoking\b|\bmust not (?:fire|trigger|invoke)\b", + re.IGNORECASE, +) + +PASS, WARN, FAIL, UNCHECKED = "PASS", "WARN", "FAIL", "UNCHECKED" + + +def as_list(value): + return [v for v in value if isinstance(v, dict)] if isinstance(value, list) else [] + + +def as_dict(value): + return value if isinstance(value, dict) else {} + + +def is_number(value): + return isinstance(value, (int, float)) and not isinstance(value, bool) + + +def resolve(path, base): + if not isinstance(path, str) or not path: + return None + return path if os.path.isabs(path) else os.path.join(base, path) + + +def plural(n, word): + return "%d %s%s" % (n, word, "" if n == 1 else "s") + + +def where(case, arm, index): + return "case %s, %s-arm run %d" % (case.get(CASE_NAME), arm, index) + + +class Run: + def __init__(self, case, arm, index, row, base): + self.case, self.arm, self.row = case, arm, row + self.where = where(case, arm, index) + self.trace_path = resolve(row.get(RUN_TRACE), base) + self.trace = read_trace(self.trace_path) if self.trace_path else None + + +def read_trace(path): + """The trace's JSON lines, or None when the file cannot be read or holds none.""" + try: + with open(path, encoding="utf-8") as handle: + lines = [] + for text in handle: + try: + line = json.loads(text) + except ValueError: + continue + if isinstance(line, dict): + lines.append(line) + return lines or None + except OSError: + return None + + +def frontmatter(path): + """description and tags from a prompt.md, or {} when it cannot be read.""" + try: + with open(path, encoding="utf-8") as handle: + text = handle.read() + except (OSError, TypeError): + return {} + parts = text.split("---", 2) + if not text.startswith("---") or len(parts) < 3: + return {} + fields, key = {"tags": []}, None + for line in parts[1].splitlines(): + item = re.match(r"\s*-\s+(.+)", line) + if item and key == "tags": + fields["tags"].append(item.group(1).strip().strip("'\"")) + continue + pair = re.match(r"(\w+):\s*(.*)", line) + if not pair: + continue + key, value = pair.group(1), pair.group(2).strip() + if key == "tags" and value.startswith("["): + fields["tags"] = [ + t.strip().strip("'\"") for t in value.strip("[]").split(",") + ] + elif key == "description": + fields["description"] = value.strip("'\"") + return fields + + +def grader_input_match(case, grader, root): + """A grader's input_match, from the result config or its graders/.md.""" + config = as_dict(grader.get(GRADER_CONFIG)) + if isinstance(config.get(CONFIG_INPUT_MATCH), str): + return config[CONFIG_INPUT_MATCH] + case_dir, name = case.get(CASE_DIR), grader.get(GRADER_NAME) + if not root or not isinstance(case_dir, str) or not isinstance(name, str): + return None + path = os.path.join(root, case_dir, "graders", name + ".md") + try: + with open(path, encoding="utf-8") as handle: + text = handle.read() + except OSError: + return None + parts = text.split("---", 2) + if not text.startswith("---") or len(parts) < 3: + return None + pair = re.search(r"^input_match:\s*(.+?)\s*$", parts[1], re.MULTILINE) + if not pair: + return None + value = pair.group(1) + if value.startswith('"'): + try: + return json.loads(value) + except ValueError: + return value.strip('"') + return value.strip("'") + + +def targets_own_skills(pattern, root): + """True when pattern matches a Skill call to one of this plugin's own skills.""" + if not root: + return False + try: + names = os.listdir(os.path.join(root, "skills")) + found = re.compile(pattern) + except (OSError, re.error): + return False + plugin = os.path.basename(os.path.normpath(root)) + return any( + found.search(json.dumps({"skill": skill})) + for name in names + for skill in (name, "%s:%s" % (plugin, name)) + ) + + +def skill_graders(case, root): + """(names of graders requiring a Skill call, True when an unscoped one forbids a call). + + A max-0 grader is unscoped when it has no input_match or its input_match targets + this plugin's own skills; a guard on another skill does not make a control. + """ + requiring, forbids = [], False + for grader in as_list(case.get(CASE_GRADER_DEFS)): + config = as_dict(grader.get(GRADER_CONFIG)) + if ( + grader.get(GRADER_TYPE) != FIRE_GRADER_TYPE + or config.get(CONFIG_TOOL) != SKILL_TOOL + ): + continue + if config.get(CONFIG_MAX) == 0: + pattern = grader_input_match(case, grader, root) + if pattern is None or targets_own_skills(pattern, root): + forbids = True + elif not (is_number(config.get(CONFIG_MIN)) and config[CONFIG_MIN] < 1): + requiring.append(grader.get(GRADER_NAME)) + return requiring, forbids + + +def is_control(case, root): + """Control: tagged or described as one, else it forbids own-skill calls and + requires none (a should-trigger case may also carry a max-0 guard).""" + case_dir = case.get(CASE_DIR) + if root and isinstance(case_dir, str): + fields = frontmatter(os.path.join(root, case_dir, CASE_PROMPT)) + tags = [t.lower() for t in fields.get("tags", [])] + if any(t in CONTROL_TAGS for t in tags) or CONTROL_DESCRIPTION.search( + fields.get("description", "") + ): + return True + requiring, forbids = skill_graders(case, root) + return forbids and not requiring + + +def check_complete(result, runs): + if not runs: + return ( + FAIL, + "the result holds no runs", + [], + "the result holds no runs, so nothing was measured", + ) + if result.get(PARTIAL) is True: + return ( + FAIL, + "partial is true (%s): the suite did not finish" + % result.get(PARTIAL_REASON), + [], + "the run is partial (%s)" % result.get(PARTIAL_REASON), + ) + return PASS, "partial is false", [], None + + +def check_paid(runs): + hits = [r.where for r in runs if r.row.get(RUN_SKIPPED_PAID) is True] + if hits: + return ( + FAIL, + "%s skipped paid graders" % plural(len(hits), "run"), + hits, + ( + "%s skipped paid graders, which then score as failures" + % plural(len(hits), "run") + ), + ) + return PASS, "no run skipped its paid graders", [], None + + +def check_errors(runs): + details = [] + for r in runs: + if r.row.get(RUN_ERROR) is not None: + details.append("%s: error %s" % (r.where, r.row.get(RUN_ERROR))) + if r.row.get(RUN_ABORTED): + details.append( + "%s: aborted %s" % (r.where, json.dumps(r.row.get(RUN_ABORTED))) + ) + if r.trace and any( + line.get(LINE_TYPE) == TRACE_RESULT and line.get(TRACE_RESULT_ERROR) is True + for line in r.trace + ): + details.append( + "%s: its trace ends in an error (%s)" % (r.where, r.trace_path) + ) + if details: + count = plural(len(details), "run error") + return FAIL, count, details, count + return PASS, "no run errored or aborted", [], None + + +def check_rows(cases, runs_wanted, arms): + details, per_case = [], set() + for case in cases: + for arm in arms: + count = len(as_list(as_dict(case.get(CASE_ARMS)).get(arm))) + if count != runs_wanted: + details.append( + "case %s, %s-arm: %s, --runs asked for %d" + % (case.get(CASE_NAME), arm, plural(count, "row"), runs_wanted) + ) + per_case.add(case.get(CASE_RUNS_STATED)) + ignored = "" + stated = [v for v in per_case if v is not None] + if any(v != runs_wanted for v in stated): + ignored = "; runsPerCase reads %s and is not used" % ", ".join( + str(v) for v in sorted(stated, key=str) + ) + if details: + return ( + FAIL, + "%s off the requested count%s" % (plural(len(details), "arm"), ignored), + details, + "%s hold a row count other than --runs %d" + % (plural(len(details), "arm"), runs_wanted), + ) + return ( + PASS, + "%s per arm in every case, as --runs %d asked%s" + % (plural(runs_wanted, "row"), runs_wanted, ignored), + [], + None, + ) + + +def denials(run): + """Every denied tool call in one trace, keyed by tool_use_id, as (detail line, + target path or None).""" + calls, found = {}, {} + for line in run.trace: + for block in as_list(as_dict(line.get(MESSAGE)).get(CONTENT)): + if block.get(BLOCK_TYPE) == TOOL_USE: + calls[block.get(USE_ID)] = ( + block.get(USE_NAME), + as_dict(block.get(USE_INPUT)), + ) + elif ( + block.get(BLOCK_TYPE) == TOOL_RESULT + and block.get(RESULT_ERROR) is True + and DENIAL_TEXT in json.dumps(block.get(CONTENT)) + ): + found.setdefault(block.get(RESULT_USE_ID), None) + if line.get(LINE_TYPE) == TRACE_RESULT: + for denial in as_list(line.get(TRACE_DENIALS)): + found[denial.get(DENIAL_ID)] = ( + denial.get(DENIAL_TOOL), + as_dict(denial.get(DENIAL_INPUT)), + ) + out = [] + for use_id, call in found.items(): + name, tool_input = call or calls.get(use_id, (None, {})) + path = next((tool_input[k] for k in TARGET_KEYS if k in tool_input), None) + target = path if path is not None else next(iter(tool_input.values()), "") + out.append( + ( + "%s: %s %s (%s)" % (run.where, name or "tool", target, run.trace_path), + path if isinstance(path, str) else None, + ) + ) + return out + + +def inside(path, roots): + """True unless path is absolute and neither under nor above any plugin root. + An ancestor of a root covers the plugin's files, so it counts as inside; so + do no path, a relative path and no known root. Symlinks resolve on both sides.""" + if not path or not os.path.isabs(path) or not roots: + return True + path = os.path.realpath(path) + return any( + os.path.commonpath([path, r]) in (path, r) + for r in (os.path.realpath(root) for root in roots) + ) + + +def same_as_clean(run, found, runs, roots): + """True when a denied run scores the same as every denial-free run of its case + in the same arm, the case has at least one, every sibling in that arm has a + readable trace, and no with-arm denial is aimed inside the plugin.""" + if run.arm == "with" and any(inside(path, roots) for _, path in found[run]): + return False + siblings = [ + r for r in runs if r.case is run.case and r.arm == run.arm and r is not run + ] + if any(r not in found for r in siblings): + return False + score = run.row.get(RUN_SCORE) + clean = [r.row.get(RUN_SCORE) for r in siblings if not found[r]] + return ( + is_number(score) + and bool(clean) + and all(is_number(s) and abs(s - score) < 1e-9 for s in clean) + ) + + +def check_denials(runs, roots): + traced = [r for r in runs if r.trace is not None] + found = {r: denials(r) for r in traced} + denied = [r for r in traced if found[r]] + warned = [r for r in denied if same_as_clean(r, found, runs, roots)] + failed = [r for r in denied if r not in warned] + details = [line for r in denied for line, _ in found[r]] + summary = "%s in %s" % (plural(len(details), "denial"), plural(len(denied), "run")) + warned_cases = sorted({r.case.get(CASE_NAME) for r in warned}, key=str) + if warned: + summary += ( + "; runs scoring the same as their case's denial-free runs in the same" + " arm, with no with-arm denial inside the plugin, are warnings: %s" + % ", ".join(warned_cases) + ) + if failed: + reason = "%s in the traces" % plural( + sum(len(found[r]) for r in failed), "permission denial" + ) + if warned: + reason += " (warnings only, not counted: case %s)" % ", ".join(warned_cases) + return FAIL, summary, details, reason + if warned and len(traced) == len(runs): + return ( + WARN, + summary, + details, + "permission denials in case %s" % ", ".join(warned_cases), + ) + if len(traced) < len(runs): + missing = len(runs) - len(traced) + return ( + UNCHECKED, + "%s; %s not read" + % ( + summary if warned else "no denial in the %d traces read" % len(traced), + plural(missing, "trace"), + ), + details, + "permission denials unchecked in %s with no readable trace" + % plural(missing, "run"), + ) + return PASS, "no denied tool call in %s" % plural(len(traced), "trace"), [], None + + +def check_fired(cases, runs, root): + unfired, unread, controls, unknown, checked = [], [], [], [], 0 + for case in cases: + requiring, _ = skill_graders(case, root) + if is_control(case, root): + controls.append(case.get(CASE_NAME)) + continue + if not requiring: + unknown.append(case.get(CASE_NAME)) + continue + for r in runs: + if r.case is not case or r.arm != "with": + continue + results = {g.get(GRADER_NAME): g for g in as_list(r.row.get(RUN_GRADERS))} + verdicts = [ + results[n].get(GRADER_PASSED) for n in requiring if n in results + ] + if not verdicts: + unread.append("%s: no skill-fired grader result" % r.where) + elif all(v is True for v in verdicts): + checked += 1 + else: + unfired.append("%s (%s)" % (r.where, r.trace_path or "no trace")) + notes = [] + if controls: + notes.append("exempt as no-trigger controls: %s" % ", ".join(controls)) + if unknown: + notes.append("no skill-fired grader, so not checked: %s" % ", ".join(unknown)) + tail = ("; " + "; ".join(notes)) if notes else "" + if unfired: + return ( + FAIL, + "%d of %d with-arm runs of should-trigger cases did not fire%s" + % (len(unfired), len(unfired) + checked + len(unread), tail), + ["unfired: " + u for u in unfired] + unread, + ( + "the skill did not fire in %s of a should-trigger case" + % plural(len(unfired), "with-arm run") + ), + ) + if unread: + return ( + UNCHECKED, + "%s carry no skill-fired result%s" % (plural(len(unread), "run"), tail), + unread, + "skill firing unchecked in %s" % plural(len(unread), "run"), + ) + if unknown: + return WARN, "fired in %s%s" % (plural(checked, "with-arm run"), tail), [], None + return ( + PASS, + "fired in all %s of should-trigger cases%s" + % (plural(checked, "with-arm run"), tail), + [], + None, + ) + + +def recorded(result, keys): + for scope in (result, as_dict(result.get(SUITE))): + for key in keys: + if isinstance(scope.get(key), str) and scope.get(key): + return scope[key] + return None + + +def check_models(result, runs): + traced = [r for r in runs if r.trace is not None] + models = {} + for r in traced: + for line in r.trace: + if (line.get(LINE_TYPE), line.get(LINE_SUBTYPE)) == TRACE_INIT and line.get( + TRACE_MODEL + ): + models.setdefault(line[TRACE_MODEL], []).append(r) + break + stated = recorded(result, MODEL_KEYS) + if stated: + models.setdefault(stated, []) + judge = recorded(result, JUDGE_MODEL_KEYS) + version = result.get(CLAUDE_VERSION) or "not recorded" + judge_text = ( + "judge model %s" % judge + if judge + else "judge model not recorded in the result or the traces; record the --judge-model the run used" + ) + if len(models) > 1: + details = [ + "%s: %s" % (model, ", ".join(r.where for r in rs) or "the result file") + for model, rs in sorted(models.items()) + ] + return ( + FAIL, + "runs used %d different models; %s" % (len(models), judge_text), + details, + ("runs used different models (%s)" % ", ".join(sorted(models))), + ) + if not models: + return ( + WARN, + "model not recorded in the result or the traces; Claude Code %s; %s" + % (version, judge_text), + [], + None, + ) + ((model, rs),) = models.items() + status = PASS if judge else WARN + return ( + status, + "model %s in %d of %d traces; Claude Code %s; %s" + % (model, len(rs), len(runs), version, judge_text), + [], + None, + ) + + +def check_votes(runs): + details = [] + for r in runs: + for grader in as_list(r.row.get(RUN_GRADERS)): + votes = grader.get(GRADER_VOTES) + if isinstance(votes, list) and len(set(map(bool, votes))) > 1: + details.append( + "%s, grader %s: %s" + % ( + r.where, + grader.get(GRADER_NAME), + " ".join("PASS" if v else "FAIL" for v in votes), + ) + ) + if details: + return ( + WARN, + "%s; read its explanation and evidence before trusting it" + % plural(len(details), "split vote"), + details, + None, + ) + return PASS, "no split llm judge vote", [], None + + +def arm_score(case, arm): + scores = [r.get(RUN_SCORE) for r in as_list(as_dict(case.get(CASE_ARMS)).get(arm))] + if not scores or not all(is_number(s) for s in scores): + return None + return sum(scores) / len(scores) + + +def check_ceiling(cases, arms): + if "without" not in arms: + return PASS, "no without-arm in this run, so no ceiling to read", [], None + at_ceiling, deltas = [], [] + for case in cases: + with_score, without_score = arm_score(case, "with"), arm_score(case, "without") + if without_score is None or with_score is None: + continue + if without_score >= FULL_SCORE - 1e-9: + at_ceiling.append(case.get(CASE_NAME)) + else: + deltas.append(with_score - without_score) + if not at_ceiling: + return PASS, "no case has its without-arm at 1.00", [], None + rest = ( + "the delta over the other %s is %+.2f" + % (plural(len(deltas), "case"), sum(deltas) / len(deltas)) + if deltas + else "no case is left for a delta" + ) + return ( + WARN, + "without-arm already at 1.00 in %s; excluded from the delta, and %s" + % (", ".join(at_ceiling), rest), + [], + None, + ) + + +def main(argv=None): + if sys.version_info < MIN_PYTHON: + sys.stderr.write( + "error: run-validity needs Python %d.%d or newer\n" % MIN_PYTHON + ) + return 2 + parser = argparse.ArgumentParser( + prog="run-validity", + description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + parser.add_argument("result", help="path to aggregate-result.json") + parser.add_argument( + "--runs", + type=int, + required=True, + help="the run count the eval was invoked with", + ) + args = parser.parse_args(argv) + if args.runs < 1: + sys.stderr.write("error: --runs must be 1 or more\n") + return 2 + try: + with open(args.result, encoding="utf-8") as handle: + result = json.load(handle) + except (OSError, ValueError) as error: + sys.stderr.write("error: cannot read %s (%s)\n" % (args.result, error)) + return 2 + if not isinstance(result, dict): + sys.stderr.write("error: %s is not a result document\n" % args.result) + return 2 + + base = os.path.dirname(os.path.abspath(args.result)) + suite = as_dict(result.get(SUITE)) + root = resolve(suite.get(SUITE_ROOT), base) + plugin_roots = [ + os.path.normpath(p) + for p in [root] + + [ + resolve(plugin.get(PLUGIN_PATH), base) + for plugin in as_list(suite.get(SUITE_PLUGINS)) + ] + if p + ] + arms = ("with",) if suite.get(SUITE_ABLATION) == ONE_ARM_ABLATION else ARMS + cases = as_list(result.get(CASES)) + runs = [ + Run(case, arm, index, row, base) + for case in cases + for arm in ARMS + for index, row in enumerate(as_list(as_dict(case.get(CASE_ARMS)).get(arm)), 1) + ] + unresolved = [r for r in runs if r.trace is None] + + checks = [ + ("complete", check_complete(result, runs)), + ("paid graders", check_paid(runs)), + ("run errors", check_errors(runs)), + ("row count", check_rows(cases, args.runs, arms)), + ] + lines = [] + for name, (status, summary, details, _) in checks: + lines.append("check %s: %s (%s)" % (name, status, summary)) + lines += [" " + d for d in details] + if unresolved: + lines.append( + "traces: %d of %d tracePaths do not resolve (the run needs --keep-temp, and its" + " directories must still exist); permission denials and the trace model are" + " unchecked for those runs" % (len(unresolved), len(runs)) + ) + lines += [ + " %s: %s" % (r.where, r.trace_path or "no tracePath") for r in unresolved + ] + else: + lines.append("traces: %d of %d tracePaths resolve" % (len(runs), len(runs))) + later = [ + ("permission denials", check_denials(runs, plugin_roots)), + ("skill fired", check_fired(cases, runs, root)), + ("models", check_models(result, runs)), + ("judge votes", check_votes(runs)), + ("ceiling", check_ceiling(cases, arms)), + ] + for name, (status, summary, details, _) in later: + lines.append("check %s: %s (%s)" % (name, status, summary)) + lines += [" " + d for d in details] + checks += later + + reasons = [ + reason for _, (status, _, _, reason) in checks if status in (FAIL, UNCHECKED) + ] + warned = [ + reason or name for name, (status, _, _, reason) in checks if status == WARN + ] + for line in lines: + print(line) + if reasons: + print("verdict: INVALID (%s)" % "; ".join(reasons)) + return 1 + print("verdict: VALID" + (" (warnings: %s)" % ", ".join(warned) if warned else "")) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/evals/skills/plugin-eval/scripts/run-validity.test.sh b/plugins/evals/skills/plugin-eval/scripts/run-validity.test.sh new file mode 100755 index 0000000000..b6ed470ea3 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/run-validity.test.sh @@ -0,0 +1,41 @@ +#!/usr/bin/env bash +# Cross-platform wrapper for run-validity.py's unittest suite, so the repo's +# run-plugin-tests.sh discovery (plugins/**/*.test.sh) runs it. The interpreter +# discovery follows plugins/evals/skills/validate/scripts/validate-cases.test.sh. +# +# Exit: 0 all tests passed; 1 a test failed; 2 no usable interpreter (a named +# environment error, never a silent skip). +set -uo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ENGINE="$SCRIPT_DIR/run-validity.py" +SUITE="$SCRIPT_DIR/test_run_validity.py" + +# The Python floor has one origin: MIN_PYTHON in the engine. +FLOOR="$(sed -n 's/^MIN_PYTHON = (\([0-9]*\), \([0-9]*\)).*/\1.\2/p' "$ENGINE")" +if [[ -z "$FLOOR" ]]; then + echo "error: could not parse MIN_PYTHON from $ENGINE" >&2 + exit 2 +fi + +# A zero-length candidate under a WindowsApps path component is the Store's App +# Execution Alias stub, which opens the Microsoft Store instead of running an +# interpreter, so each candidate is inspected before anything executes it. +PYTHON="" +for candidate in python3 python; do + resolved="$(command -v "$candidate" 2>/dev/null)" || continue + lower="$(printf '%s' "$resolved" | tr '[:upper:]' '[:lower:]')" + if [[ "$lower" == *windowsapps* && ! -s "$resolved" ]]; then + continue + fi + if "$candidate" -c "import sys; floor = tuple(int(part) for part in '$FLOOR'.split('.')); raise SystemExit(0 if sys.version_info >= floor else 1)" 2>/dev/null; then + PYTHON="$candidate" + break + fi +done +if [[ -z "$PYTHON" ]]; then + echo "error: Python ${FLOOR}+ not found (tried python3, python) -- run-validity.py's suite cannot run" >&2 + exit 2 +fi + +"$PYTHON" "$SUITE" diff --git a/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py b/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py new file mode 100755 index 0000000000..7b8d775fea --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py @@ -0,0 +1,537 @@ +#!/usr/bin/env python3 +"""Fixture suite for calibrate-judge.py. + +fixtures/calibrate-judge/ holds: + + suite/capital-city/ one case with one llm grader (names-paris, two + must-pass and three must-fail samples, one of them + empty and so skipped) and one tool_used grader that + build must ignore + result.json a hand-written aggregate-result.json for the suite + build writes from it: agreement, a false negative, a + split vote, a reply that is not the sample (read from + a kept trace), a run that skipped its paid graders, + and a run whose reproduction cannot be checked + traces/ the two kept traces result.json points at + +result.json uses the case names build derives, so it also pins the naming. +Other tests copy a fixture into a temporary directory and change one thing. + +The suite is executed by calibrate-judge.test.sh, which run-plugin-tests.sh +discovers. Run it directly with: python3 test_calibrate_judge.py +""" + +import copy +import importlib.util +import json +import shutil +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent +SCRIPT = HERE / "calibrate-judge.py" +VALIDATOR = HERE.parent.parent / "validate" / "scripts" / "validate-cases.py" +FIXTURES = HERE / "fixtures" / "calibrate-judge" +SUITE = FIXTURES / "suite" +RESULT = FIXTURES / "result.json" +PLUGIN_SUITE = HERE.parents[2] / "evals" +PROMPT = "What is the capital of France? Answer in one sentence." + + +def load_module(name, path): + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +calibrate = load_module("calibrate_judge", SCRIPT) +validate_cases = load_module("validate_cases", VALIDATOR) + + +def run(*args): + return subprocess.run( + [sys.executable, str(SCRIPT), *map(str, args)], + capture_output=True, + text=True, + check=False, + ) + + +def frontmatter_and_body(path): + block, body = calibrate.split_body(path.read_text(encoding="utf-8")) + return validate_cases.parse_yaml(block), body + + +class Base(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.dir = Path(self.tmp.name) + + def build(self, suite=SUITE, *extra, out_name="out"): + out = self.dir / out_name + return run("build", "--suite", suite, "--out", out, *extra), out + + def manifest(self, out): + return json.loads((out / "manifest.json").read_text(encoding="utf-8")) + + def copy_suite(self): + target = self.dir / "suite" + shutil.copytree(SUITE, target) + return target + + +class BuildTest(Base): + def test_one_case_per_labelled_sample_of_the_llm_grader(self): + proc, out = self.build() + self.assertEqual(proc.returncode, 0, proc.stdout + proc.stderr) + self.assertIn("wrote 4 calibration cases", proc.stdout) + self.assertIn( + "capital-city/names-paris: 4 (2 must-pass, 2 must-fail)", proc.stdout + ) + self.assertIn("validate-cases: PASS (0 FAIL, 4 WARN)", proc.stdout) + self.assertIn("--ablation none --threshold 0", proc.stdout) + cases = self.manifest(out)["cases"] + self.assertEqual( + sorted((c["label"], c["sampleIndex"], c["expected"]) for c in cases), + [ + ("fail", 2, "FAIL"), + ("fail", 3, "FAIL"), + ("pass", 1, "PASS"), + ("pass", 2, "PASS"), + ], + ) + self.assertEqual({c["grader"] for c in cases}, {"names-paris"}) + self.assertEqual({c["sourceCase"] for c in cases}, {"capital-city"}) + self.assertEqual( + sorted(p.name for p in (out / "evals").iterdir()), + sorted(c["case"] for c in cases), + ) + plugin = json.loads((out / ".claude-plugin" / "plugin.json").read_text()) + self.assertEqual(plugin["name"], "judge-calibration") + + def test_case_names_match_the_scored_fixture(self): + _, out = self.build() + names = {c["case"] for c in self.manifest(out)["cases"]} + fixture = {c["name"] for c in json.loads(RESULT.read_text())["cases"]} + self.assertEqual(names, fixture) + + def test_generated_suite_passes_validate_cases(self): + _, out = self.build() + proc = subprocess.run( + [sys.executable, str(VALIDATOR), str(out / "evals")], + capture_output=True, + text=True, + check=False, + ) + self.assertEqual(proc.returncode, 0, proc.stdout + proc.stderr) + self.assertNotIn("FAIL", proc.stdout) + + def test_each_case_reproduces_its_sample_and_sends_the_source_prompt(self): + _, out = self.build() + for entry in self.manifest(out)["cases"]: + case = out / "evals" / entry["case"] + fields, body = frontmatter_and_body(case / "prompt.md") + self.assertEqual(body.strip(), PROMPT) + self.assertEqual(fields["allowed_tools"], []) + self.assertEqual( + fields["append_system_prompt"], calibrate.system_prompt(entry["answer"]) + ) + self.assertIn( + "\nBEGIN-REFERENCE\n%s\nEND-REFERENCE" % entry["answer"], + fields["append_system_prompt"], + ) + self.assertEqual( + (case / "graders" / "names-paris.md").read_text(), + (SUITE / "capital-city" / "graders" / "names-paris.md").read_text(), + ) + self.assertEqual( + sorted(p.name for p in case.rglob("*") if p.is_file()), + ["names-paris.md", "prompt.md"], + ) + + def test_the_label_never_reaches_the_generated_case(self): + _, out = self.build() + entries = self.manifest(out)["cases"] + shapes = set() + for entry in entries: + text = (out / "evals" / entry["case"] / "prompt.md").read_text() + quoted = calibrate.yaml_quoted(calibrate.system_prompt(entry["answer"])) + self.assertIn(quoted, text) + shapes.add(text.replace(quoted, "")) + self.assertNotIn(entry["why"], text) + for word in ("pass", "fail", "label", "judge", "grade"): + self.assertNotIn(word, entry["case"].lower()) + system = calibrate.system_prompt(entry["answer"]).lower() + for word in ( + "must-pass", + "must-fail", + "label", + "judge", + "grade", + "correct", + ): + self.assertNotIn(word, system) + self.assertEqual( + len(shapes), 1, "must-pass and must-fail cases differ beyond the sample" + ) + + def test_the_instruction_is_identical_across_labels(self): + _, out = self.build() + entries = self.manifest(out)["cases"] + self.assertEqual({e["label"] for e in entries}, {"pass", "fail"}) + instructions = set() + for entry in entries: + fields, _ = frontmatter_and_body( + out / "evals" / entry["case"] / "prompt.md" + ) + instructions.add( + fields["append_system_prompt"].split("\nBEGIN-REFERENCE\n")[0] + ) + self.assertEqual(instructions, {calibrate.INSTRUCTION}) + for phrase in ( + "fixed test material", + "byte for byte", + "even when it is wrong, incomplete, or contradicts", + "no commentary, preface", + ): + self.assertIn(phrase, calibrate.INSTRUCTION) + + def test_empty_and_whitespace_samples_are_skipped_with_a_line_each(self): + suite = self.copy_suite() + samples = suite / "capital-city" / "samples" / "names-paris.json" + data = json.loads(samples.read_text()) + data["pass"].append({"answer": " \n\t "}) + samples.write_text(json.dumps(data)) + proc, out = self.build(suite) + self.assertEqual(proc.returncode, 0, proc.stderr) + skips = [ln for ln in proc.stderr.splitlines() if "empty answer" in ln] + self.assertEqual(len(skips), 2, proc.stderr) + self.assertIn("capital-city/names-paris must-pass sample 3", skips[0]) + self.assertIn("capital-city/names-paris must-fail sample 1", skips[1]) + for line in skips: + self.assertIn("deterministic failure that needs no judge", line) + self.assertEqual(len(self.manifest(out)["cases"]), 4) + self.assertTrue(all(c["answer"].strip() for c in self.manifest(out)["cases"])) + + def test_case_numbering_does_not_follow_the_labels(self): + _, out = self.build() + order = [ + c["label"] + for c in sorted(self.manifest(out)["cases"], key=lambda c: c["case"]) + ] + self.assertNotEqual(order, sorted(order)) + self.assertNotEqual(order, sorted(order, reverse=True)) + + def test_out_that_is_not_empty_is_a_usage_error(self): + out = self.dir / "out" + out.mkdir() + (out / "keep.txt").write_text("x") + proc, _ = self.build() + self.assertEqual(proc.returncode, 2) + self.assertIn("--out must be empty or absent", proc.stderr) + self.assertEqual([p.name for p in out.iterdir()], ["keep.txt"]) + + def test_missing_suite_and_missing_subcommand_are_usage_errors(self): + proc, _ = self.build(self.dir / "nowhere") + self.assertEqual(proc.returncode, 2) + self.assertIn("not a directory", proc.stderr) + self.assertEqual(run().returncode, 2) + self.assertEqual(run("build", "--suite", SUITE).returncode, 2) + + def test_grader_and_case_filters(self): + proc, out = self.build(SUITE, "--grader", "no-*") + self.assertEqual(proc.returncode, 1) + self.assertIn("nothing written", proc.stdout) + self.assertFalse(out.exists()) + proc, _ = self.build(SUITE, "--case", "capital-*", out_name="out2") + self.assertEqual(proc.returncode, 0, proc.stderr) + proc, _ = self.build(SUITE, "--case", "other", out_name="out3") + self.assertEqual(proc.returncode, 1) + + def test_a_grader_that_does_not_read_the_reply_is_skipped(self): + suite = self.copy_suite() + grader = suite / "capital-city" / "graders" / "names-paris.md" + grader.write_text( + "---\ntype: llm\nfocus: { source: file, path: answer.md }\n---\n\nPASS if Paris.\n" + ) + proc, _ = self.build(suite) + self.assertEqual(proc.returncode, 1) + self.assertIn("skip capital-city/names-paris: focus", proc.stderr) + + def test_a_trace_focus_is_built_with_a_warning(self): + suite = self.copy_suite() + grader = suite / "capital-city" / "graders" / "names-paris.md" + grader.write_text("---\ntype: llm\nfocus: trace\n---\n\nPASS if Paris.\n") + proc, out = self.build(suite) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("warn capital-city/names-paris: focus trace", proc.stderr) + self.assertEqual({c["focus"] for c in self.manifest(out)["cases"]}, {"trace"}) + + def test_unusable_samples_are_skipped_and_conflicts_warned(self): + suite = self.copy_suite() + samples = suite / "capital-city" / "samples" / "names-paris.json" + data = json.loads(samples.read_text()) + data["pass"].append({"answer": "Paris\u0007"}) + data["pass"].append({"answer": ["not", "text"]}) + data["fail"].append({"answer": "The capital of France is Paris."}) + samples.write_text(json.dumps(data)) + proc, out = self.build(suite) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("must-pass sample 3: control character U+0007", proc.stderr) + self.assertIn("must-pass sample 4: answer is not text", proc.stderr) + self.assertIn("same answer is labelled both pass and fail", proc.stderr) + self.assertEqual(len(self.manifest(out)["cases"]), 5) + + def test_a_case_yaml_grader_and_prompt_are_read(self): + suite = self.dir / "yaml-suite" + case = suite / "capital-yaml" + (case / "samples").mkdir(parents=True) + (case / "case.yaml").write_text( + 'schema_version: "1.1"\n' + "name: capital-yaml\n" + "execution:\n" + ' prompt: "%s"\n' + "graders:\n" + " - name: names-paris\n" + " type: llm\n" + ' criteria: "PASS if the answer names Paris."\n' % PROMPT + ) + shutil.copy( + SUITE / "capital-city" / "samples" / "names-paris.json", + case / "samples" / "names-paris.json", + ) + proc, out = self.build(suite) + self.assertEqual(proc.returncode, 0, proc.stdout + proc.stderr) + entry = self.manifest(out)["cases"][0] + fields, body = frontmatter_and_body(out / "evals" / entry["case"] / "prompt.md") + self.assertEqual(body.strip(), PROMPT) + grader = ( + out / "evals" / entry["case"] / "graders" / "names-paris.md" + ).read_text() + self.assertEqual( + grader, "---\ntype: llm\n---\n\nPASS if the answer names Paris.\n" + ) + + def test_this_plugins_own_suite_builds_and_validates(self): + proc, out = self.build(PLUGIN_SUITE) + self.assertEqual(proc.returncode, 0, proc.stdout + proc.stderr) + self.assertRegex(proc.stdout, r"validate-cases: PASS \(0 FAIL") + self.assertTrue(self.manifest(out)["cases"]) + + +class YamlQuotingTest(unittest.TestCase): + def test_round_trip_through_the_validator_parser(self): + for text in ( + "plain", + 'a "quote" and C:\\path', + "two\nlines\ttab\r\n", + "", + "caf\u00e9 \u2192", + ): + parsed = validate_cases.parse_yaml("k: " + calibrate.yaml_quoted(text)) + self.assertEqual(parsed["k"], text) + + def test_a_control_character_is_refused(self): + with self.assertRaises(ValueError): + calibrate.yaml_quoted("bell\u0007") + + +class ScoreTest(Base): + def setUp(self): + Base.setUp(self) + proc, self.out = self.build() + self.assertEqual(proc.returncode, 0, proc.stderr) + self.manifest_path = self.out / "manifest.json" + self.result = json.loads(RESULT.read_text(encoding="utf-8")) + + def score(self, result=None, manifest=None): + path = RESULT + if result is not None: + path = self.dir / "result.json" + for case in result["cases"]: + for row in case["arms"]["with"]: + if "tracePath" in row: + row["tracePath"] = str(FIXTURES / row["tracePath"]) + path.write_text(json.dumps(result), encoding="utf-8") + return run("score", "--manifest", manifest or self.manifest_path, path) + + def rows(self, result, case_number): + name = "capital-city--names-paris--%02d" % case_number + return next(c for c in result["cases"] if c["name"] == name)["arms"]["with"] + + def test_fixture_meets_the_target_at_exactly_ninety_percent(self): + proc = self.score() + self.assertEqual(proc.returncode, 0, proc.stdout + proc.stderr) + out = proc.stdout + self.assertIn( + "grader capital-city/names-paris: agreement 9/10 runs (90.0%) over 4 samples", + out, + ) + self.assertIn( + "false negative: capital-city--names-paris--03 with-arm run 1 (must-pass " + 'sample 1: "The capital of France is Paris.") votes FAIL FAIL PASS', + out, + ) + self.assertNotIn("false positive", out) + self.assertIn( + "split vote: capital-city--names-paris--01 with-arm run 2: PASS FAIL PASS, " + "agrees with the label", + out, + ) + self.assertIn( + "not reproduced, left out: capital-city--names-paris--03 with-arm run 2: reply " + 'begins "Paris is the capital of France, and also its largest city." (trace)', + out, + ) + self.assertIn( + "not judged, paid graders skipped: capital-city--names-paris--02 with-arm run 3", + out, + ) + self.assertIn( + "reproduction unchecked, no trace or evidence: capital-city--names-paris--01 " + "with-arm run 3", + out, + ) + self.assertIn( + "reproduction: 2 checked against traces, 8 against judge evidence, 1 unchecked", + out, + ) + self.assertNotIn("FAIL grader", out) + self.assertIn("samples with no reproduced run: 0 of 4", out) + self.assertNotIn("untested:", out) + self.assertEqual( + out.strip().splitlines()[-1], "verdict: PASS (every grader at or above 90%)" + ) + + def test_a_sample_with_no_reproduced_run_is_untested_and_not_counted(self): + result = copy.deepcopy(self.result) + for row in self.rows(result, 3): + row["graders"][0]["evidence"] = "Paris is the capital, and a big city." + row.pop("tracePath", None) + proc = self.score(result) + self.assertEqual(proc.returncode, 0, proc.stdout) + out = proc.stdout + self.assertIn("agreement 8/8 runs (100.0%) over 3 samples", out) + self.assertIn("samples with no reproduced run: 1 of 4", out) + self.assertIn( + "untested: capital-city--names-paris--03 (must-pass sample 1: " + '"The capital of France is Paris."), counted in no agreement', + out, + ) + self.assertEqual( + out.strip().splitlines()[-1], + "verdict: PASS (every grader at or above 90%; 1 untested)", + ) + + def test_a_false_positive_drops_the_grader_under_the_target(self): + result = copy.deepcopy(self.result) + grader = self.rows(result, 4)[0]["graders"][0] + grader["judgeVotes"], grader["passed"] = [True, True, False], True + proc = self.score(result) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("agreement 8/10 runs (80.0%)", proc.stdout) + self.assertIn( + "false positive: capital-city--names-paris--04 with-arm run 1 (must-fail " + 'sample 3: "The capital of France is Marseille.") votes PASS PASS FAIL', + proc.stdout, + ) + self.assertIn( + "FAIL grader capital-city/names-paris: agreement 80.0% is under the 90% target", + proc.stdout, + ) + self.assertTrue( + proc.stdout.strip().endswith("verdict: FAIL (capital-city/names-paris)") + ) + + def test_the_verdict_is_the_vote_majority_not_passed(self): + result = copy.deepcopy(self.result) + grader = self.rows(result, 3)[0]["graders"][0] + grader["passed"] = True # votes stay FAIL FAIL PASS + proc = self.score(result) + self.assertIn( + "false negative: capital-city--names-paris--03 with-arm run 1", proc.stdout + ) + + def test_without_votes_the_passed_field_decides(self): + result = copy.deepcopy(self.result) + grader = self.rows(result, 3)[0]["graders"][0] + del grader["judgeVotes"] + proc = self.score(result) + self.assertIn("no judgeVotes, read passed", proc.stdout) + + def test_a_reproduced_reply_differing_only_in_whitespace_counts(self): + result = copy.deepcopy(self.result) + self.rows(result, 3)[2]["graders"][0]["evidence"] = ( + " The capital\nof France is Paris.\n" + ) + proc = self.score(result) + self.assertEqual(proc.returncode, 0, proc.stdout) + self.assertIn("agreement 9/10 runs", proc.stdout) + + def test_a_reproduced_reply_that_adds_bold_counts(self): + result = copy.deepcopy(self.result) + self.rows(result, 3)[2]["graders"][0]["evidence"] = ( + "**The capital** of France is __Paris__." + ) + proc = self.score(result) + self.assertEqual(proc.returncode, 0, proc.stdout) + self.assertIn("agreement 9/10 runs", proc.stdout) + + def test_a_trace_focus_never_reads_the_evidence(self): + manifest = self.manifest(self.out) + for entry in manifest["cases"]: + entry["focus"] = "trace" + path = self.dir / "trace-manifest.json" + path.write_text(json.dumps(manifest)) + proc = self.score(manifest=path) + self.assertIn( + "reproduction: 2 checked against traces, 0 against judge evidence, 9 unchecked", + proc.stdout, + ) + + def test_a_case_missing_from_the_result_is_named(self): + result = copy.deepcopy(self.result) + result["cases"] = [c for c in result["cases"] if not c["name"].endswith("--02")] + proc = self.score(result) + self.assertIn( + "missing from the result: capital-city--names-paris--02", proc.stdout + ) + + def test_a_grader_with_no_judged_run_fails(self): + result = copy.deepcopy(self.result) + for case in result["cases"]: + for row in case["arms"]["with"]: + row["skippedPaidGraders"] = True + proc = self.score(result) + self.assertEqual(proc.returncode, 1) + self.assertIn("grader capital-city/names-paris: no judged run", proc.stdout) + self.assertIn( + "FAIL grader capital-city/names-paris: no judged run is under the 90% target", + proc.stdout, + ) + + def test_unreadable_inputs_are_usage_errors(self): + bad = self.dir / "bad.json" + bad.write_text("not json") + self.assertEqual(run("score", "--manifest", bad, RESULT).returncode, 2) + self.assertEqual( + run("score", "--manifest", self.manifest_path, bad).returncode, 2 + ) + self.assertEqual( + run("score", "--manifest", self.dir / "none.json", RESULT).returncode, 2 + ) + self.assertEqual(run("score", RESULT).returncode, 2) + empty = self.dir / "empty.json" + empty.write_text('{"cases": []}') + proc = run("score", "--manifest", empty, RESULT) + self.assertEqual(proc.returncode, 2) + self.assertIn("lists no cases", proc.stderr) + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/evals/skills/plugin-eval/scripts/test_noise_report.py b/plugins/evals/skills/plugin-eval/scripts/test_noise_report.py new file mode 100755 index 0000000000..a5fa321aa4 --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/test_noise_report.py @@ -0,0 +1,460 @@ +#!/usr/bin/env python3 +"""Fixture suite for noise-report.py. + +Every fixture is a small synthetic aggregate-result.json built in a temporary +directory, in the shape a real `claude plugin eval` result file carries: a +per-run `score`, a per-run `graders` list joined by name to the case-level +`graders` definitions, and `with` and `without` arms. No real result file is +tracked. + +The suite is executed by noise-report.test.sh, which run-plugin-tests.sh +discovers. Run it directly with: python3 test_noise_report.py +""" + +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parent / "noise-report.py" + +GRADER_NAMES = ("g1", "g2", "g3", "g4") + + +def make_run(score, error=None, skipped=False, include_score=True, graders=None): + """A run whose four equal-weight graders pass in proportion to `score`.""" + if graders is None: + passed = round(score * len(GRADER_NAMES)) + graders = [ + { + "name": name, + "passed": index < passed, + "weight": 1, + "explanation": "", + "withOnly": False, + "scored": True, + } + for index, name in enumerate(GRADER_NAMES) + ] + run = { + "passed": score >= 1.0, + "turns": 3, + "costUsd": 0.1, + "judgeCostUsd": 0, + "durationSeconds": 10, + "startedAt": "2026-10-01T00:00:00.000Z", + "error": error, + "skippedPaidGraders": skipped, + "graders": graders, + } + if include_score: + run["score"] = score + return run + + +def make_case(name, with_scores, without_scores=None, grader_defs=None): + """A case whose arms hold one run per listed score.""" + if grader_defs is None: + grader_defs = [ + {"name": n, "type": "regex", "weight": 1, "config": {}} + for n in GRADER_NAMES + ] + arms = {"with": [make_run(s) for s in with_scores]} + if without_scores is not None: + arms["without"] = [make_run(s) for s in without_scores] + return {"name": name, "graders": grader_defs, "arms": arms, "aggregates": {}} + + +def make_result(cases, partial=False, partial_reason=None, cost=1.5): + return { + "schemaVersion": 1, + "partial": partial, + "partialReason": partial_reason, + "costUsd": cost, + "suite": {"ablation": "with-without", "threshold": 1.0}, + "aggregates": {}, + "cases": cases, + } + + +class NoiseReportTest(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + + def report(self, result, *args): + path = Path(self.tmp.name) / "aggregate-result.json" + path.write_text(json.dumps(result), encoding="utf-8") + proc = subprocess.run( + [sys.executable, str(SCRIPT), str(path), *args], + capture_output=True, + text=True, + check=False, + ) + return proc + + def test_every_delta_plus_one_is_too_small_to_call(self): + cases = [make_case("c%d" % i, [1, 1, 1], [0, 0, 0]) for i in range(4)] + proc = self.report(make_result(cases)) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("n too small to call", proc.stdout) + self.assertNotIn("within noise", proc.stdout) + + def test_delta_interval_containing_zero_is_within_noise(self): + cases = [ + make_case("a", [1, 1, 1], [0.75, 0.75, 0.75]), + make_case("b", [1, 1, 1], [1, 1, 1]), + make_case("c", [0.75, 0.75, 0.75], [1, 1, 1]), + make_case("d", [1, 1, 1], [0.75, 0.75, 0.75]), + ] + proc = self.report(make_result(cases)) + self.assertEqual(proc.returncode, 0, proc.stderr) + # deltas +0.25, 0, -0.25, +0.25: mean 0.0625, sd 0.2394, half-width 0.2346 + self.assertIn( + "delta (with minus without), paired over 4 cases: +0.06, 95% interval -0.17 to +0.30", + proc.stdout, + ) + self.assertIn("within noise", proc.stdout) + self.assertNotIn("n too small to call", proc.stdout) + + def test_delta_interval_excluding_zero_says_so(self): + cases = [ + make_case("a", [1, 1, 1], [0, 0, 0]), + make_case("b", [1, 1, 1], [0.25, 0.25, 0.25]), + make_case("c", [1, 1, 1], [0, 0, 0]), + make_case("d", [1, 1, 1], [0.25, 0.25, 0.25]), + ] + proc = self.report(make_result(cases)) + self.assertIn("95% interval +0.73 to +1.00", proc.stdout) + self.assertIn("the interval excludes 0", proc.stdout) + self.assertNotIn("within noise", proc.stdout) + + def test_fewer_than_three_cases_is_too_small_to_call(self): + cases = [ + make_case("a", [1, 1, 1], [0, 0, 0]), + make_case("b", [1, 1, 1], [0.75, 0.75, 0.75]), + ] + proc = self.report(make_result(cases)) + self.assertIn("n too small to call", proc.stdout) + self.assertNotIn("within noise", proc.stdout) + + def test_each_arm_mean_gets_a_normal_interval(self): + cases = [ + make_case("a", [1, 1, 1], [0.75, 0.75, 0.75]), + make_case("b", [1, 1, 1], [1, 1, 1]), + make_case("c", [0.75, 0.75, 0.75], [1, 1, 1]), + make_case("d", [1, 1, 1], [0.75, 0.75, 0.75]), + ] + proc = self.report(make_result(cases)) + self.assertIn( + "with-arm mean: 0.94, 95% interval 0.82 to 1.00 (normal, 4 cases)", + proc.stdout, + ) + self.assertIn( + "without-arm mean: 0.88, 95% interval 0.73 to 1.00 (normal, 4 cases)", + proc.stdout, + ) + + def test_baseline_at_095_or_higher_leaves_no_headroom(self): + near = [ + make_case("a", [1, 1, 1], [1, 1, 1]), + make_case("b", [1, 1, 1], [1, 1, 1]), + make_case("c", [1, 1, 1], [0.75, 1, 1]), + make_case("d", [1, 1, 1], [1, 1, 1]), + ] + proc = self.report(make_result(near)) + self.assertIn("near ceiling: without-arm mean 0.98", proc.stdout) + self.assertIn("the baseline leaves no headroom", proc.stdout) + + below = [ + make_case("a", [1, 1, 1], [0.75, 0.75, 0.75]), + make_case("b", [1, 1, 1], [1, 1, 1]), + make_case("c", [1, 1, 1], [1, 1, 1]), + ] + proc = self.report(make_result(below)) + self.assertNotIn("no headroom", proc.stdout) + + def pass_count_cases(self): + return [ + make_case("a", [1, 1, 1], [0.75, 0.75, 0.75]), + make_case("b", [1, 1, 1], [1, 1, 1]), + make_case("c", [0.75, 0.75, 0.75], [1, 1, 1]), + make_case("d", [1, 1, 1], [0.75, 0.75, 0.75]), + ] + + def test_pass_count_uses_the_chosen_interval_method(self): + result = make_result(self.pass_count_cases()) + proc = self.report(result) + self.assertIn( + "with-arm pass count at threshold 1.00: 3 of 4, 95% interval 0.33 to 1.00 (normal)", + proc.stdout, + ) + proc = self.report(result, "--interval-method", "wilson") + self.assertIn( + "with-arm pass count at threshold 1.00: 3 of 4, 95% interval 0.30 to 0.95 (wilson)", + proc.stdout, + ) + self.assertIn( + "without-arm pass count at threshold 1.00: 2 of 4, 95% interval 0.15 to 0.85 (wilson)", + proc.stdout, + ) + # Beta(k + 0.5, n - k + 0.5) quantiles, checked by direct numerical integration + proc = self.report(result, "--interval-method", "jeffreys") + self.assertIn( + "with-arm pass count at threshold 1.00: 3 of 4, 95% interval 0.28 to 0.97 (jeffreys)", + proc.stdout, + ) + self.assertIn( + "without-arm pass count at threshold 1.00: 2 of 4, 95% interval 0.12 to 0.88 (jeffreys)", + proc.stdout, + ) + # score intervals stay normal whatever the method + self.assertIn( + "with-arm mean: 0.94, 95% interval 0.82 to 1.00 (normal, 4 cases)", + proc.stdout, + ) + + def test_threshold_moves_the_pass_count(self): + proc = self.report(make_result(self.pass_count_cases()), "--threshold", "0.75") + self.assertIn("with-arm pass count at threshold 0.75: 4 of 4", proc.stdout) + self.assertIn("without-arm pass count at threshold 0.75: 4 of 4", proc.stdout) + + def weighted_graders(self, weight_on_run=True): + graders = [ + { + "name": "result", + "passed": True, + "weight": 3, + "scored": True, + "withOnly": False, + }, + { + "name": "process", + "passed": False, + "weight": 1, + "scored": True, + "withOnly": False, + }, + { + "name": "skill-fired", + "passed": False, + "weight": 1, + "scored": False, + "withOnly": True, + }, + ] + if not weight_on_run: + for grader in graders: + del grader["weight"] + return graders + + def weighted_case(self, weight_on_run=True): + defs = [ + {"name": "result", "type": "llm", "weight": 3, "config": {}}, + {"name": "process", "type": "tool_order", "weight": 1, "config": {}}, + {"name": "skill-fired", "type": "tool_used", "weight": 1, "config": {}}, + ] + case = make_case("w", [], [1, 1, 1], grader_defs=defs) + case["arms"]["with"] = [ + make_run( + 0, include_score=False, graders=self.weighted_graders(weight_on_run) + ) + for _ in range(3) + ] + return case + + def test_missing_run_score_is_recomputed_from_weighted_graders(self): + proc = self.report(make_result([self.weighted_case()])) + self.assertEqual(proc.returncode, 0, proc.stderr) + # result (weight 3) passed, process (weight 1) failed, skill-fired not scored + self.assertIn("with-arm mean: 0.75 (1 case", proc.stdout) + + def test_grader_weight_falls_back_to_the_case_definition(self): + proc = self.report(make_result([self.weighted_case(weight_on_run=False)])) + self.assertIn("with-arm mean: 0.75 (1 case", proc.stdout) + + def test_reported_score_is_cross_checked_against_the_graders(self): + case = self.weighted_case() + case["arms"]["with"][1]["score"] = 1.0 + case["arms"]["with"][2]["score"] = 0.75 + proc = self.report(make_result([case])) + self.assertIn( + "score check: case w, with-arm run 2 reports 1.00, its graders give 0.75;" + " the reported score is used", + proc.stdout, + ) + self.assertNotIn("run 3 reports", proc.stdout) + # run 1 has no score, run 2 reports 1.00, run 3 reports 0.75 + self.assertIn("with-arm mean: 0.83 (1 case", proc.stdout) + + def test_incomparable_cases_are_named_and_left_out(self): + errored = make_case("errored", [1, 1, 1], [0, 0, 0]) + errored["arms"]["with"][0] = make_run( + 0, error="timed out after 300s", graders=[] + ) + skipped = make_case("skipped", [1, 1, 1], [0, 0, 0]) + skipped["arms"]["without"][2]["skippedPaidGraders"] = True + omitted = make_case("omitted", [1, 1, 1], [0, 0, 0]) + omitted["aggregates"] = {"score": 1, "passRate": 1} + cases = [ + errored, + skipped, + omitted, + make_case("a", [1, 1, 1], [0, 0, 0]), + make_case("b", [1, 1, 1], [0.25, 0.25, 0.25]), + make_case("c", [1, 1, 1], [0, 0, 0]), + ] + proc = self.report(make_result(cases)) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn( + "not comparable: case errored (with-arm run 1 ended with an error)", + proc.stdout, + ) + self.assertIn( + "not comparable: case skipped (without-arm run 3 skipped its paid graders)", + proc.stdout, + ) + self.assertIn( + "not comparable: case omitted (the result omits its delta)", proc.stdout + ) + self.assertIn("paired over 3 cases", proc.stdout) + self.assertIn("with-arm pass count at threshold 1.00: 3 of 3", proc.stdout) + + def test_one_arm_result_reports_the_with_arm_only(self): + cases = [make_case(name, [1, 1, 0.75]) for name in ("a", "b", "c")] + proc = self.report(make_result(cases)) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("with-arm mean: 0.92", proc.stdout) + self.assertIn( + "no without-arm runs in this result, so there is no delta to read", + proc.stdout, + ) + self.assertNotIn("without-arm mean", proc.stdout) + self.assertNotIn("headroom", proc.stdout) + + def test_partial_result_stops_before_any_number(self): + cases = [make_case(name, [1, 1, 1], [0, 0, 0]) for name in ("a", "b", "c")] + proc = self.report( + make_result(cases, partial=True, partial_reason="cost_ceiling") + ) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("partial result (cost_ceiling)", proc.stdout) + self.assertNotIn("mean", proc.stdout) + self.assertNotIn("verdict", proc.stdout) + + def test_unreadable_result_file_exits_2(self): + path = Path(self.tmp.name) / "broken.json" + path.write_text("{not json", encoding="utf-8") + proc = subprocess.run( + [sys.executable, str(SCRIPT), str(path)], + capture_output=True, + text=True, + check=False, + ) + self.assertEqual(proc.returncode, 2) + self.assertIn("error:", proc.stderr) + proc = subprocess.run( + [sys.executable, str(SCRIPT), str(Path(self.tmp.name) / "missing.json")], + capture_output=True, + text=True, + check=False, + ) + self.assertEqual(proc.returncode, 2) + + def voted_case(self, votes): + case = self.weighted_case() + for run, run_votes in zip(case["arms"]["with"], votes): + run["graders"][0]["judgeVotes"] = run_votes + return case + + def test_judge_vote_agreement_per_llm_grader(self): + case = self.voted_case( + [[True, True, True], [True, True, False], [True, True, True]] + ) + proc = self.report(make_result([case]), "--grader-agreement") + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn( + "judge agreement: case w, grader result: unanimous in 2 of 3 runs" + " (8 of 9 votes match their run's majority)", + proc.stdout, + ) + # tool_order and tool_used graders carry no judge + self.assertNotIn("grader process", proc.stdout) + + def test_judge_votes_as_verdict_objects_are_read(self): + verdicts = [[{"verdict": "PASS"}, {"verdict": "FAIL"}, {"verdict": "PASS"}]] * 3 + proc = self.report( + make_result([self.voted_case(verdicts)]), "--grader-agreement" + ) + self.assertIn("unanimous in 0 of 3 runs (6 of 9 votes", proc.stdout) + + def test_agreement_line_is_off_without_the_flag(self): + case = self.voted_case([[True, True, True]] * 3) + proc = self.report(make_result([case])) + self.assertNotIn("judge", proc.stdout) + + def test_missing_judge_votes_are_reported_not_invented(self): + proc = self.report(make_result([self.weighted_case()]), "--grader-agreement") + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn( + "the result file holds no judge votes, so agreement cannot be read", + proc.stdout, + ) + self.assertNotIn("unanimous", proc.stdout) + + def test_suite_without_llm_graders_has_no_agreement_to_read(self): + cases = [make_case(name, [1, 1, 1], [0, 0, 0]) for name in ("a", "b", "c")] + proc = self.report(make_result(cases), "--grader-agreement") + self.assertIn( + "no llm grader in this result, so there is no judge agreement to read", + proc.stdout, + ) + + def test_cost_prints_beside_the_scores(self): + cases = [make_case(name, [1, 1, 1], [0, 0, 0]) for name in ("a", "b", "c")] + proc = self.report(make_result(cases, cost=3.2)) + self.assertIn( + "cost: 3.20 USD for the whole suite (list-price estimate)", proc.stdout + ) + + def test_unknown_interval_method_falls_back_to_normal(self): + proc = self.report( + make_result(self.pass_count_cases()), "--interval-method", "bayes" + ) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn( + "interval method 'bayes' is not normal, wilson or jeffreys; using normal", + proc.stdout, + ) + self.assertIn("3 of 4, 95% interval 0.33 to 1.00 (normal)", proc.stdout) + + def test_case_missing_its_without_arm_is_not_comparable(self): + cases = [make_case(n, [1, 1, 1], [0, 0, 0]) for n in ("a", "b", "c")] + cases.append(make_case("lonely", [1, 1, 1])) + proc = self.report(make_result(cases)) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn( + "not comparable: case lonely (no without-arm runs for this case)", + proc.stdout, + ) + self.assertIn("paired over 3 cases", proc.stdout) + + def test_non_list_graders_value_does_not_crash(self): + case = make_case("odd", [1, 1, 1], [0, 0, 0]) + run = case["arms"]["with"][0] + del run["score"] + run["graders"] = 5 + cases = [case] + [make_case(n, [1, 1, 1], [0, 0, 0]) for n in ("a", "b", "c")] + proc = self.report(make_result(cases)) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("not comparable: case odd", proc.stdout) + + def test_too_small_verdict_names_the_count_without_a_plural_slip(self): + proc = self.report(make_result([make_case("a", [1, 1, 1], [0, 0, 0])])) + self.assertIn("n too small to call (comparable cases: 1;", proc.stdout) + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/evals/skills/plugin-eval/scripts/test_run_validity.py b/plugins/evals/skills/plugin-eval/scripts/test_run_validity.py new file mode 100755 index 0000000000..89aac8a05e --- /dev/null +++ b/plugins/evals/skills/plugin-eval/scripts/test_run_validity.py @@ -0,0 +1,613 @@ +#!/usr/bin/env python3 +"""Fixture suite for run-validity.py. + +fixtures/run-validity/ holds two result files trimmed from one real +`claude plugin eval` run (Claude Code 2.1.287, --runs 2, --keep-temp) and the +structural lines of its kept traces, with no answer text: + + r2-invalid.json the whole run: 3 denied Reads of the methodology skill's + reference files and one with-arm run whose skill never + fired, so it must come out INVALID + r2-clean-runs.json one run per arm per case, taken from the runs of that same + result that pass every check, so it must come out VALID + traces/ clean.jsonl stands in for every trace with no denial + +The other tests copy a fixture into a temporary directory and change one field. + +The suite is executed by run-validity.test.sh, which run-plugin-tests.sh +discovers. Run it directly with: python3 test_run_validity.py +""" + +import copy +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path +from typing import Optional + +HERE = Path(__file__).resolve().parent +SCRIPT = HERE / "run-validity.py" +FIXTURES = HERE / "fixtures" / "run-validity" +TRACES = FIXTURES / "traces" +INVALID = FIXTURES / "r2-invalid.json" +CLEAN = FIXTURES / "r2-clean-runs.json" + + +def load(path): + return json.loads(path.read_text(encoding="utf-8")) + + +def runs_of(result): + for case in result["cases"]: + for arm in ("with", "without"): + yield from case["arms"].get(arm, []) + + +class RunValidityTest(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.dir = Path(self.tmp.name) + + def run_script(self, path, *args): + return subprocess.run( + [sys.executable, str(SCRIPT), str(path), *args], + capture_output=True, + text=True, + check=False, + ) + + def write(self, result, absolute_traces=True): + """Write a result into the temp dir, its tracePaths pointed at the fixture traces.""" + result = copy.deepcopy(result) + if absolute_traces: + for run in (r for r in runs_of(result) if "tracePath" in r): + run["tracePath"] = str(FIXTURES / run["tracePath"]) + path = self.dir / "aggregate-result.json" + path.write_text(json.dumps(result), encoding="utf-8") + return path + + def clean(self): + return load(CLEAN) + + def assertVerdict(self, proc, verdict, code): + self.assertEqual(proc.returncode, code, proc.stdout + proc.stderr) + self.assertTrue( + proc.stdout.strip().splitlines()[-1].startswith("verdict: " + verdict), + proc.stdout, + ) + + def test_r2_is_invalid_for_its_denials_and_its_unfired_skill(self): + proc = self.run_script(INVALID, "--runs", "2") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + "check permission denials: FAIL (3 denials in 3 runs)", proc.stdout + ) + for run, file in ( + ("case grading-method-choice, with-arm run 1", "grading.md"), + ("case grading-method-choice, with-arm run 2", "grading.md"), + ("case measurable-criterion, with-arm run 1", "success-criteria.md"), + ): + self.assertIn( + " %s: Read /tmp/eval-r2/plugins/evals/skills/methodology/reference/%s" + % (run, file), + proc.stdout, + ) + self.assertIn( + "check skill fired: FAIL (1 of 6 with-arm runs of should-trigger cases did not fire;" + " exempt as no-trigger controls: control-no-trigger)", + proc.stdout, + ) + self.assertIn(" unfired: case noise-before-gain, with-arm run 1", proc.stdout) + self.assertIn( + "verdict: INVALID (3 permission denials in the traces; the skill did not fire" + " in 1 with-arm run of a should-trigger case)", + proc.stdout, + ) + + def test_r2_reads_rows_not_runs_per_case(self): + proc = self.run_script(INVALID, "--runs", "2") + self.assertIn( + "check row count: PASS (2 rows per arm in every case, as --runs 2 asked;" + " runsPerCase reads 3 and is not used)", + proc.stdout, + ) + proc = self.run_script(INVALID, "--runs", "3") + self.assertIn( + "check row count: FAIL (8 arms off the requested count)", proc.stdout + ) + self.assertIn( + " case noise-before-gain, without-arm: 2 rows, --runs asked for 3", + proc.stdout, + ) + self.assertIn("8 arms hold a row count other than --runs 3", proc.stdout) + + def test_r2_warnings_do_not_decide_the_verdict(self): + proc = self.run_script(INVALID, "--runs", "2") + self.assertIn( + "check models: WARN (model claude-opus-5-5 in 16 of 16 traces; Claude Code 2.1.287;" + " judge model not recorded in the result or the traces", + proc.stdout, + ) + self.assertIn( + " case noise-before-gain, with-arm run 2, grader noise-verdict: FAIL FAIL PASS", + proc.stdout, + ) + self.assertIn( + "check ceiling: WARN (without-arm already at 1.00 in control-no-trigger," + " measurable-criterion; excluded from the delta, and the delta over the other" + " 2 cases is -0.25)", + proc.stdout, + ) + verdict = proc.stdout.strip().splitlines()[-1] + self.assertNotIn("model", verdict) + self.assertNotIn("ceiling", verdict) + + def test_clean_runs_are_valid_with_their_warnings_named(self): + proc = self.run_script(CLEAN, "--runs", "1") + self.assertVerdict(proc, "VALID", 0) + self.assertIn("traces: 6 of 6 tracePaths resolve", proc.stdout) + self.assertIn( + "check permission denials: PASS (no denied tool call in 6 traces)", + proc.stdout, + ) + self.assertIn( + "check skill fired: PASS (fired in all 2 with-arm runs", proc.stdout + ) + self.assertIn("the delta over the other 1 case is +0.00", proc.stdout) + self.assertIn( + "verdict: VALID (warnings: models, judge votes, ceiling)", proc.stdout + ) + + def test_unresolved_traces_leave_denials_unchecked_and_invalid(self): + path = self.write(self.clean(), absolute_traces=False) + proc = self.run_script(path, "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("traces: 6 of 6 tracePaths do not resolve", proc.stdout) + self.assertIn("check permission denials: UNCHECKED", proc.stdout) + self.assertIn( + "permission denials unchecked in 6 runs with no readable trace", proc.stdout + ) + self.assertNotIn("check permission denials: PASS", proc.stdout) + + def test_a_missing_trace_path_is_named(self): + result = self.clean() + del result["cases"][0]["arms"]["with"][0]["tracePath"] + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + " case control-no-trigger, with-arm run 1: no tracePath", proc.stdout + ) + + def test_a_result_with_no_runs_is_invalid(self): + result = self.clean() + result["cases"] = [] + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check complete: FAIL (the result holds no runs)", proc.stdout) + + def test_an_empty_trace_leaves_denials_unchecked(self): + empty = self.dir / "empty.jsonl" + empty.write_text("", encoding="utf-8") + result = self.write(self.clean()) + data = json.loads(result.read_text(encoding="utf-8")) + data["cases"][0]["arms"]["with"][0]["tracePath"] = str(empty) + result.write_text(json.dumps(data), encoding="utf-8") + proc = self.run_script(result, "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check permission denials: UNCHECKED", proc.stdout) + + def test_partial_run_is_invalid(self): + result = self.clean() + result["partial"], result["partialReason"] = True, "cost_ceiling" + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check complete: FAIL", proc.stdout) + self.assertIn("the run is partial (cost_ceiling)", proc.stdout) + + def test_skipped_paid_graders_are_invalid(self): + result = self.clean() + result["cases"][1]["arms"]["without"][0]["skippedPaidGraders"] = True + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn(" case measurable-criterion, without-arm run 1", proc.stdout) + self.assertIn("1 run skipped paid graders", proc.stdout) + + def test_errored_or_aborted_runs_are_invalid(self): + result = self.clean() + result["cases"][0]["arms"]["with"][0]["error"] = "rate limited" + result["cases"][2]["arms"]["with"][0]["aborted"] = { + "server": "s", + "tool": "t", + "reason": "r", + } + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + " case control-no-trigger, with-arm run 1: error rate limited", proc.stdout + ) + self.assertIn(" case noise-before-gain, with-arm run 1: aborted", proc.stdout) + self.assertIn("2 run errors", proc.stdout) + + def test_a_trace_that_ends_in_an_error_is_invalid(self): + trace = self.dir / "errored.jsonl" + trace.write_text( + '{"type":"system","subtype":"init","model":"claude-opus-5-5"}\n' + '{"type":"result","subtype":"error_during_execution","is_error":true}\n', + encoding="utf-8", + ) + result = self.write(self.clean()) + data = load(result) + data["cases"][0]["arms"]["without"][0]["tracePath"] = str(trace) + result.write_text(json.dumps(data), encoding="utf-8") + proc = self.run_script(result, "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + "case control-no-trigger, without-arm run 1: its trace ends in an error", + proc.stdout, + ) + + def test_a_denial_seen_only_as_a_tool_result_is_counted(self): + trace = self.dir / "denied.jsonl" + trace.write_text( + '{"type":"system","subtype":"init","model":"claude-opus-5-5"}\n' + '{"type":"assistant","message":{"content":[{"type":"tool_use","id":"t1","name":"Grep","input":{"pattern":"x"}}]}}\n' + '{"type":"user","message":{"content":[{"type":"tool_result","tool_use_id":"t1","is_error":true,' + '"content":[{"type":"text","text":"Path is denied by your permission settings."}]}]}}\n', + encoding="utf-8", + ) + result = self.write(self.clean()) + data = load(result) + data["cases"][1]["arms"]["with"][0]["tracePath"] = str(trace) + result.write_text(json.dumps(data), encoding="utf-8") + proc = self.run_script(result, "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check permission denials: FAIL (1 denial in 1 run)", proc.stdout) + self.assertIn( + " case measurable-criterion, with-arm run 1: Grep x", proc.stdout + ) + + def test_a_denial_names_the_path_it_was_aimed_at(self): + trace = self.dir / "denied-paths.jsonl" + trace.write_text( + '{"type":"system","subtype":"init","model":"claude-opus-5-5"}\n' + '{"type":"result","is_error":false,"permission_denials":[' + '{"tool_name":"Grep","tool_use_id":"g1","tool_input":' + '{"pattern":"GRADER_TYPES","path":"/p/validate-cases.py"}},' + '{"tool_name":"Glob","tool_use_id":"g2","tool_input":' + '{"pattern":"**/package.json","path":"/usr"}},' + '{"tool_name":"Read","tool_use_id":"r1","tool_input":' + '{"limit":40,"file_path":"/p/reference/ci.md"}}]}\n', + encoding="utf-8", + ) + result = self.write(self.clean()) + data = load(result) + data["cases"][1]["arms"]["with"][0]["tracePath"] = str(trace) + result.write_text(json.dumps(data), encoding="utf-8") + proc = self.run_script(result, "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + "check permission denials: FAIL (3 denials in 1 run)", proc.stdout + ) + where = " case measurable-criterion, with-arm run 1: " + self.assertIn(where + "Grep /p/validate-cases.py (", proc.stdout) + self.assertIn(where + "Glob /usr (", proc.stdout) + self.assertIn(where + "Read /p/reference/ci.md (", proc.stdout) + + def denied_second_run( + self, + arm, + score, + first_denied=False, + also_with=None, + missing=None, + path: Optional[str] = "/tmp/claude-eval-x", + root=None, + ): + """The clean result at two runs per arm, measurable-criterion's second + `arm` run denied a Grep aimed at `path` (None: no path) and scoring + `score`; with first_denied, its first run of that arm is denied too; with + also_with, the second with-arm run of the case at that index is denied as + well; with missing, a (case index, arm, run index) whose tracePath does + not resolve; with root, the suite.root the plugin's directory is.""" + tool_input = {"pattern": "judge"} + if path is not None: + tool_input["path"] = path + trace = self.dir / "denied-root.jsonl" + trace.write_text( + '{"type":"system","subtype":"init","model":"claude-opus-5-5"}\n' + '{"type":"result","is_error":false,"permission_denials":[' + '{"tool_name":"Grep","tool_use_id":"g1","tool_input":%s}]}\n' + % json.dumps(tool_input), + encoding="utf-8", + ) + data = load(self.write(self.clean())) + if root is not None: + data["suite"]["root"] = root + for case in data["cases"]: + for name in ("with", "without"): + case["arms"][name].append(copy.deepcopy(case["arms"][name][0])) + rows = data["cases"][1]["arms"][arm] + rows[1]["tracePath"], rows[1]["score"] = str(trace), score + if first_denied: + rows[0]["tracePath"] = str(trace) + if also_with is not None: + data["cases"][also_with]["arms"]["with"][1]["tracePath"] = str(trace) + if missing is not None: + index, name, run = missing + data["cases"][index]["arms"][name][run]["tracePath"] = str( + self.dir / "absent.jsonl" + ) + result = self.dir / "aggregate-result.json" + result.write_text(json.dumps(data), encoding="utf-8") + return self.run_script(result, "--runs", "2") + + def test_a_with_arm_denial_with_no_plugin_root_is_a_fail_even_when_its_score_matches( + self, + ): + proc = self.denied_second_run("with", 1) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check permission denials: FAIL (1 denial in 1 run)", proc.stdout) + + PLUGIN = "/tmp/eval-r2/plugins/evals" + + def test_a_with_arm_denial_outside_the_plugin_scoring_the_same_is_a_warning(self): + proc = self.denied_second_run("with", 1, root=self.PLUGIN) + self.assertVerdict(proc, "VALID", 0) + self.assertIn("check permission denials: WARN (1 denial in 1 run", proc.stdout) + + def test_a_with_arm_denial_outside_the_plugin_scoring_differently_is_a_fail(self): + proc = self.denied_second_run("with", 0, root=self.PLUGIN) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check permission denials: FAIL (1 denial in 1 run)", proc.stdout) + + def test_a_with_arm_denial_inside_the_plugin_is_a_fail(self): + proc = self.denied_second_run( + "with", + 1, + root=self.PLUGIN, + path=self.PLUGIN + "/skills/methodology/reference/grading.md", + ) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check permission denials: FAIL (1 denial in 1 run)", proc.stdout) + + def test_a_with_arm_denial_above_the_plugin_is_a_fail(self): + for ancestor in ("/", "/tmp", "/tmp/eval-r2/plugins"): + proc = self.denied_second_run("with", 1, root=self.PLUGIN, path=ancestor) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + "check permission denials: FAIL (1 denial in 1 run)", proc.stdout + ) + + def test_a_with_arm_denial_beside_the_plugin_name_prefix_is_a_warning(self): + proc = self.denied_second_run( + "with", 1, root=self.PLUGIN, path=self.PLUGIN + "-other" + ) + self.assertVerdict(proc, "VALID", 0) + + def test_a_with_arm_denial_with_no_path_is_a_fail(self): + proc = self.denied_second_run("with", 1, root=self.PLUGIN, path=None) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check permission denials: FAIL (1 denial in 1 run)", proc.stdout) + + def test_a_without_arm_denial_scoring_like_its_clean_runs_is_a_warning(self): + proc = self.denied_second_run("without", 1) + self.assertVerdict(proc, "VALID", 0) + self.assertIn("check permission denials: WARN (1 denial in 1 run", proc.stdout) + self.assertIn( + " case measurable-criterion, without-arm run 2: Grep /tmp/claude-eval-x", + proc.stdout, + ) + self.assertIn( + "permission denials in case measurable-criterion", + proc.stdout.splitlines()[-1], + ) + + def test_an_invalid_verdict_still_names_the_warned_case(self): + proc = self.denied_second_run("without", 1, also_with=2) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + "check permission denials: FAIL (2 denials in 2 runs", proc.stdout + ) + self.assertIn( + "1 permission denial in the traces (warnings only, not counted:" + " case measurable-criterion)", + proc.stdout.splitlines()[-1], + ) + + def test_a_without_arm_denial_scoring_differently_is_a_fail(self): + proc = self.denied_second_run("without", 0) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check permission denials: FAIL (1 denial in 1 run)", proc.stdout) + + def test_a_without_arm_denial_with_no_denial_free_run_is_a_fail(self): + proc = self.denied_second_run("without", 1, first_denied=True) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + "check permission denials: FAIL (2 denials in 2 runs)", proc.stdout + ) + + def test_a_warned_denial_beside_an_unread_trace_is_unchecked(self): + proc = self.denied_second_run("without", 1, missing=(2, "with", 0)) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + "check permission denials: UNCHECKED (1 denial in 1 run", proc.stdout + ) + + def test_a_without_arm_denial_with_an_unread_sibling_is_a_fail(self): + proc = self.denied_second_run("without", 1, missing=(1, "without", 0)) + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check permission denials: FAIL (1 denial in 1 run)", proc.stdout) + + def test_skill_text_quoting_the_denial_is_not_a_denial(self): + trace = self.dir / "quoted.jsonl" + trace.write_text( + '{"type":"system","subtype":"init","model":"claude-opus-5-5"}\n' + '{"type":"user","message":{"content":[{"type":"text","text":"a Read is refused with' + ' File is in a directory that is denied by your permission settings"}]}}\n', + encoding="utf-8", + ) + result = self.write(self.clean()) + data = load(result) + data["cases"][2]["arms"]["with"][0]["tracePath"] = str(trace) + result.write_text(json.dumps(data), encoding="utf-8") + proc = self.run_script(result, "--runs", "1") + self.assertVerdict(proc, "VALID", 0) + + def test_runs_on_different_models_are_invalid(self): + trace = self.dir / "other-model.jsonl" + trace.write_text( + '{"type":"system","subtype":"init","model":"claude-sonnet-5"}\n', + encoding="utf-8", + ) + result = self.write(self.clean()) + data = load(result) + data["cases"][1]["arms"]["without"][0]["tracePath"] = str(trace) + result.write_text(json.dumps(data), encoding="utf-8") + proc = self.run_script(result, "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check models: FAIL (runs used 2 different models", proc.stdout) + self.assertIn( + " claude-sonnet-5: case measurable-criterion, without-arm run 1", + proc.stdout, + ) + + def test_a_recorded_judge_model_clears_the_models_warning(self): + result = self.clean() + result["suite"]["judgeModel"] = "sonnet" + proc = self.run_script(self.write(result), "--runs", "1") + self.assertIn( + "check models: PASS (model claude-opus-5-5 in 6 of 6 traces;", proc.stdout + ) + self.assertIn("judge model sonnet)", proc.stdout) + + def fired_case_with_tags(self, tags_line): + result = self.clean() + noise = result["cases"][2] + noise["arms"]["with"][0]["graders"][1]["passed"] = False + noise["dir"] = "evals/noise-before-gain" + result["suite"]["root"] = "suite" + prompt = self.dir / "suite" / "evals" / "noise-before-gain" / "prompt.md" + prompt.parent.mkdir(parents=True, exist_ok=True) + prompt.write_text( + "---\ndescription: d\n%s\nruns: 1\n---\n\nQ\n" % tags_line, encoding="utf-8" + ) + return self.run_script(self.write(result), "--runs", "1") + + def test_a_case_tagged_control_is_exempt(self): + proc = self.fired_case_with_tags("tags: [knowledge, control]") + self.assertVerdict(proc, "VALID", 0) + self.assertIn( + "exempt as no-trigger controls: control-no-trigger, noise-before-gain", + proc.stdout, + ) + proc = self.fired_case_with_tags("tags:\n - knowledge\n - no-trigger") + self.assertVerdict(proc, "VALID", 0) + + def test_an_untagged_unfired_case_is_invalid(self): + proc = self.fired_case_with_tags("tags: [knowledge, hard]") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn(" unfired: case noise-before-gain, with-arm run 1", proc.stdout) + + def suite_with_guard(self, case, guard_match): + """Give a case a max-0 Skill guard whose input_match is read from its grader file.""" + result = self.clean() + case_result = result["cases"][case] + case_result["dir"] = "evals/" + case_result["name"] + case_result["graders"].insert( + 0, + { + "name": "guard", + "type": "tool_used", + "config": {"tool": "Skill", "min": 0, "max": 0}, + }, + ) + result["suite"]["root"] = "suite" + suite = self.dir / "suite" + (suite / "skills" / "methodology").mkdir(parents=True) + graders = suite / case_result["dir"] / "graders" + graders.mkdir(parents=True) + (graders / "guard.md").write_text( + "---\ntype: tool_used\ntool: Skill\ninput_match: %s\nmin: 0\nmax: 0\n---\n" + % json.dumps(guard_match), + encoding="utf-8", + ) + return result + + def test_a_should_trigger_case_with_a_guard_on_another_skill_is_not_exempt(self): + result = self.suite_with_guard(2, '"skill"\\s*:\\s*"claude-api".*build-eval') + result["cases"][2]["arms"]["with"][0]["graders"][1]["passed"] = False + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn(" unfired: case noise-before-gain, with-arm run 1", proc.stdout) + self.assertNotIn("control-no-trigger, noise-before-gain", proc.stdout) + + def test_a_case_with_a_guard_on_its_own_skills_and_no_min_grader_is_exempt(self): + result = self.suite_with_guard(0, '"skill"\\s*:\\s*"(?:evals:)?methodology"') + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "VALID", 0) + self.assertIn("exempt as no-trigger controls: control-no-trigger", proc.stdout) + + def test_a_with_run_missing_its_skill_fired_result_is_unchecked(self): + result = self.clean() + del result["cases"][1]["arms"]["with"][0]["graders"][1] + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn("check skill fired: UNCHECKED", proc.stdout) + + def test_a_case_with_no_skill_grader_is_a_warning(self): + result = self.clean() + result["cases"][1]["graders"] = result["cases"][1]["graders"][:1] + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "VALID", 0) + self.assertIn("check skill fired: WARN", proc.stdout) + self.assertIn( + "no skill-fired grader, so not checked: measurable-criterion", proc.stdout + ) + + def test_one_arm_run_expects_no_without_rows(self): + result = self.clean() + result["suite"]["ablation"] = "none" + for case in result["cases"]: + del case["arms"]["without"] + proc = self.run_script(self.write(result), "--runs", "1") + self.assertIn("check row count: PASS", proc.stdout) + self.assertIn("check ceiling: PASS (no without-arm in this run", proc.stdout) + self.assertVerdict(proc, "VALID", 0) + + def test_a_two_arm_case_missing_its_without_arm_is_invalid(self): + result = self.clean() + del result["cases"][0]["arms"]["without"] + proc = self.run_script(self.write(result), "--runs", "1") + self.assertVerdict(proc, "INVALID", 1) + self.assertIn( + " case control-no-trigger, without-arm: 0 rows, --runs asked for 1", + proc.stdout, + ) + + def test_usage_and_unreadable_input_exit_2(self): + broken = self.dir / "broken.json" + broken.write_text("{not json", encoding="utf-8") + for args in ( + [str(broken), "--runs", "1"], + [str(self.dir / "missing.json"), "--runs", "1"], + [str(CLEAN)], + [str(CLEAN), "--runs", "0"], + ): + proc = subprocess.run( + [sys.executable, str(SCRIPT), *args], + capture_output=True, + text=True, + check=False, + ) + self.assertEqual(proc.returncode, 2, args) + self.assertNotIn("verdict:", proc.stdout) + listed = self.dir / "list.json" + listed.write_text("[]", encoding="utf-8") + self.assertEqual(self.run_script(listed, "--runs", "1").returncode, 2) + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/evals/skills/validate/SKILL.md b/plugins/evals/skills/validate/SKILL.md index ce606b62ec..6e89f6947e 100644 --- a/plugins/evals/skills/validate/SKILL.md +++ b/plugins/evals/skills/validate/SKILL.md @@ -1,5 +1,5 @@ --- -description: "Statically validate a `claude plugin eval` suite (`prompt.md`, `case.yaml`, `graders/*.md`) before any run spends money. A standard-library Python script reports FAIL for what the binary rejects at load (unknown frontmatter key, unknown grader option, no grader, duplicate grader name, non-positive weight, out-of-range runs / max_turns / timeout_seconds, an env key outside EVAL_[A-Z0-9_]*, an unsupported schema_version major) and WARN for the documented authoring mistakes (target: files, inline (?i), judge-only graders, a gated tool in allowed_tools, file_exists in a read-only case). Use when: 'validate my eval cases', 'check my eval suite', 'will this suite load', 'lint case.yaml', 'check my graders', 'why did my case fail to load', or before paying for a run. Not for the skill-creator `evals/evals.json` format." +description: "Statically validate a `claude plugin eval` suite (`prompt.md`, `case.yaml`, `graders/*.md`) before any run spends money. A standard-library Python script reports FAIL for what the binary rejects at load (unknown frontmatter key, unknown grader option, no grader, duplicate grader name, non-positive weight, out-of-range runs / max_turns / timeout_seconds, an env key outside EVAL_[A-Z0-9_]*, an unsupported schema_version major) and WARN for the documented authoring mistakes (target: files, inline (?i), judge-only graders, a gated tool in allowed_tools, file_exists in a read-only case). It also grades each deterministic grader offline against the case's samples/.json answers. Use when: 'validate my eval cases', 'check my eval suite', 'will this suite load', 'lint case.yaml', 'check my graders', 'why did my case fail to load', or before paying for a run. Not for the skill-creator `evals/evals.json` format." argument-hint: "[eval-dir]" user-invocable: true disable-model-invocation: false @@ -42,20 +42,80 @@ usable `type`, an unknown option for the declared grader type, a case with no gr graders sharing a name, a non-positive `weight`, `runs` / `max_turns` / `timeout_seconds` outside their bounds, an `env` key that does not match `EVAL_[A-Z0-9_]*`, a `case.yaml` with no companion `prompt.md` and no `schema_version` or `name`, and a `schema_version` whose major is newer than the -binary supports. The exact key sets and bounds -live in the script's own constants, under the drift record its header carries, so there is one place -to correct when the schema moves. +binary supports. The sets the script checks, complete, so a question about a key or a type is +answered here without opening the script: + +- `prompt.md` frontmatter keys: `schema_version`, `name`, `description`, `tags`, `plugins`, + `runs`, `expected_outcome`, `model`, `max_turns`, `timeout_seconds`, `allowed_tools`, + `append_system_prompt`, `env`. A time limit is `timeout_seconds`; `timeout` is an unknown key, so + the case fails to load. +- Grader `type`: `regex`, `tool_order`, `tool_used`, `file_exists`, `llm`, `baseline`. Any other + value, such as `contains` or `string_match`, is no usable type, so the case fails to load. A + substring or pattern check is `type: regex` with the text in `pattern`. +- Options per type, beside the `type`, `weight`, `arm` and `name` every grader takes: `regex` + `pattern`, `flags`, `match`, `target`; `tool_used` `tool`, `input_match`, `min`, `max`; + `tool_order` `before`, `after`; `file_exists` `path`, `exists`; `llm` `criteria`, `focus`; + `baseline` `baseline_file`, `criteria`. +- Bounds: `runs` 1 to 50, `max_turns` 1 to 200, `timeout_seconds` 1 to 3600. + +The script's constants are the one place to correct when the schema moves, under the drift record +its header carries; this list follows them. **WARN is an authoring mistake the suite loads with** and then scores badly on: `target: files` (which reads the list of paths created, not their contents), an inline `(?i)` the grader's regex engine does not honor, a case whose graders are all judges (the two types that cost money, with no deterministic grader beside them), a tool in `allowed_tools` that the operator must grant with -`--allow-tools`, and `file_exists` in a case that requests no write tool, so nothing it could match -is ever created. The judge-only rule -is this skill's own pairing heuristic, not a rejection the binary makes. +`--allow-tools`, `file_exists` in a case that requests no write tool, so nothing it could match +is ever created, and a `prompt.md` key the binary loads but the reference page does not list, which +can change without notice (the script's `UNDOCUMENTED_PROMPT_KEYS`, under its own drift record). +The judge-only rule +is this skill's own pairing heuristic, not a rejection the binary makes. The sample-answer check +below adds its own FAIL and WARN findings; those are this skill's checks, not rejections either. A WARN never sets exit 1, so a suite can ship with warnings on purpose. Read them once and decide. +## Sample answers + +A grader that rejects a correct paraphrase scores the plugin down for nothing, and no run tells you +the grader was at fault. So each case can carry known answers, and the script runs the free graders +over them before any money is spent. + +Put them in `samples/.json` inside the case directory, beside `graders/` and never in +it. That location rests on the case layout and run isolation described in the +[eval suite reference](https://code.claude.com/docs/en/plugin-evals#eval-suite-reference) and on the +case loader in Claude Code 2.1.287, which reads only `prompt.md`, `case.yaml` and `graders/`. As of +2026-10-01; recheck when a release adds a file the loader reads from a case directory. + +```json +{ + "pass": [{"answer": "...", "why": "the ranking rule with 'and' for the comma"}], + "fail": [{"answer": ""}, {"answer": "...", "why": "near miss: base-model phrasing"}] +} +``` + +`answer` is what the grader reads: text for a `regex` (the final reply on the default target), a +list of `{"tool", "input"}` calls for `tool_used` and `tool_order`, and a list of created paths for +`file_exists`. `why` is optional and is echoed in any finding. Give each grader several paraphrases +it must pass, near misses it must reject, an empty answer, and an answer to a different question. + +| Finding | Level | +|---|---| +| A must-pass answer the grader rejects, or a must-fail answer it accepts | FAIL | +| A sample file that is not JSON in this shape, or an answer of the wrong kind for its grader | FAIL | +| A `regex`, `tool_used`, `tool_order` or `file_exists` grader with no sample file | WARN | +| A sample file with no must-pass or no must-fail answers, or named after no grader | WARN | +| A setting the script cannot reproduce offline, such as the `y` or `v` regex flag | WARN | +| An `llm` or `baseline` grader with samples: they need a paid judge calibration run | WARN | + +The script never calls a judge. Samples on an `llm` grader are the labelled answers a calibration +run feeds the judge, and that run is the operator's to start. + +**Claim:** the sample check grades as the binary does. **Basis:** the grader code in Claude Code +2.1.287, read against the [grader types](https://code.claude.com/docs/en/plugin-evals#grader-types) +table; Python's `re` stands in for the JavaScript regex engine, and the script's header lists where +the two differ. **As of:** 2026-10-01. **Recheck trigger:** the next Claude Code release, or a +grader whose sample verdict disagrees with its verdict in a real run. + ## When the parser stops The script reads a bounded YAML subset: scalars, quoted scalars, flow lists and mappings, block @@ -68,8 +128,8 @@ which carries the full parser. ## Mirroring claim, and its recheck trigger -**Claim:** the FAIL tier matches what the binary rejects when it loads a case, so a suite at exit 0 -loads. **Basis:** the case schema recovered from the shipped Claude Code binary, read against the +**Claim:** the load-time FAIL findings match what the binary rejects when it loads a case, so a +suite at exit 0 loads. **Basis:** the case schema recovered from the shipped Claude Code binary, read against the eval-suite reference at . **As of:** 2026-09-12. **Recheck trigger:** the next Claude Code release, which can add a frontmatter key, add a grader type, or move a bound. On a firing, re-derive both lists from the schema rather than patching one @@ -94,6 +154,8 @@ case validator at an `evals.json` reports only that the directory holds no cases - Exit 0 means the suite **loads**, not that it measures anything. A case whose graders pass in both arms proves the plugin contributed nothing; the delta is read after a run, not here. +- Samples that all sort correctly prove the grader handles those answers, not the ones the model + will write. When a real run fails a correct answer, add that answer's phrasing as a sample. - A case is a directory holding `prompt.md` or `case.yaml`. A directory holding neither is skipped silently, so a case whose prompt file is misnamed reads as absent rather than as an error. An eval dir with no case at all is a FAIL. diff --git a/plugins/evals/skills/validate/scripts/test_validate_cases.py b/plugins/evals/skills/validate/scripts/test_validate_cases.py index 0a457e78c8..94b3946687 100755 --- a/plugins/evals/skills/validate/scripts/test_validate_cases.py +++ b/plugins/evals/skills/validate/scripts/test_validate_cases.py @@ -30,7 +30,7 @@ # what keeps the fixtures discriminating: this pattern alone does not. FAIL_LINE = re.compile( r"^FAIL .*(unknown frontmatter key|duplicate grader|no grader|runs|not parsed" - r"|schema_version|without a prompt.md)", + r"|schema_version|without a prompt.md|sample)", re.MULTILINE, ) @@ -97,6 +97,15 @@ def case(self, name, prompt=None, case_yaml=None, graders=None): write(case_dir / "graders" / (grader_name + ".md"), body) return case_dir + def samples(self, case_name, grader_name, passing=(), failing=()): + path = self.eval_dir / case_name / "samples" / (grader_name + ".json") + path.parent.mkdir(parents=True, exist_ok=True) + data = { + "pass": [{"answer": answer} for answer in passing], + "fail": [{"answer": answer} for answer in failing], + } + path.write_text(json.dumps(data), encoding="utf-8") + def validate(self, *args): return run_validator(str(self.eval_dir), *args) @@ -113,6 +122,9 @@ def assert_clean(self, result): 0, result.returncode, "expected exit 0\n" + result.stdout + result.stderr ) self.assertNotIn("FAIL", result.stdout) + # A sample set that could not be graded is a WARN, which would otherwise + # let a broken mirror read as a clean pass. + self.assertNotIn("samples not checked", result.stdout) class UnknownKeyFixture(ValidatorTestCase): @@ -133,6 +145,33 @@ def test_unknown_prompt_frontmatter_key_fails(self): self.assert_fail(result, 'unknown frontmatter key "run_count"') self.assertIn("unknown-key/prompt.md", result.stdout) + def test_undocumented_keys_the_binary_accepts_warn_without_failing(self): + # Claude Code 2.1.287's case schema accepts both keys; the reference + # page's prompt.md field table lists neither. + self.case( + "undocumented-keys", + prompt="""\ + --- + description: Two keys that load but are not documented + artifact_publish: true + growthbook_overrides: {some_flag: true} + --- + + Do the thing. + """, + graders={"criteria": REGEX_GRADER}, + ) + self.samples("undocumented-keys", "criteria", ["conftest.py"], ["no"]) + result = self.validate() + self.assertEqual(0, result.returncode, result.stdout + result.stderr) + self.assertNotIn("FAIL", result.stdout) + for key in ("artifact_publish", "growthbook_overrides"): + self.assertIn( + 'WARN undocumented-keys/prompt.md: frontmatter key "%s" loads but ' + "is undocumented" % key, + result.stdout, + ) + class DuplicateGraderFixture(ValidatorTestCase): def test_duplicate_grader_name_across_yaml_and_dir_fails(self): @@ -738,6 +777,307 @@ def test_help_prints_the_module_header(self): self.assertIn("exit", result.stdout.lower()) +def regex_grader(pattern, extra=""): + return '---\ntype: regex\npattern: "%s"\n%s---\n' % (pattern, extra) + + +class SampleAnswers(ValidatorTestCase): + def test_a_must_pass_answer_the_regex_rejects_fails(self): + self.case("c", prompt=CLEAN_PROMPT, graders={"g": regex_grader("conftest")}) + self.samples("c", "g", passing=["use conftest.py", "a fixtures module"]) + result = self.validate() + self.assert_fail(result, "must-pass sample 2 fails the grader") + self.assertIn("c/samples/g.json", result.stdout) + self.assertNotIn("must-pass sample 1", result.stdout) + + def test_a_must_fail_answer_the_regex_accepts_fails(self): + self.case("c", prompt=CLEAN_PROMPT, graders={"g": regex_grader("conftest")}) + self.samples("c", "g", passing=["conftest.py"], failing=["a conftest file"]) + self.assert_fail(self.validate(), "must-fail sample 1 passes the grader") + + def test_samples_the_grader_sorts_correctly_are_clean(self): + self.case("c", prompt=CLEAN_PROMPT, graders={"g": regex_grader("conftest")}) + self.samples("c", "g", passing=["conftest.py"], failing=["", "I don't know."]) + result = self.validate() + self.assert_clean(result) + self.assertNotIn("no sample answers", result.stdout) + + def test_the_why_label_appears_in_the_finding(self): + self.case("c", prompt=CLEAN_PROMPT, graders={"g": regex_grader("conftest")}) + path = self.eval_dir / "c" / "samples" / "g.json" + path.parent.mkdir(parents=True) + path.write_text( + json.dumps({"pass": [{"answer": "nope", "why": "R2 phrasing"}]}), + encoding="utf-8", + ) + self.assert_fail(self.validate(), 'must-pass sample 1 ("R2 phrasing") fails') + + def test_flags_i_is_honored_and_no_flags_is_case_sensitive(self): + self.case( + "c", + prompt=CLEAN_PROMPT, + graders={ + "loose": regex_grader("conftest", "flags: i\n"), + "strict": regex_grader("conftest"), + }, + ) + self.samples("c", "loose", passing=["CONFTEST.PY"], failing=["pytest"]) + self.samples("c", "strict", passing=["conftest.py"], failing=["CONFTEST.PY"]) + self.assert_clean(self.validate()) + + def test_multiline_and_dotall_flags_change_the_match(self): + self.case( + "c", + prompt=CLEAN_PROMPT, + graders={ + "line": regex_grader("^two$", "flags: m\n"), + "span": regex_grader("one.two", "flags: s\n"), + }, + ) + self.samples("c", "line", passing=["one\ntwo\nthree"], failing=["one two"]) + self.samples("c", "span", passing=["one\ntwo"], failing=["one\n\ntwo"]) + self.assert_clean(self.validate()) + + def test_not_contains_and_count_modes(self): + self.case( + "c", + prompt=CLEAN_PROMPT, + graders={ + "absent": regex_grader("TODO", "match: not_contains\n"), + "twice": regex_grader("ok", 'match: "count:2"\n'), + }, + ) + self.samples("c", "absent", passing=["done"], failing=["a TODO left"]) + self.samples("c", "twice", passing=["ok, ok"], failing=["ok", "ok ok ok"]) + self.assert_clean(self.validate()) + + def test_a_javascript_named_group_is_translated(self): + self.case( + "c", + prompt=CLEAN_PROMPT, + graders={"g": regex_grader("(?ab)\\\\k")}, + ) + self.samples("c", "g", passing=["abab"], failing=["ab"]) + self.assert_clean(self.validate()) + + def test_tool_used_matches_the_compact_json_input_and_its_bounds(self): + self.case( + "c", + prompt=CLEAN_PROMPT, + graders={ + "fired": """\ + --- + type: tool_used + tool: Skill + input_match: '"skill":"(?:evals:)?methodology"' + --- + """, + "never": """\ + --- + type: tool_used + tool: Skill + min: 0 + max: 0 + arm: both + --- + """, + }, + ) + skill = {"tool": "Skill", "input": {"skill": "evals:methodology"}} + other = {"tool": "Skill", "input": {"skill": "evals:design"}} + read = {"tool": "Read", "input": {"file_path": "a.md"}} + self.samples( + "c", "fired", passing=[[skill], [read, skill]], failing=[[], [other]] + ) + self.samples("c", "never", passing=[[], [read]], failing=[[skill], [other]]) + self.assert_clean(self.validate()) + + def test_tool_order_compares_the_first_matching_calls(self): + self.case( + "c", + prompt=CLEAN_PROMPT, + graders={ + "order": """\ + --- + type: tool_order + before: Read + after: { tool: Skill, input_match: "design" } + --- + """ + }, + ) + read = {"tool": "Read", "input": {}} + design = {"tool": "Skill", "input": {"skill": "evals:design"}} + self.samples( + "c", + "order", + passing=[[read, design], [read, design, read]], + failing=[[design, read], [read], [design]], + ) + self.assert_clean(self.validate()) + + def test_file_exists_uses_the_anchored_glob(self): + self.case( + "c", + prompt="""\ + --- + allowed_tools: [Read, Write] + --- + + Write the report. + """, + graders={ + "made": """\ + --- + type: file_exists + path: "out/**/*.txt" + --- + """, + "absent": """\ + --- + type: file_exists + path: "*.log" + exists: false + --- + """, + }, + ) + self.samples( + "c", + "made", + passing=[["out/a.txt"], ["out/x/y/b.txt"]], + failing=[[], ["out/a.md"], ["src/out/a.txt"]], + ) + self.samples( + "c", "absent", passing=[[], ["logs/run.log"]], failing=[["run.log"]] + ) + self.assert_clean(self.validate()) + + def test_a_deterministic_grader_without_samples_warns(self): + self.case("c", prompt=CLEAN_PROMPT, graders={"g": REGEX_GRADER}) + result = self.validate() + self.assertEqual(0, result.returncode, result.stdout) + self.assertIn("WARN c/graders/g.md: no sample answers", result.stdout) + + def test_a_one_sided_sample_file_warns(self): + self.case("c", prompt=CLEAN_PROMPT, graders={"g": regex_grader("conftest")}) + self.samples("c", "g", passing=["conftest.py"]) + result = self.validate() + self.assertEqual(0, result.returncode, result.stdout) + self.assertIn("no must-fail answers", result.stdout) + + def test_judge_samples_warn_for_calibration_and_are_never_graded(self): + self.case( + "c", + prompt=CLEAN_PROMPT, + graders={ + "judge": "---\ntype: llm\n---\n\nPASS when the answer names conftest.py.\n", + "g": REGEX_GRADER, + }, + ) + # Samples a regex would sort the other way: a judge's are not graded. + self.samples("c", "judge", passing=[""], failing=["conftest.py"]) + result = self.validate() + self.assertEqual(0, result.returncode, result.stdout) + self.assertIn("2 sample answers need judge calibration", result.stdout) + self.assertNotIn("graders/judge.md: no sample answers", result.stdout) + + def test_samples_for_a_grader_in_case_yaml_are_found_by_name(self): + self.case( + "c", + case_yaml="""\ + schema_version: "1.1" + name: c + execution: + prompt: Do the thing. + graders: + - name: listed + type: regex + pattern: "thing" + """, + ) + self.samples("c", "listed", passing=["other"], failing=["nothing"]) + result = self.validate() + self.assert_fail( + result, "must-pass sample 1 fails", "must-fail sample 1 passes" + ) + + def test_malformed_samples_fail_closed(self): + self.case( + "c", + prompt=CLEAN_PROMPT, + graders={ + "a": regex_grader("x"), + "b": regex_grader("x"), + "d": regex_grader("x"), + }, + ) + directory = self.eval_dir / "c" / "samples" + directory.mkdir() + (directory / "a.json").write_text("{not json", encoding="utf-8") + (directory / "b.json").write_text('{"passes": []}', encoding="utf-8") + (directory / "d.json").write_text('{"pass": [{"answer": 3}]}', encoding="utf-8") + result = self.validate() + self.assert_fail( + result, + "c/samples/a.json: samples not parsed", + 'unknown key "passes"', + "answer must be a string for a regex grader", + ) + + def test_samples_for_no_grader_warn(self): + self.case("c", prompt=CLEAN_PROMPT, graders={"g": regex_grader("x")}) + self.samples("c", "g", passing=["x"], failing=["y"]) + self.samples("c", "renamed", passing=["x"], failing=["y"]) + result = self.validate() + self.assertEqual(0, result.returncode, result.stdout) + self.assertIn('no grader named "renamed"', result.stdout) + + def test_an_unmirrorable_flag_warns_instead_of_grading(self): + self.case( + "c", + prompt=CLEAN_PROMPT, + graders={"g": regex_grader("x", "flags: y\n")}, + ) + self.samples("c", "g", passing=["no match here", "x"], failing=["x"]) + result = self.validate() + self.assertEqual(0, result.returncode, result.stdout) + self.assertIn("samples not checked: flag y", result.stdout) + self.assertEqual(1, result.stdout.count("samples not checked")) + + +class R2Regression(ValidatorTestCase): + """The tracked samples catch the regex that scored R2's with-arm 0.""" + + R2_PATTERN = ( + "showpiece|fastest, most reliable, most scalable|headline (metric|number)" + "|read(ing)? (a )?samples?" + ) + + def test_the_r2_regex_fails_the_tracked_samples(self): + case_dir = self.eval_dir / "grading-method-choice" + shutil.copytree(PILOT_SUITE / "grading-method-choice", case_dir) + write( + case_dir / "graders" / "methodology-wording.md", + regex_grader(self.R2_PATTERN, "flags: i\narm: both\n"), + ) + result = self.validate() + # Samples 1 and 5 carry the two phrasings the R2 with-arm used. + self.assert_fail( + result, + "must-pass sample 1 (", + "must-pass sample 5 (", + "must-fail sample 7 (", + ) + + def test_the_tracked_regex_passes_the_tracked_samples(self): + shutil.copytree( + PILOT_SUITE / "grading-method-choice", + self.eval_dir / "grading-method-choice", + ) + self.assert_clean(self.validate()) + + class PilotSuite(unittest.TestCase): def test_tracked_pilot_suite_passes(self): self.assertTrue( @@ -751,6 +1091,9 @@ def test_tracked_pilot_suite_passes(self): "the live fixture must validate clean\n" + result.stdout + result.stderr, ) self.assertNotIn("FAIL", result.stdout) + # Every deterministic grader in the pilot suite carries checked samples. + self.assertNotIn("no sample answers", result.stdout) + self.assertNotIn("samples not checked", result.stdout) if __name__ == "__main__": diff --git a/plugins/evals/skills/validate/scripts/validate-cases.py b/plugins/evals/skills/validate/scripts/validate-cases.py index a9774e63c5..8898f1954d 100755 --- a/plugins/evals/skills/validate/scripts/validate-cases.py +++ b/plugins/evals/skills/validate/scripts/validate-cases.py @@ -9,6 +9,20 @@ FAIL what the binary itself rejects, so the suite cannot load WARN an authoring mistake the suite loads with and scores badly on +It also runs every deterministic grader (regex, tool_used, tool_order, +file_exists) offline against the case's sample answers, one JSON file per grader +at `/samples/.json`: + + {"pass": [{"answer": ..., "why": "..."}], "fail": [{"answer": ...}]} + +`answer` is what the grader reads: the target text for regex (the final reply +by default), a list of `{"tool": ..., "input": {...}}` calls for tool_used and +tool_order, and a list of created paths for file_exists. A must-pass answer the +grader rejects, or a must-fail answer it accepts, is a FAIL. A deterministic +grader with no samples is a WARN, and so is an llm or baseline grader that has +samples, since judging them needs a paid calibration run this script never +makes. + Output is one finding per line, ` /: `, sorted by case then file. A finding about the suite as a whole carries the eval dir in place of `/`. `--json` prints the same findings as a JSON array of @@ -38,6 +52,26 @@ rather than patching one value. AS OF: 2026-09-12. Page: https://code.claude.com/docs/en/plugin-evals +Drift record. CLAIM: two more `prompt.md` keys, `artifact_publish` and +`growthbook_overrides`, load but are not on the page's field table, so they are +a WARN rather than a FAIL. BASIS: the case schema read out of Claude Code +2.1.287 against the page's prompt.md field table. AS OF: 2026-10-02. Recheck +when the page lists either key, or a release drops either from the schema. + +Drift record. CLAIM: the sample check grades the way the binary does - a regex +is a JavaScript RegExp built from `pattern` and `flags` (default none) and +tested against the target; `match` defaults to contains, and `count:N` counts +global matches; `input_match` is a flagless regex over each call's compact +JSON input; `min` defaults to 1 and `max` to unbounded; `tool_order` compares +the first matching call of each side; a `file_exists` glob is anchored, with +`*` inside one path segment and `**` across them. Python's `re` stands in for +the JavaScript engine: named groups are translated, the `y` and `v` flags and +any pattern `re` cannot compile are reported as not checkable (WARN), and the +two engines still differ at the edges (`\\w` and `\\b` on non-ASCII letters, +`$` before a final newline). BASIS: the grader code in Claude Code 2.1.287 +read together with the grader-types table of the eval-suite reference page. +AS OF: 2026-10-01. Recheck on the next Claude Code release. + Python 3.8+, standard library only: these scripts run from a consumer checkout where no third-party package is provisioned. """ @@ -70,6 +104,11 @@ ] ) +# Keys the binary accepts in `prompt.md` that the reference page does not list. +# They load, so they are no FAIL; undocumented, they can change without notice, +# which is a WARN. +UNDOCUMENTED_PROMPT_KEYS = frozenset(["artifact_publish", "growthbook_overrides"]) + # `case.yaml` nests the run fields under `execution`, where `prompt.md` keeps # them flat. Both spellings feed the same bounds check below. EXECUTION_KEYS = ("max_turns", "timeout_seconds", "allowed_tools", "env") @@ -138,6 +177,23 @@ # tool calls. Anything else inside a case directory belongs to that case. SKIP_DIRS = frozenset(["results", "mocks", "graders", "node_modules", "__pycache__"]) +# Sample answers live beside graders/, never inside it, so the binary never +# loads one as a grader, and a run cannot read the eval directory, so the agent +# never sees them. +SAMPLES_DIR = "samples" +SAMPLE_KEYS = ("pass", "fail") +DETERMINISTIC_TYPES = frozenset(["regex", "tool_used", "tool_order", "file_exists"]) + +# JavaScript RegExp flags that change matching, and their `re` equivalents. +# `g` and `d` do not change whether a fresh regex matches; `u` is the default +# for a Python str pattern. +JS_FLAGS = {"i": re.IGNORECASE, "m": re.MULTILINE, "s": re.DOTALL} +NEUTRAL_FLAGS = frozenset("gdu") +COUNT_MATCH = re.compile(r"^count:([0-9]+)$") +JS_NAMED_GROUP = re.compile(r"\(\?<(?![=!])") +JS_NAMED_BACKREF = re.compile(r"\\k<([A-Za-z_][A-Za-z0-9_]*)>") +GLOB_SPECIAL = frozenset(".+^${}()|[]\\") + class ParseError(Exception): """A construct outside the supported YAML subset.""" @@ -147,6 +203,10 @@ def __init__(self, construct): self.construct = construct +class Unmirrored(Exception): + """A grader setting this script cannot reproduce offline.""" + + # -------------------------------------------------------------------------- # YAML subset parser # -------------------------------------------------------------------------- @@ -696,6 +756,291 @@ def collect_graders(case_dir, case, yaml_data, findings): return graders +# -------------------------------------------------------------------------- +# Sample answers: deterministic graders run offline +# -------------------------------------------------------------------------- + + +def js_regex(pattern, flags=""): + """Compile a JavaScript regex with Python's re, or raise Unmirrored.""" + if not isinstance(pattern, str) or not isinstance(flags, str): + raise Unmirrored("pattern and flags must be strings") + compiled_flags = 0 + for flag in flags: + if flag in JS_FLAGS: + compiled_flags |= JS_FLAGS[flag] + elif flag not in NEUTRAL_FLAGS: + raise Unmirrored("flag %s has no offline equivalent" % flag) + source = JS_NAMED_GROUP.sub("(?P<", pattern) + source = JS_NAMED_BACKREF.sub(r"(?P=\1)", source) + try: + return re.compile(source, compiled_flags) + except re.error as error: + raise Unmirrored("pattern does not compile offline (%s)" % error) + + +def grade_regex(options, text): + regex = js_regex(options.get("pattern"), options.get("flags") or "") + mode = options.get("match", "contains") + if mode in ("contains", "not_contains"): + found = regex.search(text) is not None + if mode == "contains": + return found, "pattern found" if found else "pattern not found" + return not found, "pattern found, expected absent" if found else "absent" + count = COUNT_MATCH.match(str(mode)) + if not count: + raise Unmirrored('match mode "%s"' % mode) + found = sum(1 for _ in regex.finditer(text)) + return ( + found == int(count.group(1)), + "found %d matches, expected %s" % (found, count.group(1)), + ) + + +def call_matches(call, spec): + """One call against a tool name or a {tool, input_match} mapping.""" + if isinstance(spec, str): + spec = {"tool": spec} + if not isinstance(spec, dict): + raise Unmirrored("a tool spec must be a name or a mapping") + if call["tool"] != spec.get("tool"): + return False + if not spec.get("input_match"): + return True + # The binary matches the compact JSON encoding of the call's input. + text = json.dumps(call.get("input", {}), separators=(",", ":"), ensure_ascii=False) + return js_regex(spec["input_match"]).search(text) is not None + + +def grade_tool_used(options, calls): + spec = {"tool": options.get("tool"), "input_match": options.get("input_match")} + count = sum(1 for call in calls if call_matches(call, spec)) + low = options.get("min") + low = 1 if low is None else low + high = options.get("max") + for bound in (low, high): + if bound is not None and ( + not isinstance(bound, int) or isinstance(bound, bool) + ): + raise Unmirrored("min and max must be whole numbers") + passed = count >= low and (high is None or count <= high) + return passed, "%s called %dx, expected %s..%s" % ( + spec["tool"], + count, + low, + "unbounded" if high is None else high, + ) + + +def grade_tool_order(options, calls): + def first(spec): + for index, call in enumerate(calls): + if call_matches(call, spec): + return index + return -1 + + before, after = first(options.get("before")), first(options.get("after")) + if before == -1 or after == -1: + return False, '"%s" tool never called' % ("before" if before == -1 else "after") + return before < after, "before@%d, after@%d" % (before, after) + + +def glob_regex(glob): + """The binary's glob: anchored, `*` within a segment, `**/` any depth.""" + if not isinstance(glob, str): + raise Unmirrored("path must be a string") + out, index = "^", 0 + while index < len(glob): + char = glob[index] + if glob.startswith("**/", index): + out, index = out + "(?:.*/)?", index + 3 + continue + if glob.startswith("**", index): + out, index = out + ".*", index + 2 + continue + if char == "*": + out += "[^/]*" + elif char == "?": + out += "." + else: + out += "\\" + char if char in GLOB_SPECIAL else char + index += 1 + return re.compile(out + "$") + + +def grade_file_exists(options, paths): + regex = glob_regex(options.get("path")) + found = any(regex.search(path.strip()) for path in paths if path.strip()) + expected = options.get("exists", True) + if not isinstance(expected, bool): + raise Unmirrored("exists must be true or false") + return found == expected, "%s %s" % ( + options.get("path"), + "created" if found else "not created", + ) + + +GRADE = { + "regex": grade_regex, + "tool_used": grade_tool_used, + "tool_order": grade_tool_order, + "file_exists": grade_file_exists, +} + + +def answer_error(kind, answer): + """What `answer` must be for this grader type, or None when it fits.""" + if kind == "regex": + return None if isinstance(answer, str) else "a string" + if kind == "file_exists": + fits = isinstance(answer, list) and all(isinstance(p, str) for p in answer) + return None if fits else "a list of created paths" + fits = isinstance(answer, list) and all( + isinstance(call, dict) + and isinstance(call.get("tool"), str) + and set(call) <= {"tool", "input"} + for call in answer + ) + return None if fits else 'a list of {"tool", "input"} calls' + + +def sample_shape_error(data): + if not isinstance(data, dict): + return "the file must hold a JSON object" + for key in sorted(data): + if key not in SAMPLE_KEYS: + return 'unknown key "%s"; use pass and fail' % key + if not isinstance(data[key], list): + return "%s must be a list" % key + for number, item in enumerate(data[key], start=1): + if ( + not isinstance(item, dict) + or "answer" not in item + or not set(item) <= {"answer", "why"} + ): + return '%s entry %d must be {"answer": ..., "why": ...}' % (key, number) + return None + + +def load_samples(case_dir, case, findings): + """Every samples/.json in the case, as {name: (path, data)}.""" + directory = os.path.join(case_dir, SAMPLES_DIR) + loaded = {} + if not os.path.isdir(directory): + return loaded + for filename in sorted(os.listdir(directory)): + if not filename.endswith(".json"): + continue + path = SAMPLES_DIR + "/" + filename + try: + data = json.loads(read_text(os.path.join(directory, filename))) + problem = sample_shape_error(data) + except (OSError, ValueError) as error: + problem = str(error) + if problem: + findings.append( + Finding("FAIL", case, path, "samples not parsed (%s)" % problem) + ) + continue + loaded[filename[:-5]] = (path, data) + return loaded + + +def run_samples(case, path, kind, options, data, findings): + """Grade every sample; a mismatch with its expected verdict is a FAIL.""" + for key in SAMPLE_KEYS: + if not data.get(key): + findings.append( + Finding( + "WARN", + case, + path, + "no must-%s answers, so the check cannot catch a grader that " + "%s everything" % (key, "rejects" if key == "pass" else "accepts"), + ) + ) + for key in SAMPLE_KEYS: + for number, item in enumerate(data.get(key) or [], start=1): + label = "must-%s sample %d" % (key, number) + if item.get("why"): + label += ' ("%s")' % item["why"] + problem = answer_error(kind, item["answer"]) + if problem: + findings.append( + Finding( + "FAIL", + case, + path, + "%s: answer must be %s for a %s grader" + % (label, problem, kind), + ) + ) + continue + try: + passed, explanation = GRADE[kind](options, item["answer"]) + except Unmirrored as error: + findings.append( + Finding("WARN", case, path, "samples not checked: %s" % error) + ) + return + if passed != (key == "pass"): + findings.append( + Finding( + "FAIL", + case, + path, + "%s %s the grader (%s)" + % (label, "fails" if key == "pass" else "passes", explanation), + ) + ) + + +def check_samples(case, graders, samples, findings): + """Run deterministic graders over their samples; flag what goes unchecked.""" + names = set() + for name, options, grader_path in graders: + names.add(name) + if not isinstance(options, dict) or options.get("type") not in GRADER_TYPES: + continue + kind = options["type"] + if name not in samples: + if kind in DETERMINISTIC_TYPES: + findings.append( + Finding( + "WARN", + case, + grader_path, + "no sample answers; add %s/%s.json with answers this " + "grader must pass and must reject" % (SAMPLES_DIR, name), + ) + ) + continue + path, data = samples[name] + if kind in JUDGE_TYPES: + count = sum(len(data.get(key) or []) for key in SAMPLE_KEYS) + findings.append( + Finding( + "WARN", + case, + path, + "%d sample answers need judge calibration; this check never " + "calls a model" % count, + ) + ) + continue + run_samples(case, path, kind, options, data, findings) + for name in sorted(set(samples) - names): + findings.append( + Finding( + "WARN", + case, + samples[name][0], + 'no grader named "%s" in this case, so these samples check nothing' + % name, + ) + ) + + def analyze_case(case_dir, case, findings): """Validate one case, appending its findings.""" fields = {} @@ -757,7 +1102,17 @@ def analyze_case(case_dir, case, findings): prompt_fm = None if prompt_fm is not None: for key in sorted(prompt_fm): - if key not in PROMPT_KEYS: + if key in UNDOCUMENTED_PROMPT_KEYS: + findings.append( + Finding( + "WARN", + case, + "prompt.md", + 'frontmatter key "%s" loads but is undocumented, so ' + "it can change without notice" % key, + ) + ) + elif key not in PROMPT_KEYS: findings.append( Finding( "FAIL", @@ -823,6 +1178,8 @@ def analyze_case(case_dir, case, findings): ) ) + check_samples(case, graders, load_samples(case_dir, case, findings), findings) + if kinds and all(kind in JUDGE_TYPES for kind in kinds): findings.append( Finding( diff --git a/scripts/affected-tests-no-suite.txt b/scripts/affected-tests-no-suite.txt index 2308904a9f..05a56ff031 100644 --- a/scripts/affected-tests-no-suite.txt +++ b/scripts/affected-tests-no-suite.txt @@ -135,6 +135,13 @@ LICENSE # still selected by the reference rule before this line is consulted. plugins/*/evals/fixtures/*.txt +# Kept-trace fixtures for the evals plugin's result checkers. test_run_validity.py +# and test_calibrate_judge.py read them through their fixture directories, never +# by basename, so no rule above selects a suite. The lane covering them is the +# two wrapping shell suites, run-validity.test.sh and calibrate-judge.test.sh, +# which run those tests whenever any script in that directory changes. +plugins/evals/skills/plugin-eval/scripts/fixtures/*.jsonl + # The task-end judge's calibration cases: test and production code copied # verbatim from other repositories' histories, read as data by the judge, never # run. The lane covering them is the testing plugin's shell lane: metrics.test.sh From 931672e7d4a848932fe8204be1363a0b4ea0e3cc Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Fri, 2 Oct 2026 02:17:55 -0400 Subject: [PATCH 03/11] feat(discovery): accept single-publisher content claims with a visible flag Contract change. Research gate rows 4 and 7 let a first-party content claim (what a named Anthropic page, file or changelog says) pass with a "single source" flag when the claim states why only one publisher exists. Reposts never count as a second source, behavior claims still need two corroborators, and an unconvincing reason fails row 4 as before. The researcher, the verifier, the parent contract, the artifact shape and the evals carry the flag. The recommendation-basis convention (1.1.0, additive) lets a flagged claim ground a code edit as `verified (single source)`, with the flag carried into the record beside the edit. Co-Authored-By: Claude Opus 5.5 --- .../recommendation-basis/CHANGELOG.md | 9 +++ .../recommendation-basis/README.md | 6 ++ plugins/discovery/.claude-plugin/plugin.json | 2 +- plugins/discovery/CHANGELOG.md | 23 +++++++ plugins/discovery/agents/research-verifier.md | 6 ++ plugins/discovery/agents/researcher.md | 10 ++- .../discovery/reference/parent-contract.md | 2 +- plugins/discovery/scripts/contract.test.sh | 62 +++++++++++++++++++ .../discovery/skills/research-deep/SKILL.md | 2 +- plugins/discovery/skills/research/SKILL.md | 24 +++---- .../skills/research/context/artifact-shape.md | 14 ++++- .../skills/research/context/discipline.md | 20 +++++- .../skills/research/context/gotchas.md | 6 ++ .../skills/research/evals/evals.json | 29 +++++++++ 14 files changed, 191 insertions(+), 24 deletions(-) diff --git a/docs/conventions/recommendation-basis/CHANGELOG.md b/docs/conventions/recommendation-basis/CHANGELOG.md index 43d63714f3..c4be1c0cd7 100644 --- a/docs/conventions/recommendation-basis/CHANGELOG.md +++ b/docs/conventions/recommendation-basis/CHANGELOG.md @@ -4,6 +4,15 @@ Notable changes to the recommendation-basis contract (SemVer). Changing the grou label's values, or the re-emit shape is a major bump; additive guidance is a minor bump; docs-only clarification is a patch. +## [1.1.0] - 2026-10-02 + +Minor, additive. The Basis label section adds the `single source` qualifier: a recommendation +resting on a first-party content claim that `/discovery:research` accepted with a `single source` +flag stays `verified` and may ground a code edit, and its label and the record beside the edit +carry the flag, `Basis: verified (single source), `. The label's two values, the grounding bar +and the re-emit shape are unchanged, so earlier adopters still conform. The plugin-shipped copies +leave the qualifier out because it applies only to research output. + ## [1.0.2] - 2026-10-01 Patch, docs-only. The Boundary bullet on durable records of upstream-derived facts names the record diff --git a/docs/conventions/recommendation-basis/README.md b/docs/conventions/recommendation-basis/README.md index 067c8fc910..95f1f0def5 100644 --- a/docs/conventions/recommendation-basis/README.md +++ b/docs/conventions/recommendation-basis/README.md @@ -77,6 +77,12 @@ visible `Basis:` label; the third is not presented: Example: `Basis: verified, .github/workflows/ci.yml:42 and https://docs.github.com/... (fetched this session)`, or `Basis: judgment`. +**A claim accepted with a `single source` flag keeps it.** `/discovery:research` accepts a +first-party content claim that can have only one publisher, flagged `single source`, when the claim +states why no second publisher exists. A recommendation resting on that claim is still verified +and may ground a code edit, but its label carries the flag, +`Basis: verified (single source), `, and so does the record written beside the edit. + ## Re-emitting a changed recommendation When evidence changes a recommendation the user still has pending, restate it as **old → new → diff --git a/plugins/discovery/.claude-plugin/plugin.json b/plugins/discovery/.claude-plugin/plugin.json index 74c58e4690..8fb8c27753 100644 --- a/plugins/discovery/.claude-plugin/plugin.json +++ b/plugins/discovery/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "discovery", - "version": "0.25.26", + "version": "0.26.0", "description": "Structured discovery before changes: explore the local codebase, run disciplined multi-source external research, and reconstruct why a past decision was made from evidence outside the code. Each dispatches a purpose-built subagent by default so the reading stays out of the main conversation, with source tiers, falsification, recency gates, an intent-evidence tier, and a corpus-coverage ledger, and each persists EXPLORE.md / RESEARCH.md / INTENT.md index-plus-sidecar handoff artifacts.", "author": { "name": "Melodic Software", diff --git a/plugins/discovery/CHANGELOG.md b/plugins/discovery/CHANGELOG.md index 252ec25f02..bfd28c01bc 100644 --- a/plugins/discovery/CHANGELOG.md +++ b/plugins/discovery/CHANGELOG.md @@ -1,5 +1,28 @@ # Changelog: discovery plugin +## [0.26.0] - 2026-10-02 + +### Changed + +- **Contract change: the `/discovery:research` outcome gate accepts a single-publisher claim, + flagged.** A first-party content claim, one that states what a named Anthropic page, file or + changelog says, passes rows 4 and 7 at the new confidence level `HIGH (single source)` when it + states why only one publisher exists. A repost of the page is never a second source, and a + behavior claim still needs two independent corroborators. The verifier grades the stated reason, + and a reason that does not hold fails row 4. The flag stays visible in the evidence table, the + answer and any synthesis, and a flagged claim may ground a code edit with the flag carried into + the record beside it. +- **The sidecar header gains a per-claim `single_source:` field**, and its `confidence:` vocabulary + gains `HIGH (single source)`. The definition lives in `context/discipline.md`, "Single-source + first-party content claims"; the verifier, the researcher, the parent contract's `pass` value, + the gotchas and `research-deep`'s synthesis rule point at it. +- Two research evals grade the new branch: a flagged changelog claim with two reposts passes, and a + behavior claim flagged from one docs page is a Gap. +- `research/SKILL.md` moves its `breadth=` paragraph below the Effort table so the outcome gate's + longer rows 4 and 7 keep the effort ceiling inside the compaction re-attach slice. +- Outcome-gate row 7 keeps the Gaps rule: a MEDIUM or LOW claim listed in the Gaps section is not + an accepted claim, and `HIGH (single source)` under row 4's flag is. + ## [0.25.26] - 2026-10-01 ### Changed diff --git a/plugins/discovery/agents/research-verifier.md b/plugins/discovery/agents/research-verifier.md index ba6a81ea9f..97c42211dc 100644 --- a/plugins/discovery/agents/research-verifier.md +++ b/plugins/discovery/agents/research-verifier.md @@ -40,6 +40,12 @@ MEDIUM or LOW and listed in the Gaps section is not an accepted claim, so it doe A quote found at its link settles only that the quote exists; it does not show the claim follows from it, which is the question row 12 asks. +A claim at `HIGH (single source)` has no corroborator to count, so row 4 turns on its +`single_source:` reason. Judge that reason against the definition in +[`${CLAUDE_PLUGIN_ROOT}/skills/research/context/discipline.md`](${CLAUDE_PLUGIN_ROOT}/skills/research/context/discipline.md), +"Single-source first-party content claims". A reason that does not hold, a behavior claim carrying +the flag, or a repost counted as a source fails row 4. + Fetch each page once, and read each file once; the rule is stated once in [`${CLAUDE_PLUGIN_ROOT}/reference/parent-contract.md`](${CLAUDE_PLUGIN_ROOT}/reference/parent-contract.md) ("Read each file once, stated once"). Your limit is `maxTurns: 30`, from this definition's diff --git a/plugins/discovery/agents/researcher.md b/plugins/discovery/agents/researcher.md index c55301aeaf..0516911f6c 100644 --- a/plugins/discovery/agents/researcher.md +++ b/plugins/discovery/agents/researcher.md @@ -267,15 +267,19 @@ Run the skill's outcome gate against your own artifacts before the final write. **not yours to render a verdict on**, because grading them means judging the quality of your own choices, and you are the context that made them: -- the criterion requiring ≥2 **independent** corroborators per claim, -- the criterion requiring every accepted claim to be HIGH confidence, and +- the criterion requiring ≥2 **independent** corroborators per claim, or a `single source` reason + that holds for a first-party content claim, +- the criterion requiring every accepted claim to be HIGH confidence, `HIGH (single source)` + included, and - the criterion requiring every accepted claim to follow jointly from its cited sources. The gate's Owner column is the authority; where this list and that column differ, the column wins. Assemble the evidence those criteria need, since per-claim source URLs with their tier, publishing pool, what each measured, when it was published, and which product versions it applies to go in the sidecar headers, which is what lets a verifier who never saw your run grade them off -the artifact, then hand them back as a verification request. Project fit against the consuming +the artifact, then hand them back as a verification request. A claim you flag `single source` +carries its reason in the header's `single_source:` field, because that reason is what the +verifier grades in place of a corroborator count. Project fit against the consuming project's conventions is the parent's; it alone holds them. Every other criterion is yours, and the coverage ledger's and source applicability's verdicts are their scripts' exit statuses, not your reading of the table or the headers. diff --git a/plugins/discovery/reference/parent-contract.md b/plugins/discovery/reference/parent-contract.md index 47c00e4741..c6686eaefd 100644 --- a/plugins/discovery/reference/parent-contract.md +++ b/plugins/discovery/reference/parent-contract.md @@ -877,7 +877,7 @@ write `general-purpose` as the worker. | Value | State | Meaning | |---|---|---| -| `pass (research-verifier, )` | `pass` | The verifier passed every criterion it was briefed on. | +| `pass (research-verifier, )` | `pass` | The verifier passed every criterion it was briefed on. A pass keeps each claim's `single source` flag: the parent presents the flag with the claim and carries it into any record an edit rests on. | | `fail rows [,…] (research-verifier, )` | `fail` | The verifier failed those rows, named as it returned them. `explore` and `trace-intent` have no rows: they write `fail (general-purpose, )` and the failed claims stay in the verifier's return. | | `skipped (cost)` | outside the shape | Research only. The parent chose not to pay for a verifier: no worker, no date. | | `unverified (none, )` | `unverified` | No verifier could be dispatched, in all three families. The index carries a numbered gap. | diff --git a/plugins/discovery/scripts/contract.test.sh b/plugins/discovery/scripts/contract.test.sh index 000060d016..d50aa9c961 100755 --- a/plugins/discovery/scripts/contract.test.sh +++ b/plugins/discovery/scripts/contract.test.sh @@ -1042,6 +1042,68 @@ for agent in explorer researcher intent-tracer research-verifier; do fi done +# --------------------------------------------------------------------------- +# 19. A first-party content claim with one possible publisher passes flagged +# +# A claim about what a named Anthropic page, file or changelog says has no +# second publisher to find. It passes rows 4 and 7 at `HIGH (single source)` +# when it states why only one publisher exists; a repost is never a second +# source, and a behavior claim still needs two corroborators. The flag travels +# with the claim into the answer and into any record an edit rests on. +# --------------------------------------------------------------------------- +single_heading='## Single-source first-party content claims' +assert_present 'gate row 4 has a single source branch that states why only one publisher exists' \ + 'skills/research/SKILL.md' '^\| 4 \|.*flagged `single source`.*why only one publisher exists' +assert_present 'gate row 4 says a repost is not a second source' \ + 'skills/research/SKILL.md' '^\| 4 \|.*a repost is not a second source' +assert_present 'gate row 4 keeps behavior claims at two corroborators' \ + 'skills/research/SKILL.md' '^\| 4 \|.*a behavior claim' +assert_present 'gate row 7 accepts HIGH (single source)' \ + 'skills/research/SKILL.md' '^\| 7 \|.*`HIGH \(single source\)`' +assert_present 'discipline 5 points at the single-source exception' \ + 'skills/research/SKILL.md' "^5\. .*\"Single-source first-party content claims\"" +assert_present 'discipline.md carries the single-source section' \ + 'skills/research/context/discipline.md' "^$single_heading$" +assert_present 'discipline.md defines the HIGH (single source) level' \ + 'skills/research/context/discipline.md' '^- \*\*HIGH \(single source\)\*\*' +assert_present 'discipline.md lets a flagged claim ground a code edit with the flag carried' \ + 'skills/research/context/discipline.md' 'may ground a code edit' +assert_present 'discipline.md keeps behavior claims at two corroborators' \ + 'skills/research/context/discipline.md' '^\*\*A behavior claim is not a content claim\.\*\*' +assert_present 'the corroboration rule points at the exception' \ + 'skills/research/context/discipline.md' '^\*\*Authoritative is not a waiver for corroboration\.\*\*.*Single-source first-party content claims' +assert_present 'the sidecar header carries the single_source reason' \ + 'skills/research/context/artifact-shape.md' '^ {4}single_source: ' +assert_present 'the sidecar confidence vocabulary lists HIGH (single source)' \ + 'skills/research/context/artifact-shape.md' 'confidence: HIGH +# HIGH \| HIGH \(single source\) \| MEDIUM \| LOW' +assert_present 'the verifier grades the stated single_source reason' \ + 'agents/research-verifier.md' '`single_source:` reason' +assert_present 'the researcher records the reason in the header' \ + 'agents/researcher.md' '`single_source:`' +assert_present 'a research pass keeps the single source flag' \ + 'reference/parent-contract.md' '^\| `pass \(research-verifier, \)` \|.*`single source` flag' +assert_present 'gotchas name the repost trap' \ + 'skills/research/context/gotchas.md' '^- \*\*Counting a repost as the second source\.\*\*' +assert_present 'research-deep keeps the flag through the synthesis' \ + 'skills/research-deep/SKILL.md' 'keeps its `single source` flag' +for name in single-source-content-claim-passes-flagged behavior-claim-from-one-page-is-a-gap; do + assert_present "research evals grade $name" 'skills/research/evals/evals.json' "\"name\": \"$name\"" +done +single_owners="$(surface | xargs grep -lE -- "^$single_heading$" 2>/dev/null | wc -l | tr -d ' ')" +if [[ "$single_owners" -eq 1 ]]; then + pass 'the single-source definition has exactly one owner' +else + fail "the single-source definition has exactly one owner — $single_owners files carry the heading" +fi +recbasis="$PLUGIN_ROOT/../../docs/conventions/recommendation-basis/README.md" +if [[ ! -f "$recbasis" ]]; then + pass 'recommendation-basis carry-forward not checked outside the monorepo' +elif grep -qE 'Basis: verified \(single source\)' "$recbasis"; then + pass 'recommendation-basis carries the single source flag into the Basis label' +else + fail 'recommendation-basis carries the single source flag into the Basis label' +fi + printf '\n' if [[ "$fails" -eq 0 ]]; then printf 'All contract assertions passed.\n' diff --git a/plugins/discovery/skills/research-deep/SKILL.md b/plugins/discovery/skills/research-deep/SKILL.md index 7bcd9c1b72..5f938c31ec 100644 --- a/plugins/discovery/skills/research-deep/SKILL.md +++ b/plugins/discovery/skills/research-deep/SKILL.md @@ -96,7 +96,7 @@ Invoke `/discovery:research` via the Skill tool, inline in this session. No disp **A dispatched run is not finished when it returns.** No producing context, whether engine, isolated subagent, or topic worker, can complete the `/discovery:research` outcome gate's verifier-owned rows (independent corroboration, HIGH confidence, joint inference) or its parent-owned row (project fit). The verifier rows are assigned to a fresh context precisely because a producer may not grade its own choices; project fit needs the consuming project's conventions, which only this session holds. Nor can the producer be relied on to dispatch that verifier itself. Whether a non-fork subagent holds `Agent` depends on the harness's nesting allowance (`CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH`), a session property this skill does not design against. -So for **every** dispatched run, one per topic on the N-topic path, once on Tier 1 and Tier 2, this session dispatches the sibling verifier against the artifact on disk, applies project fit, and writes both results back into that artifact's index **before** surfacing anything. Surfacing a producer's summary and artifact path directly presents claims as gate-passed when the rows that matter were never graded by anyone. A single-topic ask earns no weaker boundary than a multi-topic one, and an engine earns no weaker boundary than a subagent. The verifier is `discovery:research-verifier`; its dispatch, the `verification:` write-back and the `skipped (cost)` path are the verifier block in [`${CLAUDE_PLUGIN_ROOT}/skills/research/SKILL.md`](${CLAUDE_PLUGIN_ROOT}/skills/research/SKILL.md). On the N-topic path the synthesized root index also goes to a fresh verifier for criterion 12 before it is surfaced, per the research dispatch contract's fan-out section. +So for **every** dispatched run, one per topic on the N-topic path, once on Tier 1 and Tier 2, this session dispatches the sibling verifier against the artifact on disk, applies project fit, and writes both results back into that artifact's index **before** surfacing anything. Surfacing a producer's summary and artifact path directly presents claims as gate-passed when the rows that matter were never graded by anyone. A single-topic ask earns no weaker boundary than a multi-topic one, and an engine earns no weaker boundary than a subagent. The verifier is `discovery:research-verifier`; its dispatch, the `verification:` write-back and the `skipped (cost)` path are the verifier block in [`${CLAUDE_PLUGIN_ROOT}/skills/research/SKILL.md`](${CLAUDE_PLUGIN_ROOT}/skills/research/SKILL.md). On the N-topic path the synthesized root index also goes to a fresh verifier for criterion 12 before it is surfaced, per the research dispatch contract's fan-out section. A claim a topic index flags keeps its `single source` flag in the synthesis and in anything surfaced from it. **Grade the run off disk before any of that.** Every obligation above acts on an artifact, so all of them are worthless against a dispatch that produced none, and `status: complete` is the producer's claim about its own run. The parent skill's **post-dispatch acceptance gate** is what turns that claim into evidence: create the slice and touch a `.research-dispatch` baseline BEFORE the dispatch. Both shell forms of that one command are in [`${CLAUDE_PLUGIN_ROOT}/reference/parent-contract.md`](${CLAUDE_PLUGIN_ROOT}/reference/parent-contract.md), and the POSIX one does not run in PowerShell, then `scripts/check-dispatch-artifact.sh --index-name RESEARCH.md` against the slice path this session resolved (never one read out of the payload), then a parent-side regrade of the coverage ledger and of source applicability (`${CLAUDE_PLUGIN_ROOT}/scripts/check-source-applicability.py` with `--expect-evidence-use` set to the envelope's value; a Tier 1 engine artifact without the header fields fails it by design, so route that topic to Tier 2). Cite exit statuses; any non-zero halts. **On the N-topic path run it against the sub-slice assigned to each topic, before synthesizing the slice-root index**, the gate grades exactly the path it is handed and never scans, so a sub-slice invocation grades that topic's run while a slice-root invocation would grade only the synthesized index, never any dispatched run. **That one baseline at the slice root serves every sub-slice**, the gate compares each sub-slice index's mtime against the file it is handed, and a baseline touched now is newer than anything an earlier run left anywhere under the slice, so a per-sub-slice baseline is optional, not owed. diff --git a/plugins/discovery/skills/research/SKILL.md b/plugins/discovery/skills/research/SKILL.md index 2a93d7c26c..f8177b0781 100644 --- a/plugins/discovery/skills/research/SKILL.md +++ b/plugins/discovery/skills/research/SKILL.md @@ -72,10 +72,10 @@ Each criterion is binary. **Any FAIL returns to the named phase (bounded at `Bud | 1 | Every claim row has ≥1 Tier 0/1 source whose URL/command was captured THIS turn | run | Phase 2. Fetch the primary directly | | 2 | No claim row's sources are ALL Tier-2 secondary | run | Phase 2. Get a primary | | 3 | Every Phase 2/3 query traces to a numbered gap/conflict in a written analysis block | run | re-run the phase chained to the list | -| 4 | Every claim has ≥2 INDEPENDENT `current` corroborators (not 2 cites of one upstream pool; a `historical` source never counts) | **verifier** | Phase 2. Widen sources | +| 4 | Every claim has ≥2 INDEPENDENT `current` corroborators (not 2 cites of one upstream pool; a `historical` source never counts), or is a first-party content claim flagged `single source` that states why only one publisher exists; a repost is not a second source, and a behavior claim gets no flag | **verifier** | Phase 2. Widen sources | | 5 | The Phase 2 falsification query ran and is recorded | run | Phase 2. Run it | | 6 | Recency gate satisfied for every tool/library/API claim: the LATEST upstream changelog/release was fetched THIS turn and cross-checked against the claim. Read the confirmed-latest release and the verdict off the fetch log's changelog entry, an absent verdict or an `invalidated` one FAILs, and `unresolved` passes only as an enumerated Gap, never under an accepted claim. Windows, and what a major bump invalidates: the discipline file's "Recency gate" | run | Phase 2. Fetch changelog | -| 7 | Every accepted claim is HIGH confidence; a MEDIUM or LOW claim listed in the Gaps section is not accepted | **verifier** | Phase 4 follow-up. Iterate to HIGH or list as a Gap | +| 7 | Every accepted claim is HIGH confidence, or `HIGH (single source)` under row 4's flag; a MEDIUM or LOW claim listed in the Gaps section is not accepted | **verifier** | Phase 4 follow-up. Iterate to HIGH or list as a Gap | | 8 | Project fit checked against the consuming project's own conventions and stated direction | **parent** | revisit before presenting | | 9 | For every ACCEPTED claim taken from any publisher's own artifacts, vendor, OSS maintainer, standards body alike, the fetch log ACCOUNTS FOR every artifact-ladder rung above the one the claim came from, each carrying one of the outcome values and none left unaccounted. Rungs, outcome vocabulary, and what earns nonexistence rather than `unresolved`: the discipline file's "Primary-source-first protocol". A rung that exists, is reachable, and carries the claim IS where the claim comes from | run | Phase 2. Walk the ladder from rung 1, fetching and searching each reachable rung and recording its outcome | | 10 | Every reported absence names both the sources checked and the sources left unchecked. No bare "unsourced" / "not found" | run | revisit before presenting | @@ -99,7 +99,7 @@ Full recipes and rationale: `${CLAUDE_PLUGIN_ROOT}/skills/research/context/disci 2. **Queries scale to open questions: the floor is a starting point, not a target.** Phase 1 opens with ≥3 queries to seed the evidence base; Phase 2 and Phase 3 each run **one query per unresolved gap/conflict** surfaced by the prior phase's written analysis (≥3, no upper cap). Every floor below is a minimum; a run that stops at the floor while numbered gaps remain has not finished the phase 3. **3 distinct tool types minimum per phase**. One search engine plus one synthesis tool does not meet it; mix in direct fetches, doc-MCP servers, `gh api`, or documentation agents your environment provides 4. **4+ distinct tool types across the topic**. Phases cannot share the same 3 tools end-to-end. Cross-phase tool diversity is the consensus-driving mechanism -5. **Source-tier ratio per claim**. Every accepted claim has ≥1 Tier 0/1 (primary source captured this turn) PLUS ≥2 independent corroborators that cover the claim's target version (a `historical` source is recorded, never counted), however authoritative the primary is, because a canonical doc can be stale. Three synthesis-tool citations of three blogs = 1 Tier 2 source, NOT 3. Track diversity per claim +5. **Source-tier ratio per claim**. Every accepted claim has ≥1 Tier 0/1 (primary source captured this turn) PLUS ≥2 independent corroborators that cover the claim's target version (a `historical` source is recorded, never counted), however authoritative the primary is, because a canonical doc can be stale. Three synthesis-tool citations of three blogs = 1 Tier 2 source, NOT 3. Track diversity per claim. The one exception: the discipline file's "Single-source first-party content claims" 6. **Recency gate, first-party docs lag releases**, one query fetches the latest upstream changelog or release notes this turn and confirms the claims are current as of it. A major version bump invalidates prior docs, first-party included; treat any doc-vs-changelog lag as a conflict to resolve, not a closed answer. The 30/14/90-day staleness windows: the discipline file's "Recency gate" 7. **One falsification query in Phase 2**. Phase 2 includes exactly one query that attempts to falsify the leading hypothesis from Phase 1; without it Phase 2 confirms Phase 1 by default 8. **Broad-topic auto-detect → doubled minimums**, when the topic involves 2+ vendors / 2+ tools / 3+ proper-noun products / comparison ("X vs Y") / migration ("X replaces Y") → 6+ queries per phase, 12+ total, 5+ tool types, 4+ Tier 0/1 sources per claim @@ -120,13 +120,6 @@ Dated record: `${CLAUDE_PLUGIN_ROOT}/reference/parent-contract.md`, "Harness facts the dispatch design rests on". A dispatched run follows envelope `Source breadth:` from this load, not the researcher pin. Missing line: `high`, named in the artifact. -**`breadth=` narrows, never widens.** A `breadth=low` or `breadth=medium` token in `$ARGUMENTS` -selects that row when it is below caller effort; source breadth is the lower of the two, and -nothing here widens the researcher's `maxTurns: 40`. A dispatched run writes the resolved row to `Source breadth:` and the matching word -to `Budget:` (parent contract, "`Budget:` vocabulary"). Without the token, lowering session effort -before invoking is the other lever. Use `low` for a question about one named artifact, folder, or -version. - | Effort | Source breadth | |---|---| | `low` | Phase 0 if bounded, Phase 1 at existing floors and under the cap below, Phase 2 as the mandatory falsification query only (no per-gap expansion). Skip Phase 3 and Phase 4 | @@ -136,6 +129,13 @@ version. The Effort row is the ceiling over discipline 8. Rationale and skipped-phase N/A: the discipline file's "Effort, source breadth". +**`breadth=` narrows, never widens.** A `breadth=low` or `breadth=medium` token in `$ARGUMENTS` +selects that row when it is below caller effort; source breadth is the lower of the two, and +nothing here widens the researcher's `maxTurns: 40`. A dispatched run writes the resolved row to `Source breadth:` and the matching word +to `Budget:` (parent contract, "`Budget:` vocabulary"). Without the token, lowering session effort +before invoking is the other lever. Use `low` for a question about one named artifact, folder, or +version. + **Phase 1 at `low` is capped at 6 web queries and fetches combined**, above the 3-query floor and below the doubled minimums. For a single named artifact or folder, read it directly first (`Read`, `Glob`, `Grep`, or a listing) and let what it shows choose the queries; local reads do not count @@ -233,10 +233,10 @@ Local counterpart: `/discovery:explore` (what IS in the repo); this skill covers Present research findings as, and if invoked standalone present them directly, while inside a larger workflow they feed the subsequent planning step: 1. **Summary**. 2-3 sentence answer to the research question, preceded by one line naming any decision the findings leave to the user (e.g. two primary sources conflict, or a gap blocks the answer), or omitted when none -2. **Evidence table**. `Claim | Sources (Tier 0/1 entries cite the URL/command fetched THIS turn) | Tier | Tool diversity | Confidence`. A source whose `standing:` is `historical` carries the label historical in its Sources cell +2. **Evidence table**. `Claim | Sources (Tier 0/1 entries cite the URL/command fetched THIS turn) | Tier | Tool diversity | Confidence`. A source whose `standing:` is `historical` carries the label historical in its Sources cell, and a flagged claim's Confidence cell reads `HIGH (single source)` 3. **Fetch log**, the written record criteria 6 and 9 are graded against, so it is WRITTEN, not recalled. One entry per fetch PER CLAIM: `Claim | URL or command | artifact-ladder rung | tool used | outcome`, and each accepted claim carries the entry for the rung it came from AND one for every rung above it. **The outcome vocabulary is a parsed schema, not free text**. Five values, three of which look interchangeable and are not, plus the composite changelog entry criterion 6 grades. Write it to the spec in `${CLAUDE_PLUGIN_ROOT}/skills/research/context/artifact-shape.md` ("The fetch log") 4. **Conflicts**. Disagreements between sources (flagged explicitly; primary wins over blog consensus) -5. **Gaps**. Claims not at ≥1 primary + 2 independent corroborators, OR LOW confidence (flagged for follow-up). A gap asserting absence names the sources checked AND the sources left unchecked, never a bare "not found" +5. **Gaps**. Claims not at ≥1 primary + 2 independent corroborators (or a holding `single source` flag), OR LOW confidence (flagged for follow-up). A gap asserting absence names the sources checked AND the sources left unchecked, never a bare "not found" 6. **Recency status**. Primary-source age per tool/library claim 7. **Project fit**. How findings align with the consuming project's conventions and stated direction 8. **Outcome gate result**. Pass, or which criterion failed and what was re-run, plus effort and any skipped phases diff --git a/plugins/discovery/skills/research/context/artifact-shape.md b/plugins/discovery/skills/research/context/artifact-shape.md index 84a6beafee..ebc535572f 100644 --- a/plugins/discovery/skills/research/context/artifact-shape.md +++ b/plugins/discovery/skills/research/context/artifact-shape.md @@ -66,7 +66,8 @@ section: abstract: claims: - claim: "" - confidence: HIGH # HIGH | MEDIUM | LOW + confidence: HIGH # HIGH | HIGH (single source) | MEDIUM | LOW + single_source: "" # only at HIGH (single source); omit otherwise tiers: [0, 1] # source tiers backing this claim applies_to: " " # the claim's target, or version-independent sources: # what makes gate criterion 4 gradeable off the artifact @@ -84,8 +85,15 @@ produced_by: --- ``` -The vocabulary is reused, never reinvented: `HIGH | MEDIUM | LOW` and `Tier 0..3` are the research -skill's own, defined in `discipline.md`. +The vocabulary is reused, never reinvented: `HIGH | HIGH (single source) | MEDIUM | LOW` and +`Tier 0..3` are the research skill's own, defined in `discipline.md`. + +**`single_source:` is the flag's reason, and criterion 4 grades it.** A claim at +`HIGH (single source)` carries it; no other claim does. It states why only one publisher of the +claim's content exists, so a verifier that never saw the run can judge that reason instead of +counting corroborators the claim cannot have. A claim at that level without the field fails +criterion 4. A repost of the primary is recorded under the primary's `pool`, so it never reads as a +second source. Definition and limits: `discipline.md`'s "Single-source first-party content claims". **`sources[]` is not redundant with `tiers[]`.** It is what lets outcome-gate criterion 4, "≥2 INDEPENDENT corroborators, not two cites of one upstream pool", be graded **by a verifier that never diff --git a/plugins/discovery/skills/research/context/discipline.md b/plugins/discovery/skills/research/context/discipline.md index 323eac45d4..92d402f468 100644 --- a/plugins/discovery/skills/research/context/discipline.md +++ b/plugins/discovery/skills/research/context/discipline.md @@ -7,6 +7,7 @@ Recipes and rationale behind the bars stated in the research skill's SKILL.md bo - [Source tiers (canonical for this plugin)](#source-tiers-canonical-for-this-plugin) - [Source-tier ratio (per claim)](#source-tier-ratio-per-claim) +- [Single-source first-party content claims](#single-source-first-party-content-claims) - [Recency gate (for libraries, tools, CLIs, APIs)](#recency-gate-for-libraries-tools-clis-apis) - [Falsification step (mandatory Phase 2 query)](#falsification-step-mandatory-phase-2-query) - [Broad-topic auto-detect](#broad-topic-auto-detect) @@ -37,12 +38,24 @@ Recipes and rationale behind the bars stated in the research skill's SKILL.md bo ## Source-tier ratio (per claim) -Every accepted claim has at least one Tier 0/1 source plus two independent corroborators of any tier. +Every accepted claim has at least one Tier 0/1 source plus two independent corroborators of any tier. The one exception is the next section. **Anti-pattern:** three AI-synthesis citations of three different secondary blogs = 1 Tier 2 source, not 3. They're synthesizing from the same upstream pool. Count INDEPENDENT primary sources, not citation count. **Track tool diversity per topic in the evidence table.** Two sources both from one synthesis tool / both from one search engine / both from one author's blog network = 1 corroborator, not 2. +## Single-source first-party content claims + +Some claims can have only one publisher. A **first-party content claim** states what a named Anthropic page, file or changelog says. The content exists in one place, so searching for an independent second source finds only copies of it. Such a claim passes criterion 4 with no counted corroborator when all of these hold: + +- **The claim states why only one publisher exists**, in the sidecar header's `single_source:` field. The verifier grades that reason; one that does not hold fails criterion 4 like any other uncorroborated claim. +- **A repost is not a second source.** A blog post, forum answer, mirror or synthesis answer that quotes or restates the page shares its pool. Record it with the page's `pool`, never count it, and keep the flag: a repost does not turn the claim into a corroborated one. +- **The primary is the named artifact itself, fetched this turn**, and it passes every run-owned row as any other primary does. + +The claim's confidence is `HIGH (single source)` (see "Confidence calibration"), and the flag stays visible wherever the claim goes: the evidence table's Confidence cell, the sidecar header, the answer, and any synthesis built from it. A flagged claim may ground a code edit, and the record written beside that edit carries the flag. + +**A behavior claim is not a content claim.** What a product does when it runs (a default that takes effect, a limit it enforces, an error it raises) can be checked by a live probe or found in an issue report, so it still needs two independent corroborators, whatever its docs page says. A content claim about a page and a behavior claim drawn from that page are two claims: split them, and only the content claim can carry the flag. + ## Recency gate (for libraries, tools, CLIs, APIs) When the topic touches a library, tool, CLI, API or framework that ships releases, one Phase 1 or Phase 2 query fetches the latest upstream changelog or release notes this turn and confirms the claims are current as of it. Acceptable forms: `gh api repos///releases/latest`, WebFetch on a raw `CHANGELOG.md` URL, the vendor's "What's New" page. The windows below bound how stale a cited doc may be before this cross-check is required. A stable project whose latest release is older than the window still passes once that release is confirmed to be the current one. @@ -218,7 +231,7 @@ The "top of Google" is a ranking artifact, not an authority signal. SEO content **An announcement is the shallowest rung that still carries the claim.** It states the headline figure; the specific run, its conditions, and its methodology live at rung 1. Checking an announcement, an intro page, and a couple of searches, then reporting the figure as unsourced, is a ladder that was never walked. -**Authoritative is not a waiver for corroboration.** Even the canonical doc still needs ≥2 independent corroborators and a freshness check. First-party docs routinely lag major releases. When the topic post-dates a major version, cross-check the canonical doc against the upstream changelog/release and treat any lag as a conflict to resolve. +**Authoritative is not a waiver for corroboration.** Even the canonical doc still needs ≥2 independent corroborators and a freshness check; the one exception is a claim about what the doc itself says, under "Single-source first-party content claims" above. First-party docs routinely lag major releases. When the topic post-dates a major version, cross-check the canonical doc against the upstream changelog/release and treat any lag as a conflict to resolve. **Escalate on block, never downgrade.** A direct-fetch 403/429 means wrong fetcher, not vanished source. Escalation order: (1) a headless-browser URL reader if connected; (2) a managed scraping tool if available; (3) a synthesis tool forced to the blocked domain (domain-filter option). Only after those fail, fall back to secondary sources, and document the gap. @@ -292,10 +305,11 @@ If a required tool category is unavailable this session (no synthesis MCP server The evidence-table `Confidence` column must be set per claim: - **HIGH**: 3+ independent Tier 0/1 sources agree; recency gate passed; falsification query failed to find counter-evidence +- **HIGH (single source)**: a first-party content claim whose one publisher is its primary, fetched this turn, with a `single_source:` reason that holds; recency gate passed; falsification query failed to find counter-evidence. The flag is part of the level and is never dropped (see "Single-source first-party content claims") - **MEDIUM**: 3+ sources agree but mix of Tier 0/1 + Tier 2; OR 2 Tier 0/1 + open falsification gap; OR primary source > 30d old without changelog cross-check - **LOW**: fewer than 3 sources; OR sources conflict; OR Tier 2-only consensus; OR primary source > 90d old -Only HIGH-confidence claims are accepted (the outcome gate enforces this). A MEDIUM or LOW claim is a **Gap**: return to Phase 4 follow-up and iterate until HIGH, or report it as a gap; never a basis for code edits. +Only HIGH and HIGH (single source) claims are accepted (the outcome gate enforces this). A MEDIUM or LOW claim is a **Gap**: return to Phase 4 follow-up and iterate until HIGH, or report it as a gap; never a basis for code edits. ## Joint-inference check diff --git a/plugins/discovery/skills/research/context/gotchas.md b/plugins/discovery/skills/research/context/gotchas.md index 509cffff75..d7ca706dd8 100644 --- a/plugins/discovery/skills/research/context/gotchas.md +++ b/plugins/discovery/skills/research/context/gotchas.md @@ -53,6 +53,12 @@ outcome gate's artifact-grounded criteria, or not at all. `historical` from each source's `published:` and `applies_to:`, and criterion 12's era and scenario checks ask whether the source covers the claim's product line and situation. A `historical` source is labeled and never counted. +- **Counting a repost as the second source.** A blog post or synthesis answer that restates an + Anthropic page is that page again. Counting it lets a single-publisher claim pass criterion 4 as + corroborated, and the `single source` flag that should travel with the claim disappears. Record + the repost under the page's pool and flag the claim. The opposite slip costs as much: flagging a + behavior claim because its docs page is the only one found, when a probe or an issue could + corroborate it. - **Reading the coverage ledger instead of running the gate.** A model cannot reliably audit its own checklist, and the context most motivated to call it finished is the one reading it. Criterion 11 cites the script's exit status. Exit 2, a ledger the script could not parse, is a FAIL, never a diff --git a/plugins/discovery/skills/research/evals/evals.json b/plugins/discovery/skills/research/evals/evals.json index 21b7737e59..3745f17ac4 100644 --- a/plugins/discovery/skills/research/evals/evals.json +++ b/plugins/discovery/skills/research/evals/evals.json @@ -399,6 +399,35 @@ "The Phase 1 cap for a named folder or artifact is a direct read of it first, then at most 6 web queries and fetches", "A verifier FAIL on a verifier-owned row at low is recorded as a Gap, Conflicts entry or named verification fail, not resumed" ] + }, + { + "id": 28, + "name": "single-source-content-claim-passes-flagged", + "prompt": "Research what Anthropic's Claude Code changelog says about one named release. [The dispatched discovery:researcher returns a well-formed payload with verification: pending, and both gates exit 0. One accepted claim reads 'the changelog entry for that release lists the change'. Its sidecar header records confidence: HIGH (single source) and single_source: 'only Anthropic publishes this changelog; every copy found elsewhere quotes it'. Its sources are the changelog file, fetched this turn, as role: primary, plus a blog post and a newsletter issue, each quoting that changelog entry and each recorded under the changelog's pool.]", + "files": [], + "expected_output": "Briefs the sibling verifier on rows 4, 7 and 12. The verifier grades row 4 on its single-source branch: the claim states what a named Anthropic file says, the stated reason holds, and the blog post and newsletter are reposts recorded under the changelog's pool and not counted as corroborators. Row 4 passes without two corroborators, and row 7 passes at HIGH (single source). The parent writes verification: pass and presents the claim with its single source flag visible in the evidence table and the answer. A code edit that later rests on the claim carries the flag in the record beside it.", + "expectations": [ + "The verifier is briefed on rows 4, 7 and 12 by number", + "Row 4 is graded on the claim's single_source: reason, not failed for lacking two independent corroborators", + "The blog post and the newsletter are treated as reposts of the changelog: recorded, not counted as second sources", + "Row 7 passes with the claim at HIGH (single source), not plain HIGH", + "The claim is presented with the single source flag visible, not filed as a Gap", + "A record of a code edit grounded on the claim carries the single source flag" + ] + }, + { + "id": 29, + "name": "behavior-claim-from-one-page-is-a-gap", + "prompt": "Research whether setting one named Claude Code setting to false stops automatic updates. [The dispatched discovery:researcher returns a well-formed payload with verification: pending, and both gates exit 0. One accepted claim reads 'setting the key to false stops automatic updates'. Its sidecar header records confidence: HIGH (single source) and single_source: 'Anthropic is the only publisher of its own product documentation'. Its only source is the settings documentation page, fetched this turn, as role: primary. No probe was run and no issue tracker was searched.]", + "files": [], + "expected_output": "The verifier grades row 4 and finds the flag misapplied: the claim states what the product does when the key is set, a behavior claim that a live probe or an issue report could corroborate, so the single-source branch does not apply and the claim needs two independent corroborators, which it lacks. Row 4 fails, and so does row 7. The parent files the claim as a Gap and routes it to Phase 2 for a probe or an issue search. A separate content claim, that the settings page documents the key with that effect, may be split out and carry the flag.", + "expectations": [ + "The verifier treats the claim as a behavior claim, not a first-party content claim", + "The single_source: reason is judged not to hold for a behavior claim, so row 4 fails", + "Row 7 fails: the claim is not accepted at HIGH (single source)", + "The claim is filed as a Gap routed to Phase 2 for a live probe or an issue search, not presented as accepted", + "Only a split-out content claim about what the page states may carry the single source flag" + ] } ] } From ec87889c5187798318c4dc32a80ffb8652f1e5da Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Fri, 2 Oct 2026 02:27:06 -0400 Subject: [PATCH 04/11] fix(harness-config): make the audit-instructions native records links-only The bundled claude-api and doctor records in audit-instructions now hold our decision, a pointer, an as-of date and a recheck trigger, with no restated upstream text. The outdated "not public" subcommand row is dropped; its trigger now fires when the claude-api skill docs page lists build-eval and hillclimb. The doctor records state our rule for prompt-audit: offer it only when doctor and claude-api both resolve, and never ask the audit to apply its proposals. The claude-api to harness-config:audit-instructions row in the native-surfaces store is updated and the view regenerated. Co-Authored-By: Claude Opus 5.5 --- docs/native-surfaces.md | 13 +++---- docs/native-surfaces/records.json | 17 +++++----- .../harness-config/.claude-plugin/plugin.json | 2 +- plugins/harness-config/CHANGELOG.md | 14 ++++++++ .../skills/audit-instructions/SKILL.md | 18 +++++----- .../reference/bundled-claude-api.md | 34 ++++++++----------- .../reference/native-doctor.md | 19 +++++------ 7 files changed, 64 insertions(+), 53 deletions(-) diff --git a/docs/native-surfaces.md b/docs/native-surfaces.md index bbe587a129..c2588f22bf 100644 --- a/docs/native-surfaces.md +++ b/docs/native-surfaces.md @@ -542,18 +542,19 @@ and when. See [`docs/conventions/native-references/`](conventions/native-referen ### `claude-api` → `harness-config:audit-instructions` -- **Verdict:** `complementary`: Composite posture, decided at the ClaudeDevs cost-performance adoption interview: wrap or point to the bundled subcommand where it fits the use case, and run our own processes where they fit, rather than routing one way on paper. The bundled skill's prompt-audit subcommand is the vendor's apply-sweep over the working directory's whole prompt surface, application code included; audit-instructions is a standing report-only audit of locally-owned Claude Code instruction surfaces with the versioned I-catalog, target-model scoping, and deterministic pre-scans. ADR-0028 already composes both: run the vendor procedure per model change, feed recurring gap shapes back into the catalog. The app-code surface stays with the bundled skill (scope widening rejected at the same interview). +- **Verdict:** `complementary`: Composite posture, decided at the ClaudeDevs cost-performance adoption interview: wrap or point to the bundled subcommand where it fits the use case, and run our own processes where they fit, rather than routing one way on paper. Model migrations and application-code prompts go to the bundled skill's prompt-audit subcommand; audit-instructions stays a standing report-only audit of locally-owned Claude Code instruction surfaces with the versioned I-catalog, target-model scoping, and deterministic pre-scans, and never chains into a prompt-audit apply. ADR-0028 already composes both: run the vendor procedure per model change, feed recurring gap shapes back into the catalog. The app-code surface stays with the bundled skill (scope widening rejected at the same interview). - **Integration:** `route` - **Native surface:** `claude-api` (bundled skill; markers: gated) - **Our component:** `harness-config:audit-instructions` (skill) - **Evidence:** - - binary extraction 2026-09-09 (claude.exe 2.1.263): registerClaudeApiSkill present; subcommand array cost-optimize, migrate, managed-agents-onboard, prompt-audit, upgrade, build-eval, hillclimb - - platform docs claude-api-skill page (fetched 2026-09-09): 'The skill comes bundled with Claude Code and is also available in the open-source Anthropic skills repository' - - hillclimb and build-eval are bundled-only: absent from anthropics/skills HEAD 41bbe19 (2026-09-03) and from the skill's docs page + - binary extraction 2026-09-09 (claude.exe 2.1.263) and 2026-10-02 (Claude Code 2.1.287): registerClaudeApiSkill present; the subcommand array includes prompt-audit + - distribution pointer: https://platform.claude.com/docs/en/agents-and-tools/agent-skills/claude-api-skill#in-claude-code-bundled (as of 2026-10-02) + - subcommand pointer: https://code.claude.com/docs/en/skills#work-on-claude-api-projects; published guides: https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/prompt-audit.md and model-migration.md at the same commit (as of 2026-10-02) + - the routing, mutation gate and presence records live in the skill's reference/bundled-claude-api.md, links-only - executed composition precedent: ADR-0028 (fleet-wide prompt-audit run, 805 findings applied; repeats per model change; findings are edits, not criteria) - verdict recorded from the owner's interview answers in docs/upstream/claudedevs-cost-performance.md Lane M and Lane T2, 2026-09-10 -- **Observation:** extraction: extracted from binary 2.1.263 at node_modules/@anthropic-ai/claude-code/bin/claude.exe (registerClaudeApiSkill string plus subcommand array; bundled shared/evals/eval-hillclimb.md extracted and read); bulk registrar enumeration was broken at this build, so this row's evidence is the targeted extraction, not the inventory JSON (2026-09-09) -- **Recheck trigger:** a Claude Code release changes the bundled claude-api skill's subcommand set, or the anthropics/skills repo or the platform claude-api-skill docs page gains hillclimb/build-eval (which also fires the docs/upstream/claudedevs-cost-performance.md hillclimb row) (verified 2026-09-10) +- **Observation:** extraction: extracted from binary v2.1.287 on 2026-10-02 (the /harness-ops:inventory --binary-only extraction lists claude-api as a gated, model-invocable bundled skill; the registerClaudeApiSkill string and subcommand array read from the same binary) (2026-10-02) +- **Recheck trigger:** a Claude Code release changes the bundled claude-api skill's subcommand set or names prompt-audit or the model-migration guide, the platform claude-api-skill docs page lists build-eval and hillclimb, or a commit to anthropics/skills changes skills/claude-api/shared/prompt-audit.md or model-migration.md (verified 2026-10-02) - **Baked:** description phrase yes · Boundary section yes · Native step no · suggest sentence no - **Budget caveat:** the baked phrase may be dropped from the skill listing under budget pressure. It is the best available routing surface, not a guaranteed one diff --git a/docs/native-surfaces/records.json b/docs/native-surfaces/records.json index 57143faf66..ec21cd39a1 100644 --- a/docs/native-surfaces/records.json +++ b/docs/native-surfaces/records.json @@ -650,22 +650,23 @@ }, "verdict": "complementary", "integration": "route", - "reason": "Composite posture, decided at the ClaudeDevs cost-performance adoption interview: wrap or point to the bundled subcommand where it fits the use case, and run our own processes where they fit, rather than routing one way on paper. The bundled skill's prompt-audit subcommand is the vendor's apply-sweep over the working directory's whole prompt surface, application code included; audit-instructions is a standing report-only audit of locally-owned Claude Code instruction surfaces with the versioned I-catalog, target-model scoping, and deterministic pre-scans. ADR-0028 already composes both: run the vendor procedure per model change, feed recurring gap shapes back into the catalog. The app-code surface stays with the bundled skill (scope widening rejected at the same interview).", + "reason": "Composite posture, decided at the ClaudeDevs cost-performance adoption interview: wrap or point to the bundled subcommand where it fits the use case, and run our own processes where they fit, rather than routing one way on paper. Model migrations and application-code prompts go to the bundled skill's prompt-audit subcommand; audit-instructions stays a standing report-only audit of locally-owned Claude Code instruction surfaces with the versioned I-catalog, target-model scoping, and deterministic pre-scans, and never chains into a prompt-audit apply. ADR-0028 already composes both: run the vendor procedure per model change, feed recurring gap shapes back into the catalog. The app-code surface stays with the bundled skill (scope widening rejected at the same interview).", "evidence": [ - "binary extraction 2026-09-09 (claude.exe 2.1.263): registerClaudeApiSkill present; subcommand array cost-optimize, migrate, managed-agents-onboard, prompt-audit, upgrade, build-eval, hillclimb", - "platform docs claude-api-skill page (fetched 2026-09-09): 'The skill comes bundled with Claude Code and is also available in the open-source Anthropic skills repository'", - "hillclimb and build-eval are bundled-only: absent from anthropics/skills HEAD 41bbe19 (2026-09-03) and from the skill's docs page", + "binary extraction 2026-09-09 (claude.exe 2.1.263) and 2026-10-02 (Claude Code 2.1.287): registerClaudeApiSkill present; the subcommand array includes prompt-audit", + "distribution pointer: https://platform.claude.com/docs/en/agents-and-tools/agent-skills/claude-api-skill#in-claude-code-bundled (as of 2026-10-02)", + "subcommand pointer: https://code.claude.com/docs/en/skills#work-on-claude-api-projects; published guides: https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/prompt-audit.md and model-migration.md at the same commit (as of 2026-10-02)", + "the routing, mutation gate and presence records live in the skill's reference/bundled-claude-api.md, links-only", "executed composition precedent: ADR-0028 (fleet-wide prompt-audit run, 805 findings applied; repeats per model change; findings are edits, not criteria)", "verdict recorded from the owner's interview answers in docs/upstream/claudedevs-cost-performance.md Lane M and Lane T2, 2026-09-10" ], "observation": { "class": "extraction", - "detail": "extracted from binary 2.1.263 at node_modules/@anthropic-ai/claude-code/bin/claude.exe (registerClaudeApiSkill string plus subcommand array; bundled shared/evals/eval-hillclimb.md extracted and read); bulk registrar enumeration was broken at this build, so this row's evidence is the targeted extraction, not the inventory JSON", - "date": "2026-09-09" + "detail": "extracted from binary v2.1.287 on 2026-10-02 (the /harness-ops:inventory --binary-only extraction lists claude-api as a gated, model-invocable bundled skill; the registerClaudeApiSkill string and subcommand array read from the same binary)", + "date": "2026-10-02" }, "recheck": { - "trigger": "a Claude Code release changes the bundled claude-api skill's subcommand set, or the anthropics/skills repo or the platform claude-api-skill docs page gains hillclimb/build-eval (which also fires the docs/upstream/claudedevs-cost-performance.md hillclimb row)", - "verified": "2026-09-10" + "trigger": "a Claude Code release changes the bundled claude-api skill's subcommand set or names prompt-audit or the model-migration guide, the platform claude-api-skill docs page lists build-eval and hillclimb, or a commit to anthropics/skills changes skills/claude-api/shared/prompt-audit.md or model-migration.md", + "verified": "2026-10-02" }, "baked": { "description_phrase": true, diff --git a/plugins/harness-config/.claude-plugin/plugin.json b/plugins/harness-config/.claude-plugin/plugin.json index 65218cf346..b5a45b9fe9 100644 --- a/plugins/harness-config/.claude-plugin/plugin.json +++ b/plugins/harness-config/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "harness-config", - "version": "1.3.3", + "version": "1.3.4", "description": "Nine configuration-health skills (plus setup) for a repo's Claude Code configuration: audit (settings.json / .mcp.json / hooks / plugins / permissions drift), audit-automation-gaps (evidence-gated verdicts on automation gaps), audit-permission-grants (allow-rule / allowed-tools grants for auto-mode durability and portability), audit-permission-state (the permission rules actually in effect: every settings scope merged with per-rule provenance, what auto mode drops on entry, config written where nothing reads it, and which managed intents are enforced versus loosenable), draft-auto-mode-rules (interview and draft a paste-ready autoMode classifier block; prints only, never writes), audit-instructions (locally-owned instruction surfaces vs current model capability, proposing removals/rewrites of instructions the model no longer needs, and detecting cross-surface instruction conflicts), audit-prompting-postures (the additive lane: posture guidance the prompting guide says a component's purpose needs but the component does not carry), audit-pass (one coordinated, ordered, resumable pass over a named target: three-scope inventory, run-time-derived exclusion set, stable finding identity, suppression memory, resume, one human gate, delegating every check to the plugin that owns it), and unhobble (the empirical bare-baseline experiment: reversibly strip a repo's standing instructions, log real stumbles against the current model, re-add only what evidence earns). Boundary: harness-memory owns the health of CLAUDE.md, AGENTS.md, CLAUDE.local.md, .claude/rules/ and auto-memory (structure, size, placement, index integrity); harness-config audit-instructions judges whether instruction text across those files and skills, agents and hooks still fits the current model, and runs no memory-file hygiene checks.", "author": { "name": "Melodic Software", diff --git a/plugins/harness-config/CHANGELOG.md b/plugins/harness-config/CHANGELOG.md index 4c8d2a0a1d..228661335e 100644 --- a/plugins/harness-config/CHANGELOG.md +++ b/plugins/harness-config/CHANGELOG.md @@ -5,6 +5,20 @@ All notable changes to the `harness-config` plugin are documented here. Format f Versions 0.51.8 to 0.51.9 and 0.51.11 to 0.51.14 were reserved by parallel branches and never released. +## [1.3.4] - 2026-10-02 + +### Fixed + +- **`audit-instructions` offers `/doctor prompt-audit` only where it can run.** The offer now + requires both `doctor` and the bundled `claude-api` skill to resolve in the session, and the + report says when it was skipped. The skill never asks either audit to apply its proposals. +- **The `audit-instructions` records for the bundled `claude-api` and `doctor` skills are + links-only.** Each row in `reference/bundled-claude-api.md` and `reference/native-doctor.md` is + our decision, a pointer to the exact docs section or the guide at a pinned `anthropics/skills` + commit, an as-of date and a recheck trigger. The outdated record that two subcommands were + missing from the public repository and the docs is gone, and the record whose trigger fired at + Claude Code 2.1.283 was re-derived at 2.1.287. + ## [1.3.3] - 2026-10-02 ### Changed diff --git a/plugins/harness-config/skills/audit-instructions/SKILL.md b/plugins/harness-config/skills/audit-instructions/SKILL.md index f1edadfb09..255798185a 100644 --- a/plugins/harness-config/skills/audit-instructions/SKILL.md +++ b/plugins/harness-config/skills/audit-instructions/SKILL.md @@ -106,8 +106,8 @@ cross-surface conflicts, and harness claims that misstate Claude Code's own beha wants both, run both: recurring gap shapes the vendor sweep surfaces feed this catalog as new rows, and this skill's findings never substitute for the vendor procedure on a model change. -**Mutation gate.** `prompt-audit` edits files when the request asks for edits. This skill is -report-only: never chain into a `prompt-audit` apply; surface the finding and let the user run it. +**Mutation gate.** This skill is report-only: never ask `prompt-audit` to apply its proposed diff +and never chain into an apply; surface the finding and let the user run it. **Availability is never assumed.** Bundled surfaces are gated by settings, environment, plan, and host; this section states what to do when the surface resolves, never that it is present. Subcommand @@ -115,19 +115,19 @@ set, distribution facts, recheck triggers: [reference/bundled-claude-api.md](ref ## Boundary, the bundled `doctor` skill -`/doctor prompt-audit` also audits these files for outdated or conflicting instructions. +`/doctor prompt-audit` and this skill both audit instruction files and are easily conflated. -- **`doctor` (bundled skill, alias `checkup`)**: its `prompt-audit` subcommand audits `CLAUDE.md` files, skills, agents, and commands for - older-model prompting patterns. It is reserved for the person to run; the model does not invoke it. +- **`doctor` (bundled skill, alias `checkup`)**: its `prompt-audit` subcommand is the vendor's + audit of instruction files. The person runs it; the model does not invoke it. - **This skill (marketplace plugin).** Report-only catalog audit that adds over-prescription with target-model scope, stale Claude Code behavior claims, and the cross-surface conflict pass. **Routing.** At the end of the run, offer it to the person: you can run `/doctor prompt-audit` alongside this skill. An unattended run records the offer in its output instead of asking. -**Mutation gate.** Applying its proposed edits is the person's call; never chain into `/doctor`. -**Availability is never assumed.** Its gates (settings, environment, version, and the bundled -skill it runs through) are read live from the pointers in the records; this section states what -to offer, never that it is present. Records: [reference/native-doctor.md](reference/native-doctor.md). +**Mutation gate.** Never ask the audit to apply its proposals, and never chain into `/doctor`. +**Availability is never assumed.** Offer it only when `doctor` and the bundled `claude-api` skill +both resolve in this session, else report the offer as skipped; never state that either is present. +Records: [reference/native-doctor.md](reference/native-doctor.md). ## Arguments diff --git a/plugins/harness-config/skills/audit-instructions/reference/bundled-claude-api.md b/plugins/harness-config/skills/audit-instructions/reference/bundled-claude-api.md index 19577dc86f..b874214e63 100644 --- a/plugins/harness-config/skills/audit-instructions/reference/bundled-claude-api.md +++ b/plugins/harness-config/skills/audit-instructions/reference/bundled-claude-api.md @@ -1,20 +1,24 @@ # The bundled `claude-api` skill, as this skill relates to it -Four-part records behind the `## Boundary` section in `SKILL.md`: each claim names its basis, its -as-of date, and the observable event that obliges re-deriving it. The section carries the -conclusion; this file carries what it rests on. Nothing here asserts that the surface is present in -any session; every claim is about what the surface does where it resolves. +Records behind the `## Boundary` section in `SKILL.md`. Each row is our decision, a pointer to the +upstream section that holds the specific, the date the decision was last derived, and the +observable event that obliges re-deriving it; the specific itself is read live at the pointer. The +section carries the conclusion; this file carries what it rests on. Nothing here asserts that the +surface is present in any session. -## What the surface is +Pinned commit for the `anthropics/skills` links below: `8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4`. -| Claim | Basis | As-of | Recheck trigger | +## Records + +| Decision | Pointer | As of | Recheck when | |---|---|---|---| -| The `claude-api` skill is a bundled Claude Code skill and is also published in the open-source Anthropic skills repository | The skill's platform docs page states it "comes bundled with Claude Code and is also available in the open-source Anthropic skills repository" (`platform.claude.com/docs/en/agents-and-tools/agent-skills/claude-api-skill`); the Claude Code binary registers it | 2026-09-09, against Claude Code 2.1.263 | The docs page stops carrying the bundling statement, or a release note moves the skill between bundled and marketplace distribution | -| Run through the `/claude-api` door, its `prompt-audit` subcommand scopes to the whole working directory's prompt surface: skill bodies, `CLAUDE.md` and rule files, tool descriptions, and request-building application code | The subcommand's own reference read source-as-spec from the public skills repository (`skills/claude-api`, `shared/prompt-audit.md`, inventory step) | 2026-09-09, repository HEAD of 2026-09-03 | The reference's inventory step changes scope, or the subcommand is renamed or removed | +| This skill routes to `claude-api` as a bundled surface and keeps no copy of its guides | For distribution: [the Claude API skill page, "In Claude Code (bundled)"](https://platform.claude.com/docs/en/agents-and-tools/agent-skills/claude-api-skill#in-claude-code-bundled) and [Bundled skills](https://code.claude.com/docs/en/skills#bundled-skills) | 2026-10-02 | Either section changes how the skill is distributed, or a release note moves it between bundled and marketplace distribution | +| This file keeps no list of the skill's subcommands; a reader takes the set from the pointer. This skill routes to `prompt-audit` only | For the subcommands and the version each needs: [Work on Claude API projects](https://code.claude.com/docs/en/skills#work-on-claude-api-projects); for their published guides: [`skills/claude-api/shared/`](https://github.com/anthropics/skills/tree/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared) at the pinned commit | 2026-10-02 | The [Claude API skill page](https://platform.claude.com/docs/en/agents-and-tools/agent-skills/claude-api-skill) lists `build-eval` and `hillclimb` (then repoint there), or a release note changes the skill's subcommand set | | We route Claude Code configuration audits to `/doctor prompt-audit [path]`, the same audit's Claude Code door, and application-code prompts to `/claude-api prompt-audit`; its scope, write posture and version floor are read live at the pointer | Pointer: [Audit your instruction files](https://code.claude.com/docs/en/memory#audit-your-instruction-files) and the `/doctor` row of [Commands](https://code.claude.com/docs/en/commands#all-commands) | 2026-10-01 | Either docs section changes the scope, write posture or version floor, or a release note changes `/doctor prompt-audit` | -| `prompt-audit` produces a report and a proposed diff, applying edits only when the request asked for them | Same reference, its output and apply steps | 2026-09-09 | The reference's apply posture changes | -| The bundled skill's subcommand set is wider than the public repository's: `cost-optimize`, `migrate`, `managed-agents-onboard`, `prompt-audit`, `upgrade`, `build-eval`, `hillclimb` ship in the binary, while `build-eval` and `hillclimb` are absent from the public repository and the skill's docs page | Direct read of the bundled skill inside Claude Code 2.1.263 against a clone of the public repository at HEAD `41bbe19` | 2026-09-09 | The public repository or the docs page gains the missing subcommands, or a release changes the bundled set | -| The model-migration guide includes `## Ground the migration with an eval`, and the `prompt-audit` guide still runs Steps 0–7 over Groups 1–4 | Direct read of those two guides extracted from Claude Code 2.1.282. Changelog 2.1.260 refreshed the skill's samples and is the release that added the eval section relative to the 2.1.258 basis this catalog first cited. Changelog 2.1.283 fired the earlier trigger; on our read it names neither the steps, the groups nor the eval section, and no note through 2.1.287 names `prompt-audit` | Guides read 2026-09-28; changelog read 2026-10-01 | A release note changes `prompt-audit`'s steps or groups, or the model-migration guide's eval section | +| Application-code prompts route to `prompt-audit`; this skill keeps to Claude Code instruction surfaces and does not widen to application code | For what the subcommand inventories: [`prompt-audit.md`, Step 1](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/prompt-audit.md#step-1-inventory-the-prompt-surface) at the pinned commit | 2026-10-02 | A commit to `anthropics/skills` changes `skills/claude-api/shared/prompt-audit.md`, or a release note names `prompt-audit` | +| This skill never asks `prompt-audit` to apply its proposed diff and never chains into an apply; it names the option and the person runs it | For its output and apply steps: [`prompt-audit.md`, Step 6](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/prompt-audit.md#step-6-produce-the-proposed-diff) onward at the pinned commit | 2026-10-02 | Same as the row above | +| A model change runs the vendor's procedure, the model-migration guide with its eval grounding and `prompt-audit`; this catalog cites the per-model prompting guides directly ([criteria.md](criteria.md), Sources) and copies neither vendor guide. Re-derived from the published copies at the pinned commit with Claude Code 2.1.287 installed; the bundled copies were not extracted, and our reading of the changelog through 2.1.287 found no later release naming either guide | For the eval grounding: [`model-migration.md`, "Ground the migration with an eval"](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/model-migration.md#ground-the-migration-with-an-eval); for the audit procedure: [`prompt-audit.md`](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/prompt-audit.md); both at the pinned commit. For the last release that changed `prompt-audit`: [changelog 2.1.283](https://code.claude.com/docs/en/changelog#2-1-283) | 2026-10-02 | A release note names `prompt-audit` or the model-migration guide, or a commit to `anthropics/skills` changes either file | +| The routing in `SKILL.md` reads "when the surface resolves in this session" and never "the surface is available" | For the settings that turn bundled skills off: [`disableBundledSkills`](https://code.claude.com/docs/en/settings-reference#disablebundledskills) and [Override skill visibility from settings](https://code.claude.com/docs/en/skills#override-skill-visibility-from-settings) | 2026-10-02 | A release or docs change adds, removes, or renames a setting or host restriction that gates bundled skills | ## Why the verdict is complementary @@ -31,11 +35,3 @@ Both surfaces judge prompt text against current-model doctrine, and neither repl The composite posture follows: run the vendor procedure on every model change and for application prompts; run this skill continuously on Claude Code surfaces; feed recurring gap shapes the vendor sweep surfaces into the catalog as rows rather than re-running the sweep to find them again. - -## Presence - -Bundled skills can be removed by `disableBundledSkills`, hidden by `skillOverrides`, and vary by -plan, platform, and host surface. The routing in `SKILL.md` therefore reads "when the surface -resolves in this session" and never "the surface is available". Verified against -`code.claude.com/docs/en/settings-reference.md` on 2026-09-09; recheck when a release or docs -change adds, removes, or renames a gating axis. diff --git a/plugins/harness-config/skills/audit-instructions/reference/native-doctor.md b/plugins/harness-config/skills/audit-instructions/reference/native-doctor.md index e61c18685d..352c981ce7 100644 --- a/plugins/harness-config/skills/audit-instructions/reference/native-doctor.md +++ b/plugins/harness-config/skills/audit-instructions/reference/native-doctor.md @@ -1,17 +1,16 @@ # The bundled `doctor` skill, as this skill relates to it -Four-part records behind the `doctor` Boundary section in `SKILL.md`. The section carries the -conclusion; this file carries what it rests on. Nothing here asserts that the surface is present in -any session. +Records behind the `doctor` Boundary section in `SKILL.md`. Each row is our decision, a pointer to +where the specific is read live, the date the decision was last derived, and the observable event +that obliges re-deriving it. The section carries the conclusion; this file carries what it rests +on. Nothing here asserts that the surface is present in any session. -| Claim | Basis | As of | Recheck when | +| Decision | Pointer | As of | Recheck when | |---|---|---|---| -| `doctor` is a bundled skill with alias `checkup`, user-invocable, with model invocation disabled, gated, and it survives `disableBundledSkills` | The `/harness-ops:inventory` extraction of the installed 2.1.284 binary (`model_invocable: false`, `disable_model_invocation: true`, `gated: true`, `survives_kill_switch: true`, `aliases: ["checkup"]`) | 2026-09-29, Claude Code 2.1.284 | A release renames or removes it, changes its alias, or changes its invocability | -| Its argument hint is `[prompt-audit []]` | Same binary extraction | 2026-09-29, Claude Code 2.1.284 | A release changes the argument hint | -| `/doctor prompt-audit` (also `/checkup prompt-audit`) audits `CLAUDE.md` files, skills, agents, and commands for prompting patterns written for older models | Claude Code changelog 2.1.283 | 2026-09-29 | A release note changes or removes `prompt-audit` | -| The commands page describes it as auditing `CLAUDE.md` files, skills, and other configuration for outdated or conflicting instructions, instead of running the checkup, and requires v2.1.283 or later | The `/doctor` row of | 2026-09-29 | The commands page row changes | -| `/doctor` stays typable under `disableBundledSkills`; `DISABLE_DOCTOR_COMMAND` or a `skillOverrides` entry `"doctor": "off"` hides it | , bundled skills section | 2026-09-29 | The skills page changes its gating for `/doctor` | -| Our decision: offer `prompt-audit` to the person and leave applying its proposed edits to them; never chain into it. Its write posture and its dependency on the bundled `claude-api` skill are read live from the pointer | | 2026-10-02 | That section changes when the audit edits files, or which settings turn it off | +| This skill offers `/doctor prompt-audit` to the person and never invokes it. The probe observed `doctor` as a bundled skill with alias `checkup`, user-invocable and not model-invocable, with an argument hint naming `prompt-audit` | The `/harness-ops:inventory` binary extraction, `bundled_skills.doctor` | 2026-10-02, Claude Code 2.1.287 | A release renames or removes `doctor`, or changes its alias, argument hint, or invocability | +| This skill keeps its own catalog and offers `/doctor prompt-audit` beside it at the end of a run; it copies none of that audit's checks | For what the audit covers: [Audit your instruction files](https://code.claude.com/docs/en/memory#audit-your-instruction-files); for the command and its version floor: the `/doctor` row of [All commands](https://code.claude.com/docs/en/commands#all-commands); for the release that added it: [changelog 2.1.283](https://code.claude.com/docs/en/changelog#2-1-283) | 2026-10-02 | That section or row changes, or a release note names `prompt-audit` | +| This skill never asks the audit to apply its proposals and never chains into `/doctor`; the person decides what to apply | For the audit's write posture: [Audit your instruction files](https://code.claude.com/docs/en/memory#audit-your-instruction-files) | 2026-10-02 | That section changes what the audit does before the person asks | +| Offer `/doctor prompt-audit` only when both `doctor` and the bundled `claude-api` skill resolve in this session; when either does not, the report says the offer was skipped | For what the audit depends on: [Audit your instruction files](https://code.claude.com/docs/en/memory#audit-your-instruction-files); for how `doctor` itself is gated: [Bundled skills](https://code.claude.com/docs/en/skills#bundled-skills); for the settings that turn bundled skills off: [`disableBundledSkills`](https://code.claude.com/docs/en/settings-reference#disablebundledskills) and [Override skill visibility from settings](https://code.claude.com/docs/en/skills#override-skill-visibility-from-settings) | 2026-10-02 | That section changes what the audit depends on, or a release adds, removes, or renames a setting that gates `doctor` or bundled skills | ## Why the verdict is complementary From 5a9b9218a286f5deed81b4ddd0696860def6ea48 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Thu, 1 Oct 2026 22:05:25 -0400 Subject: [PATCH 05/11] docs(context-budget): route skill pruning to /skill-doctor and record the boundary The audit's route-out now sends "which skills to turn off" to the built-in /skill-doctor, which the person runs, and keeps unused MCP servers and plugins with the bundled /doctor. A Boundary section and its four-part records file state the split; the README, the lever catalogue's routes and the report's Routes section say the same. The native-surfaces store gains one hand-added row (skill-doctor x context-budget:audit, complementary, integration suggest), with the generated view regenerated. No skill description changes. Co-Authored-By: Claude Opus 5.5 --- docs/native-surfaces.md | 20 ++++++++- docs/native-surfaces/records.json | 42 +++++++++++++++++++ .../context-budget/.claude-plugin/plugin.json | 2 +- plugins/context-budget/CHANGELOG.md | 15 +++++++ plugins/context-budget/README.md | 3 ++ plugins/context-budget/skills/audit/SKILL.md | 35 +++++++++++++--- .../skills/audit/reference/levers.json | 3 +- .../audit/reference/native-skill-doctor.md | 22 ++++++++++ .../skills/audit/reference/report.md | 4 +- 9 files changed, 135 insertions(+), 11 deletions(-) create mode 100644 plugins/context-budget/skills/audit/reference/native-skill-doctor.md diff --git a/docs/native-surfaces.md b/docs/native-surfaces.md index c2588f22bf..c8a595ac30 100644 --- a/docs/native-surfaces.md +++ b/docs/native-surfaces.md @@ -17,7 +17,7 @@ and when. See [`docs/conventions/native-references/`](conventions/native-referen | Lane | Rows | Baked | Integration | Verdicts | |---|---|---|---|---| -| Built-in CLI commands | 24 | 23 | route 4, suggest 20 | complementary 23, defer 1 | +| Built-in CLI commands | 25 | 24 | route 4, suggest 21 | complementary 24, defer 1 | | Bundled skills | 29 | 22 | route 18, suggest 9, wrap 2 | complementary 23, defer 6 | | Bundled workflows | 1 | 1 | suggest 1 | complementary 1 | | Plugin-backed built-ins | 4 | 2 | route 4 | complementary 3, defer 1 | @@ -430,6 +430,24 @@ and when. See [`docs/conventions/native-references/`](conventions/native-referen - **Baked:** description phrase no · Boundary section yes · Native step no · suggest sentence no - **Budget caveat:** the baked phrase may be dropped from the skill listing under budget pressure. It is the best available routing surface, not a guaranteed one +### `skill-doctor` → `context-budget:audit` + +- **Verdict:** `complementary`: The two meet only on skills. We send the choice of which skills to turn off to the built-in command, which the person runs; context-budget:audit measures a fresh headless session's startup payload per item, splits the built-in tool pools, and ledgers what each toggle measurably saved. User-only, so ours offers it to the person. Human-added pair: discovery scored it under the 0.30 floor. Ruled 2026-10-02 by operator direction on the orchestrator's recommendation. +- **Integration:** `suggest` +- **Native surface:** `skill-doctor` (built-in command; markers: gated, model-invocation-disabled) +- **Our component:** `context-budget:audit` (skill) +- **Evidence:** + - `skill-doctor` present in the 2.1.285 extraction as builtin-command (source builtin) + - invocation mode (2026-10-02, Claude Code 2.1.285): user-invocable only, model invocation disabled (command type `local-jsx`) + - markers: model-invocation-disabled from the extraction; gated from the docs pointers below, since the extraction set no gated flag (the same basis as the sibling harness-ops:audit-skill-visibility row) + - detect: origin discovered only at threshold 0.01, score 0.0729 from shared tokens cost, context; not emitted at the default 0.30 + - docs pointers (read 2026-10-02, not restated here): https://code.claude.com/docs/en/skills#find-unused-skills and the /skill-doctor and /doctor rows on https://code.claude.com/docs/en/commands + - our route-out sends skill pruning to /skill-doctor and unused MCP servers and plugins to /doctor; our Boundary: 'If /skill-doctor is available in your session (gate basis: the records linked below), you can run `/skill-doctor` to choose which skills to turn off'; the description carries no /skill-doctor clause +- **Observation:** extraction: extracted from binary v2.1.285 on 2026-10-02 (the /harness-ops:inventory --binary-only extraction of the installed native build; every lane ok, overall degraded only because 2.1.285 differs from the validated 2.1.287, so counts are floors) (2026-10-02) +- **Recheck trigger:** a Claude Code release or docs change removes or renames /skill-doctor, folds it into /doctor, makes it model-invocable, changes its version or feature-flag gate, or widens it to startup measurement or a before/after comparison (verified 2026-10-02) +- **Baked:** description phrase no · Boundary section yes · Native step no · suggest sentence yes +- **Budget caveat:** the baked phrase may be dropped from the skill listing under budget pressure. It is the best available routing surface, not a guaranteed one + ### `skill-doctor` → `harness-ops:audit-skill-visibility` - **Verdict:** `complementary`: The sibling doctor row's split, narrowed to the surface that now owns the question. Built-in /skill-doctor is a one-shot report of what each loaded skill costs in context and how often it is used, so unused ones can be turned off. audit-skill-visibility answers why a skill is unseen: it reconciles three usage sources (native ~/.claude.json counters, its own JSONL store, OTEL) under a max-across-sources rule, computes an observed horizon and withholds every verdict the span cannot support, diagnoses reachability causes, and analyses listing-budget starvation. It disables nothing by contract. This row is separate from the doctor row rather than folded into it because the two surfaces carry different gates: /doctor answers to DISABLE_DOCTOR_COMMAND, /skill-doctor to a minimum version and to feature-flag fetching, so a session can resolve either, both, or neither, and each routing line needs its own presence gate. diff --git a/docs/native-surfaces/records.json b/docs/native-surfaces/records.json index ec21cd39a1..c2119892f7 100644 --- a/docs/native-surfaces/records.json +++ b/docs/native-surfaces/records.json @@ -1419,6 +1419,48 @@ }, "budget_caveat": false }, + { + "native": { + "name": "skill-doctor", + "class": "builtin-command", + "markers": [ + "gated", + "model-invocation-disabled" + ] + }, + "component": { + "plugin": "context-budget", + "skill": "audit", + "kind": "skill" + }, + "verdict": "complementary", + "integration": "suggest", + "reason": "The two meet only on skills. We send the choice of which skills to turn off to the built-in command, which the person runs; context-budget:audit measures a fresh headless session's startup payload per item, splits the built-in tool pools, and ledgers what each toggle measurably saved. User-only, so ours offers it to the person. Human-added pair: discovery scored it under the 0.30 floor. Ruled 2026-10-02 by operator direction on the orchestrator's recommendation.", + "evidence": [ + "`skill-doctor` present in the 2.1.285 extraction as builtin-command (source builtin)", + "invocation mode (2026-10-02, Claude Code 2.1.285): user-invocable only, model invocation disabled (command type `local-jsx`)", + "markers: model-invocation-disabled from the extraction; gated from the docs pointers below, since the extraction set no gated flag (the same basis as the sibling harness-ops:audit-skill-visibility row)", + "detect: origin discovered only at threshold 0.01, score 0.0729 from shared tokens cost, context; not emitted at the default 0.30", + "docs pointers (read 2026-10-02, not restated here): https://code.claude.com/docs/en/skills#find-unused-skills and the /skill-doctor and /doctor rows on https://code.claude.com/docs/en/commands", + "our route-out sends skill pruning to /skill-doctor and unused MCP servers and plugins to /doctor; our Boundary: 'If /skill-doctor is available in your session (gate basis: the records linked below), you can run `/skill-doctor` to choose which skills to turn off'; the description carries no /skill-doctor clause" + ], + "observation": { + "class": "extraction", + "detail": "extracted from binary v2.1.285 on 2026-10-02 (the /harness-ops:inventory --binary-only extraction of the installed native build; every lane ok, overall degraded only because 2.1.285 differs from the validated 2.1.287, so counts are floors)", + "date": "2026-10-02" + }, + "recheck": { + "trigger": "a Claude Code release or docs change removes or renames /skill-doctor, folds it into /doctor, makes it model-invocable, changes its version or feature-flag gate, or widens it to startup measurement or a before/after comparison", + "verified": "2026-10-02" + }, + "baked": { + "description_phrase": false, + "boundary_section": true, + "native_step": false, + "suggest_sentence": true + }, + "budget_caveat": true + }, { "native": { "name": "recap", diff --git a/plugins/context-budget/.claude-plugin/plugin.json b/plugins/context-budget/.claude-plugin/plugin.json index 28766085ab..5359999eb3 100644 --- a/plugins/context-budget/.claude-plugin/plugin.json +++ b/plugins/context-budget/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "context-budget", - "version": "0.7.3", + "version": "0.7.4", "description": "Measure a Claude Code session's fixed startup context payload per item, on the consumer's machine at a pinned, version-stamped binary, including per-tool attribution of the built-in tool pools that /context reports only as lump sums, derived live by A/B bare-name-deny differencing with enforced comparability rules (skill-listing signature, one mode, one binary), an SDK-primary exact meter degrading to a version-aware headless /context parser and then to an honest structured error (never a wrong number), and a per-project measure-toggle-remeasure ledger under the plugin data directory recording every lever's real before/after delta. Report-only by default; `fix` applies one approved project-scope trim.", "author": { "name": "Melodic Software", diff --git a/plugins/context-budget/CHANGELOG.md b/plugins/context-budget/CHANGELOG.md index b8cee32d7b..f17292f2a1 100644 --- a/plugins/context-budget/CHANGELOG.md +++ b/plugins/context-budget/CHANGELOG.md @@ -7,6 +7,21 @@ adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). Versions 0.6.38 and 0.6.40 were reserved by parallel changes and never published. +## [0.7.4] - 2026-10-02 + +### Added + +- **`audit` carries a Boundary section for the built-in `/skill-doctor` command.** The command is + user-only, so the section offers it to the person for choosing which skills to turn off and keeps + measuring what a toggle saved here. Its four-part records live in + `reference/native-skill-doctor.md`. + +### Changed + +- **`audit`'s route-out sends skill pruning to `/skill-doctor`.** Unused MCP servers and plugins + stay with the bundled `/doctor`. The README's Boundaries list, the lever catalogue's routes and + the report's Routes section name the same split. + ## [0.7.3] - 2026-10-02 ### Changed diff --git a/plugins/context-budget/README.md b/plugins/context-budget/README.md index 1ec9035550..0f491bca6b 100644 --- a/plugins/context-budget/README.md +++ b/plugins/context-budget/README.md @@ -125,6 +125,9 @@ passed. - Usage-based removal ("which plugins do I never use") belongs to the bundled `/doctor`; the skill routes there and never reimplements it. +- Which skills to turn off goes to the built-in `/skill-doctor`, which the person runs; the audit measures what a toggle saved and never picks the skill. + Pointer: . As of 2026-10-02. + Recheck when that section sends the question to another command. - Per-skill / per-agent / per-MCP-tool attribution belongs to `/context` natively. - Live in-session occupancy zones belong to the `context-guard` plugin. - Measurements describe **headless** sessions of the **local CLI**; interactive sessions and diff --git a/plugins/context-budget/skills/audit/SKILL.md b/plugins/context-budget/skills/audit/SKILL.md index 1a862ed937..7dd4a6c00a 100644 --- a/plugins/context-budget/skills/audit/SKILL.md +++ b/plugins/context-budget/skills/audit/SKILL.md @@ -37,12 +37,13 @@ Two rules govern everything this skill says, per the plugin's ## Scope boundary (route out) -- Unused skills/plugins/MCP servers by usage history → the bundled `/doctor`, which finds unused - skills, MCP servers, and plugins against their context cost and asks for confirmation before - changing anything. Tell the operator to run it themselves; never reimplement its checks. - Verified 2026-09-06 against Claude Code 2.1.263 and the commands reference - (, the `/doctor` row). Recheck when that row stops - naming the unused-component check, or when a release note names `/doctor`. +- Which skills to turn off → the built-in `/skill-doctor`; unused MCP servers and plugins → the + bundled `/doctor`. Tell the operator to run either one themselves; + never reimplement their checks, and never decide here what to turn off. Pointers: for + `/skill-doctor`, see ; for `/doctor`, + see the `/doctor` row on . As of 2026-10-02. Recheck + when either changes which command it sends that question to, or when a release note names + `/skill-doctor` or `/doctor`. - Per-skill / per-agent / per-MCP-tool attribution → `/context` natively, which the person runs. - Live in-session occupancy over time → the `context-guard` plugin, if installed. - Settings correctness, permission-rule state → the `harness-config` plugin, if installed. @@ -103,6 +104,28 @@ fetched 2026-09-30. As of 2026-09-30. Recheck when a release renames or removes changes its gate, or makes it model-invocable. The remaining records live in [reference/native-context.md](reference/native-context.md). +## Boundary, the built-in `skill-doctor` command + +"Which of my skills cost context" can land on either surface: + +- **`/skill-doctor` (built-in command, user-only).** We send the choice of which skills to turn + off there; the person runs it. +- **This skill (marketplace plugin).** Measures a fresh headless session's startup payload per + item, splits the built-in tool pools, and ledgers the measured delta of each toggle, including + what turning a skill off actually saved. + +**Routing.** When the person asks which skills to turn off, offer it to the person: +If /skill-doctor is available in your session (gate basis: the records linked below), you can run +`/skill-doctor` to choose which skills to turn off. Prefer this skill to measure what a toggle +saved. An unattended run records the offer in its output instead of asking. + +**Mutation gate.** This skill never runs `/skill-doctor` and never turns a skill off on its +behalf; it stays read-only unless `fix` is passed. + +**Availability is never assumed.** This section states what to do when the person can run the +command, never that it is present; its gate is read live at the pointer. The four-part records +live in [reference/native-skill-doctor.md](reference/native-skill-doctor.md). + ## Declared scope This skill measures **the local Claude Code CLI, in a headless session**. On cloud or web surfaces diff --git a/plugins/context-budget/skills/audit/reference/levers.json b/plugins/context-budget/skills/audit/reference/levers.json index d04b5c3e13..2539a9abf5 100644 --- a/plugins/context-budget/skills/audit/reference/levers.json +++ b/plugins/context-budget/skills/audit/reference/levers.json @@ -15,7 +15,8 @@ "unverified-undocumented": "Detected or rumored but not backed by current official documentation; reported if detected, never recommended." }, "routes": { - "unused components by usage history": "the bundled /doctor (operator-run; disableModelInvocation)", + "unused skills by usage history": "the built-in /skill-doctor (operator-run)", + "unused components by usage history": "the bundled /doctor for MCP servers and plugins (operator-run; disableModelInvocation)", "CLAUDE.md and memory-file content": "harness-memory / harness-config:audit-instructions, when installed", "context-injecting hooks": "the hook classification rubric in the consuming marketplace's plugin philosophy, or harness-config:audit-instructions when installed", "live in-session occupancy": "the context-guard plugin, when installed" diff --git a/plugins/context-budget/skills/audit/reference/native-skill-doctor.md b/plugins/context-budget/skills/audit/reference/native-skill-doctor.md new file mode 100644 index 0000000000..d58c4a86dc --- /dev/null +++ b/plugins/context-budget/skills/audit/reference/native-skill-doctor.md @@ -0,0 +1,22 @@ +# The built-in `/skill-doctor` command: verification record + +Detail behind the `skill-doctor` `## Boundary` section in [SKILL.md](../SKILL.md). Each row is a +four-part record: our decision or our probe's observation, the pointer or probe it rests on, the +date it was derived, and the event that makes it worth deriving again. No row restates an upstream +page; read the specific live at the pointer. Nothing here asserts the command is present in any +session. + +| Decision or observation | Pointer or probe | As of | Recheck when | +|---|---|---|---| +| Our probe found `/skill-doctor` as a built-in command, user-invocable, with model invocation disabled (command type `local-jsx`); the probe set no `gated` flag for it | Probe: the `/harness-ops:inventory --binary-only` extraction of the installed 2.1.285 binary, `builtin_commands` lane | 2026-10-02 | A release renames or removes the command, or makes it model-invocable | +| We send the choice of which skills to turn off to `/skill-doctor`, and never make that choice here | For skill pruning, see | 2026-10-02 | That section sends the question to another command, or the command's job there changes | +| We treat it as gated and never assert it present; the gate is read live, not copied here | For its gate, see the `/skill-doctor` row on and | 2026-10-02 | Either page changes the command's version floor or flag requirement | +| We keep unused MCP servers and plugins routed to the bundled `/doctor` | For `/doctor`, see its row on | 2026-10-02 | That row drops its unused-component check, or a release note names `/doctor` | + +## Why the verdict is complementary + +The two meet only on skills. We assign the pruning choice to the native command and keep, here, +the measurement of a fresh session's startup payload, the split of the built-in tool pools, and the +ledger of what a toggle measurably saved. The model cannot run `/skill-doctor`, so this skill +offers it to the person instead of routing to it. Recheck when the pointer section above starts +covering startup measurement or a before/after comparison. diff --git a/plugins/context-budget/skills/audit/reference/report.md b/plugins/context-budget/skills/audit/reference/report.md index 68a7bf76a8..7c39ea3769 100644 --- a/plugins/context-budget/skills/audit/reference/report.md +++ b/plugins/context-budget/skills/audit/reference/report.md @@ -53,8 +53,8 @@ mode; the displayed fraction in cli-parse mode). category, `removes-weight` first. Postures bind: `never-recommend` rows appear under a "priced, not recommended" heading; `report-only` vendor weight closes the group as the honest floor. -6. **Routes.** The catalogue's route-outs (`/doctor` for usage-based removal, which the operator - runs; memory files, hooks, live occupancy to their owners), each in one line. +6. **Routes.** The catalogue's route-outs (`/skill-doctor` for unused skills and `/doctor` for + unused MCP servers and plugins, both of which the operator runs; memory files, hooks, live occupancy to their owners), each in one line. 7. **Degradations and caveats.** Every `caveats[]` entry from the records used, plus anything the engine could not measure and why. The attribution record already merges the baseline with each deny run and the combined additivity run, so a deny-run disclosure is in that list and From 276c5fdde0fb0e0b3b6c9177b334f5b675059745 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Fri, 2 Oct 2026 02:14:09 -0400 Subject: [PATCH 06/11] docs(playbooks): carry failures back into skills and allow old-patterns tables The skill-authoring guidance now suggests an eval run when a skill's description or body changes, and says how to learn from a failure: read it in full, write back the general cause, never copy case text into the skill, and never draw from held-back test cases. It adds that whatever an eval case depends on must sit in the hub SKILL.md, since an eval run reads only the hub. Time-sensitive content allows a names-only old-patterns table per the upstream-drift convention, and no longer names a rule file as the owner of the in-body history ban, since none holds it. Co-Authored-By: Claude Opus 5.5 --- plugins/playbooks/.claude-plugin/plugin.json | 2 +- plugins/playbooks/CHANGELOG.md | 15 +++++ .../reference/authoring-checklist.md | 2 +- .../reference/authoring-guidance.md | 65 ++++++++++++++----- 4 files changed, 65 insertions(+), 19 deletions(-) diff --git a/plugins/playbooks/.claude-plugin/plugin.json b/plugins/playbooks/.claude-plugin/plugin.json index e084429960..a7815e2fd5 100644 --- a/plugins/playbooks/.claude-plugin/plugin.json +++ b/plugins/playbooks/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "playbooks", - "version": "0.17.1", + "version": "0.17.2", "description": "Doctrine and knowledge playbooks as on-demand skills, repo-sweep for running a catalog of hygiene skills through a repository one commit per step, plus a maintainer-facing update skill. boris carries Boris Cherny's Claude Code workflow tips (howborisusesclaudecode.com), skill-authoring carries Anthropic's internal skill-authoring playbook, and fable-5 carries Claude Fable 5's operating doctrine (self-authored, no upstream). The boris and skill-authoring packs vendor a verbatim upstream baseline; /playbooks:update drift-checks and syncs those baselines centrally (maintainers).", "author": { "name": "Melodic Software", diff --git a/plugins/playbooks/CHANGELOG.md b/plugins/playbooks/CHANGELOG.md index 6d9547bd88..e787dee36f 100644 --- a/plugins/playbooks/CHANGELOG.md +++ b/plugins/playbooks/CHANGELOG.md @@ -4,6 +4,21 @@ All notable changes to the `playbooks` plugin are recorded here. The `version` i `.claude-plugin/plugin.json` is the delivery vehicle. A consumer receives a change only after that version increases. +## [0.17.2] - 2026-10-02 + +### Changed + +- `skill-authoring`'s time-sensitive guidance allows one exception to the in-body history ban: an + "Old patterns" section holding a names-only table of old-to-current names, under the + upstream-drift convention's old-patterns carve-out. The ban no longer names + `.claude/rules/skill-bodies-state-current-rules.md` as its owner, since that rule does not hold + it; the pre-share checklist row says the same. +- `skill-authoring`'s evaluation guidance says how to carry a failure back (read the whole failure, + write the general cause in your own words, never copy case text, never draw on held-back test + cases), to re-run the evals when a skill's description or body changes, and to keep everything a + `claude plugin eval` case depends on in the hub `SKILL.md`, with a pointer to the evals plugin's + record. + ## [0.17.1] - 2026-10-01 ### Changed diff --git a/plugins/playbooks/skills/skill-authoring/reference/authoring-checklist.md b/plugins/playbooks/skills/skill-authoring/reference/authoring-checklist.md index cbdede4117..87eaa3a263 100644 --- a/plugins/playbooks/skills/skill-authoring/reference/authoring-checklist.md +++ b/plugins/playbooks/skills/skill-authoring/reference/authoring-checklist.md @@ -34,7 +34,7 @@ have checked a judgment row is misreporting. | A gotchas surface exists (`## Gotchas` inline or a gotchas spoke) | mechanical (check 11) | | `## Next` is present and names the successor in mention-only form | judgment | | Arguments follow the skill argument shape: one action first, earned `--flag` modifiers, at most one subject last, `argument-hint` in the same order ([`authoring-guidance.md`](authoring-guidance.md#argument-surface)) | judgment | -| No date-conditional guidance; history lives in CHANGELOG, commit, or ADR; no upstream text is restated, and a volatile specific the body depends on is our decision plus a pointer to the exact section, an as-of date, and a recheck trigger | judgment | +| No date-conditional guidance; history lives in CHANGELOG, commit, or ADR, apart from a names-only "Old patterns" table ([`authoring-guidance.md`](authoring-guidance.md#time-sensitive-content)); no upstream text is restated, and a volatile specific the body depends on is our decision plus a pointer to the exact section, an as-of date, and a recheck trigger | judgment | | Every item of the upstream checklist (Pointer below) that this file does not sharpen holds | judgment | | Each spoke pointer says what the file holds and when to read it | judgment | | Freedom level chosen per section and matched to fragility | judgment | diff --git a/plugins/playbooks/skills/skill-authoring/reference/authoring-guidance.md b/plugins/playbooks/skills/skill-authoring/reference/authoring-guidance.md index d2fadb9dc5..c367090659 100644 --- a/plugins/playbooks/skills/skill-authoring/reference/authoring-guidance.md +++ b/plugins/playbooks/skills/skill-authoring/reference/authoring-guidance.md @@ -241,18 +241,26 @@ example. No date-conditional guidance in a body ("before August, use the old API"): state the current method only. This marketplace does not keep superseded guidance in an in-body section, collapsed -or not. History routes to the plugin `CHANGELOG.md`, the commit message, and `docs/adr/`, and a -volatile specific the body depends on is replaced by a links-only record: our decision in our -words, a pointer to the exact upstream section, an as-of date, and a recheck trigger, with no -upstream text. The reason is the cost model above: a collapsed block is still tokens on every turn -after invocation, while a separate reference file is free until read, so history that must travel -with the skill goes in a spoke. The owning rule is -`.claude/rules/skill-bodies-state-current-rules.md`, and the record shape is the upstream-drift -convention's `docs/conventions/upstream-drift/README.md#required-parts`. - -**Record.** Pointer: for the page's treatment of superseded guidance, see +or not, apart from the one exception below. History routes to the plugin `CHANGELOG.md`, the +commit message, and `docs/adr/`, and a volatile specific the body depends on is replaced by a +links-only record: our decision in our words, a pointer to the exact upstream section, an as-of +date, and a recheck trigger, with no upstream text. The reason is the cost model above: a collapsed +block is still tokens on every turn after invocation, while a separate reference file is free until +read, so history that must travel with the skill goes in a spoke. The record shape is the +upstream-drift convention's `docs/conventions/upstream-drift/README.md#required-parts`, and +`.claude/rules/skill-bodies-state-current-rules.md` applies it to skill and agent bodies. + +The exception: a skill whose users still meet renamed API or interface names may keep an "Old +patterns" section holding a table that maps each old name to its current one. The table carries +names only, never a description of behavior, and the section ends in its own record. Where the +section may sit and what it may hold are owned by +`docs/conventions/upstream-drift/README.md#old-patterns-mapping-tables`; read the limits there. + +**Record.** Pointer: for the page's treatment of superseded guidance and of an "Old patterns" +section, see [Avoid time-sensitive information](https://platform.claude.com/docs/en/agents-and-tools/agent-skills/best-practices#avoid-time-sensitive-information). -As of: 2026-09-10. Recheck trigger: the page drops or changes that section. +As of: 2026-10-02. Recheck trigger: the page drops or changes that section; then re-read the +upstream-drift carve-out, which depends on the same section. ## Evaluation and iteration @@ -277,7 +285,18 @@ it or signal it better). Where the bundled skill-creator plugin is installed, it this loop with a subagent per case. Use `/skill-doctor`, where it resolves, to answer "does it activate" from usage data, never "is the output right"; its version floor and availability are at the pointer. Run this loop for a skill's eval file; route a plugin measured as a plugin to -`claude plugin eval`, whose case format is separate from `evals/evals.json`. +`claude plugin eval`, whose case format is separate from `evals/evals.json`. For a +`claude plugin eval` case, put everything the case depends on in the hub `SKILL.md`, not a spoke. + +Carrying a failure back: read the whole failure first (the case's input, the output, and the +grader's failure text), then write the general cause into the skill in your own words. Never copy +a case's prompt, output, or distinctive phrasing into the skill, and never draw a change from +held-back test cases. + +Re-run the evals whenever a skill's description or body changes: a description change re-measures +triggering (`/skill-quality:check measure-invocation`), a body change re-measures output (the loop +above, or `/evals:plugin-eval` for a plugin suite). Runs are on demand; no CI workflow in this +marketplace runs model-graded evals. When a rule is being missed, try both directive wording (a capitalized must) and reasoning-based wording (the rule plus the reason it exists). The evals settle it. @@ -289,11 +308,23 @@ for the eval file shape, see ). As of: 2026-09-10 +(plugin-evals read 2026-09-12; hub-only record, failure carry-back and held-back pointers +2026-10-02). Recheck trigger: the plugin-evals page drops the format separation or its runner +starts reading `evals/evals.json`, the skills page changes the `/skill-doctor` gate, the loop, or +the skill-creator modes, the best-practices page changes the four signals, the runner changes its +record shape, the `/evals:plugin-eval` hub-only record's trigger fires, the agentskills.io +iterating section changes, or a commit to `anthropics/skills` changes Step 4 or the failure modes +of `eval-hillclimb.md` (then move the pin). ## Model coverage From 6ca9f6a718705d1934a1f8a679a4466769ade54d Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Fri, 2 Oct 2026 16:07:35 -0400 Subject: [PATCH 07/11] fix(evals): keep calibrate-judge writes inside --out and make review cells link-inert calibrate-judge.py build skips a case.yaml grader name that is not one path segment, so a crafted suite cannot write or overwrite files outside --out. render-review.py escapes backslashes and brackets in Markdown table cells, so case text cannot load a remote image or hide a link behind other text; a backslash before a pipe now survives. Co-Authored-By: Claude Opus 5.5 --- plugins/evals/CHANGELOG.md | 12 ++++++++---- .../skills/design/scripts/render-review.py | 6 +++--- .../design/scripts/test_render_review.py | 8 ++++++++ .../plugin-eval/scripts/calibrate-judge.py | 11 ++++++++++- .../scripts/test_calibrate_judge.py | 18 ++++++++++++++++++ 5 files changed, 47 insertions(+), 8 deletions(-) diff --git a/plugins/evals/CHANGELOG.md b/plugins/evals/CHANGELOG.md index 6f227c087e..0fe7c2541a 100644 --- a/plugins/evals/CHANGELOG.md +++ b/plugins/evals/CHANGELOG.md @@ -15,9 +15,11 @@ - **`design` routes by repository kind and has every input approved.** A Claude API app is told to type `/claude-api build-eval`; a skill or plugin repository continues here and runs through `plugin-eval`. Candidate cases render for approval through a new `scripts/render-review.py`, - Markdown by default and escaped HTML as the option. Raw transcripts stay out of cases in a public - repository or one of unknown visibility. Graders are checked against a handful of cases before - their scores are trusted. + Markdown by default and escaped HTML as the option. A Markdown table cell escapes link and image + syntax, so case text cannot load a remote image or hide a link behind other text; a bare URL + still shows as itself. Raw transcripts stay out of cases in a public repository or one of + unknown visibility. Graders are checked against a handful of cases before their scores are + trusted. - **`methodology` points at the bundled hillclimb guides.** A new `reference/hillclimb.md` links each step at a pinned commit and states only this plugin's facts; a new `reference/local-decisions.md` holds this plugin's defaults and source conflicts. The skill @@ -34,7 +36,9 @@ labelled sample, whose agent replies with the sample word for word, and scores the judge's verdicts against the labels: agreement, false positives and negatives, split votes, runs whose reply was not the sample (whitespace and bold markers aside), and samples never judged. A grader under 90% agreement prints a - `FAIL grader` line and the script exits 1. `## Calibrating a judge` gives the commands. + `FAIL grader` line and the script exits 1. A `case.yaml` grader name that is not one path + segment is skipped with a note, so a suite cannot write outside `--out`. `## Calibrating a + judge` gives the commands. - **`plugin-eval` checks a run is valid before its score counts.** A new `scripts/run-validity.py` reads `aggregate-result.json` and the kept traces and prints VALID or INVALID: an incomplete or empty run, skipped paid graders, an errored run, a row count that diff --git a/plugins/evals/skills/design/scripts/render-review.py b/plugins/evals/skills/design/scripts/render-review.py index 5ac740fcf0..10e787137f 100755 --- a/plugins/evals/skills/design/scripts/render-review.py +++ b/plugins/evals/skills/design/scripts/render-review.py @@ -56,9 +56,9 @@ def as_text(value): def md_cell(value): - """One table cell: a single line, HTML-inert, with every pipe escaped.""" - flat = " ".join(as_text(value).split()) - return html.escape(flat, quote=False).replace("|", "\\|") + """One table cell: a single line, HTML-inert, with no live link or image.""" + flat = re.sub(r"([\\\[\]|])", r"\\\1", " ".join(as_text(value).split())) + return html.escape(flat, quote=False) def longest_backtick_run(text): diff --git a/plugins/evals/skills/design/scripts/test_render_review.py b/plugins/evals/skills/design/scripts/test_render_review.py index d7a1e4c2a3..705eb6bd02 100755 --- a/plugins/evals/skills/design/scripts/test_render_review.py +++ b/plugins/evals/skills/design/scripts/test_render_review.py @@ -113,6 +113,14 @@ def test_pipes_in_cells_are_escaped_so_the_row_keeps_its_cell_count(self): self.assertIn(r"x \| y z", row) self.assertEqual(unescaped_pipes(row), unescaped_pipes(header)) + def test_link_and_image_syntax_in_cells_is_inert(self): + cases = [{"id": 1, "name": "![](https://x.example/p.png) [ok](https://y.example) a\\|b"}] + code, text, _ = run(cases, "markdown") + self.assertEqual(code, 0) + _, _, row = table_rows(text) + self.assertIn(r"!\[\](https://x.example/p.png) \[ok\](https://y.example)", row) + self.assertIn(r"a\\\|b", row) + def test_one_fenced_block_per_case_longer_than_any_backtick_run(self): tricky = "before\n`````\n\nafter" cases = [{"id": 1, "prompt": tricky}, {"id": 2, "input": "plain"}] diff --git a/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py index ea894a07c2..4b438e1e6f 100755 --- a/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py +++ b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py @@ -68,6 +68,7 @@ import importlib.util import json import os +import re import sys import unicodedata @@ -79,6 +80,7 @@ HERE = os.path.dirname(os.path.abspath(__file__)) VALIDATOR = os.path.join(HERE, "..", "..", "validate", "scripts", "validate-cases.py") +SAFE_NAME = re.compile(r"[A-Za-z0-9._-]+") SKIP_DIRS = frozenset(["results", "mocks", "graders", "samples", "__pycache__"]) REPLY_FOCUSES = (None, "last_message") # the judge reads the final reply TRACE_FOCUS = "trace" @@ -204,12 +206,19 @@ def llm_graders(case_dir, validator, notes, case): data = validator.parse_yaml(read_text(yaml_path)) for entry in data.get("graders") or []: if isinstance(entry, dict) and entry.get("type") == "llm": + name = str(entry.get("name")) + if not SAFE_NAME.fullmatch(name) or name in (".", ".."): + notes.append( + "skip %s/case.yaml grader %r: a grader name must be one path " + "segment of letters, digits, '.', '_' or '-'" % (case, name) + ) + continue lines = ["---", "type: llm"] for key in ("focus", "weight", "arm"): if isinstance(entry.get(key), (str, int, float)): lines.append("%s: %s" % (key, entry[key])) body = "\n".join(lines + ["---", "", str(entry.get("criteria", ""))]) - found.append((str(entry.get("name")), entry.get("focus"), body + "\n")) + found.append((name, entry.get("focus"), body + "\n")) grader_dir = os.path.join(case_dir, "graders") for filename in sorted(os.listdir(grader_dir)) if os.path.isdir(grader_dir) else []: if not filename.endswith(".md"): diff --git a/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py b/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py index 7b8d775fea..500a4b1dd2 100755 --- a/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py +++ b/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py @@ -317,6 +317,24 @@ def test_a_case_yaml_grader_and_prompt_are_read(self): grader, "---\ntype: llm\n---\n\nPASS if the answer names Paris.\n" ) + def test_a_case_yaml_grader_name_that_is_not_one_path_segment_is_skipped(self): + case = self.dir / "capital-yaml" + case.mkdir() + names = ["a/../../../CLAUDE", "..", "a\\b", "names-paris"] + (case / "case.yaml").write_text( + 'schema_version: "1.1"\nname: capital-yaml\ngraders:\n' + + "".join( + ' - name: "%s"\n type: llm\n criteria: "PASS."\n' + % name.replace("\\", "\\\\") + for name in names + ) + ) + notes = [] + found = calibrate.llm_graders(case, validate_cases, notes, "capital-yaml") + self.assertEqual([name for name, _, _ in found], ["names-paris"]) + self.assertEqual(len(notes), 3) + self.assertTrue(all("one path segment" in note for note in notes)) + def test_this_plugins_own_suite_builds_and_validates(self): proc, out = self.build(PLUGIN_SUITE) self.assertEqual(proc.returncode, 0, proc.stdout + proc.stderr) From a5fa3c963e46920861afae649e809738822eef6f Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Fri, 2 Oct 2026 17:10:21 -0400 Subject: [PATCH 08/11] test(scripts): cover the multi-agent config-root copy in the sync test sync-config-root.sh gained plugins/multi-agent as a carrier, but the test's fixture list did not, so the sync step failed to copy into a directory the fixture never created and three cases failed. Co-Authored-By: Claude Opus 5.5 --- scripts/sync-config-root.test.sh | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/sync-config-root.test.sh b/scripts/sync-config-root.test.sh index dd5f23fff7..7b14a2badd 100755 --- a/scripts/sync-config-root.test.sh +++ b/scripts/sync-config-root.test.sh @@ -18,6 +18,7 @@ sync_cluster_suite::run \ --copy 'plugins/ai-slop/lib/config-root.sh' \ --extra-copy 'plugins/attribution/lib/config-root.sh' \ --extra-copy 'plugins/docs-naming/lib/config-root.sh' \ + --extra-copy 'plugins/multi-agent/lib/config-root.sh' \ --v1 'config_root_classify() { echo repo; }\n' \ --v2 'config_root_classify() { echo repo; }\nconfig_root_resolve() { echo /r; }\n' \ --drift 'config_root_classify() { echo home; }\n' \ From 8c792ebc73023ffef37474d90b67aa9dc8700aac Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Fri, 2 Oct 2026 17:11:32 -0400 Subject: [PATCH 09/11] fix(evals): use American spelling and portable example paths typos (locale en-us) and the machine-specific-paths check both failed on the evals plugin once the PR left draft. Respell labelled, behaviour, artefact, rigour and penalised throughout, rename the interval-setting-scope grader to expected-behavior, rename the unreleased option to labeled_grader_check, and swap the /home/dev/ example paths in sample answers for /srv/. No label, expected value or rubric criterion changes; the touched graders are recalibrated before the next full run. Co-Authored-By: Claude Opus 5.5 --- plugins/evals/.claude-plugin/plugin.json | 4 ++-- plugins/evals/CHANGELOG.md | 6 +++--- plugins/evals/README.md | 4 ++-- .../samples/criteria-first.json | 2 +- .../samples/grading-choice.json | 4 ++-- ...ected-behaviour.md => expected-behavior.md} | 0 .../evals/interval-setting-scope/prompt.md | 2 +- ...d-behaviour.json => expected-behavior.json} | 4 ++-- .../samples/no-number-from-invalid.json | 2 +- .../samples/four-properties.json | 4 ++-- .../samples/skipped-judge-is-failure.json | 4 ++-- .../reference-read-denied/samples/cause.json | 6 +++--- .../graders/routes-to-build-eval.md | 2 +- .../samples/routes-to-build-eval.json | 18 +++++++++--------- .../samples/token-build-eval.json | 2 +- .../samples/target-first.json | 2 +- .../evals/evals/worktree-guard-stop/prompt.md | 2 +- .../samples/no-way-around.json | 10 +++++----- .../samples/paste-outside.json | 8 ++++---- .../samples/check-and-rerun.json | 4 ++-- .../samples/not-marked-partial.json | 2 +- plugins/evals/skills/design/SKILL.md | 2 +- .../methodology/reference/local-decisions.md | 2 +- plugins/evals/skills/plugin-eval/SKILL.md | 2 +- .../plugin-eval/scripts/calibrate-judge.py | 8 ++++---- .../scripts/test_calibrate_judge.py | 6 +++--- plugins/evals/skills/validate/SKILL.md | 2 +- 27 files changed, 57 insertions(+), 57 deletions(-) rename plugins/evals/evals/interval-setting-scope/graders/{expected-behaviour.md => expected-behavior.md} (100%) rename plugins/evals/evals/interval-setting-scope/samples/{expected-behaviour.json => expected-behavior.json} (89%) diff --git a/plugins/evals/.claude-plugin/plugin.json b/plugins/evals/.claude-plugin/plugin.json index 002af33aeb..679923f589 100644 --- a/plugins/evals/.claude-plugin/plugin.json +++ b/plugins/evals/.claude-plugin/plugin.json @@ -55,9 +55,9 @@ "description": "When on, /evals:plugin-eval's preflight warns when the tested model and the judge model resolve to the same model, and /evals:design repeats the reminder beside the build-eval route.", "default": true }, - "labelled_grader_check": { + "labeled_grader_check": { "type": "boolean", - "title": "Labelled-set grader check", + "title": "Labeled-set grader check", "description": "When on, /evals:design checks a grader against a set of cases you label, not only the handful-of-cases agreement check.", "default": false } diff --git a/plugins/evals/CHANGELOG.md b/plugins/evals/CHANGELOG.md index 0fe7c2541a..c5c2c200fa 100644 --- a/plugins/evals/CHANGELOG.md +++ b/plugins/evals/CHANGELOG.md @@ -25,15 +25,15 @@ `reference/local-decisions.md` holds this plugin's defaults and source conflicts. The skill routes by repository kind. - **Six settings:** `split_policy`, `interval_method`, `review_format`, `grader_run_twice`, - `same_model_warning` and `labelled_grader_check`. + `same_model_warning` and `labeled_grader_check`. - **The suite grows to 30 cases.** 21 are hard cases, each saying why it is hard; 4 are routine guards where the base model already answers well; 4 are near-miss controls that must not invoke an evals skill; and 1 is a knowledge case. `noise-before-gain` is the one a person judged hard: a model tends to take a small gain over a near-ceiling baseline at face value. -- **Every `llm` grader has labelled samples.** All 47 hold must-pass and must-fail answers in +- **Every `llm` grader has labeled samples.** All 47 hold must-pass and must-fail answers in `samples/.json`, so each rubric can be calibrated against them. - **`plugin-eval` calibrates a judge.** A new `scripts/calibrate-judge.py` builds one case per - labelled sample, whose agent replies with the sample word for word, and scores the judge's + labeled sample, whose agent replies with the sample word for word, and scores the judge's verdicts against the labels: agreement, false positives and negatives, split votes, runs whose reply was not the sample (whitespace and bold markers aside), and samples never judged. A grader under 90% agreement prints a `FAIL grader` line and the script exits 1. A `case.yaml` grader name that is not one path diff --git a/plugins/evals/README.md b/plugins/evals/README.md index fa8e7edbc1..0545b2261e 100644 --- a/plugins/evals/README.md +++ b/plugins/evals/README.md @@ -87,7 +87,7 @@ where this repository departs from a source, the record is in the methodology sk from votes the run already took. - **`same_model_warning`** (boolean, default `true`): warns when the tested model and the judge are the same model. -- **`labelled_grader_check`** (boolean, default `false`): adds a check of each grader against cases +- **`labeled_grader_check`** (boolean, default `false`): adds a check of each grader against cases you label. No hooks and no MCP servers. The methodology, design, and validate surfaces make no network calls @@ -113,7 +113,7 @@ reads it from. | `review_format` | string | `"markdown"` | `CLAUDE_PLUGIN_OPTION_REVIEW_FORMAT` | Format /evals:design renders candidate eval cases in for your approval. markdown (the default) is a table plus one fenced block per case; html is one page with every field escaped. | | `grader_run_twice` | boolean | `true` | `CLAUDE_PLUGIN_OPTION_GRADER_RUN_TWICE` | When on, /evals:plugin-eval's noise report shows how often the judge votes for each llm grader agreed, read from votes the run already took, at no extra spend. | | `same_model_warning` | boolean | `true` | `CLAUDE_PLUGIN_OPTION_SAME_MODEL_WARNING` | When on, /evals:plugin-eval's preflight warns when the tested model and the judge model resolve to the same model, and /evals:design repeats the reminder beside the build-eval route. | -| `labelled_grader_check` | boolean | `false` | `CLAUDE_PLUGIN_OPTION_LABELLED_GRADER_CHECK` | When on, /evals:design checks a grader against a set of cases you label, not only the handful-of-cases agreement check. | +| `labeled_grader_check` | boolean | `false` | `CLAUDE_PLUGIN_OPTION_LABELED_GRADER_CHECK` | When on, /evals:design checks a grader against a set of cases you label, not only the handful-of-cases agreement check. | ### How to set these diff --git a/plugins/evals/evals/design-criteria-first/samples/criteria-first.json b/plugins/evals/evals/design-criteria-first/samples/criteria-first.json index f9304ed5cd..981d83e2c1 100644 --- a/plugins/evals/evals/design-criteria-first/samples/criteria-first.json +++ b/plugins/evals/evals/design-criteria-first/samples/criteria-first.json @@ -31,7 +31,7 @@ "why": "F-retract: sets measurable criteria and asks, then says to skip the criteria step and writes cases first." }, { - "answer": "Start with success criteria, then let tooling build the suite. Proposed targets for you to confirm:\n\n- 90% or more of drafts judged correct against a per-email rubric.\n- Tone averaging at least 4 on a 1 to 5 scale.\n- 95th-percentile draft time under 8 seconds.\n\nTwo questions: what does a good draft mean to your agents, and do you have a baseline, such as today's rewrite rate?\n\nOnce the targets are agreed, type `/claude-api build-eval` in Claude Code. It scaffolds the labelled inputs and graders for an LLM app like this drafter, and you can follow up with `/claude-api hillclimb` to tune the prompt against the suite. Keep the targets in a short criteria file so every later run is judged against the same bar.", + "answer": "Start with success criteria, then let tooling build the suite. Proposed targets for you to confirm:\n\n- 90% or more of drafts judged correct against a per-email rubric.\n- Tone averaging at least 4 on a 1 to 5 scale.\n- 95th-percentile draft time under 8 seconds.\n\nTwo questions: what does a good draft mean to your agents, and do you have a baseline, such as today's rewrite rate?\n\nOnce the targets are agreed, type `/claude-api build-eval` in Claude Code. It scaffolds the labeled inputs and graders for an LLM app like this drafter, and you can follow up with `/claude-api hillclimb` to tune the prompt against the suite. Keep the targets in a short criteria file so every later run is judged against the same bar.", "why": "F-near: criteria and questions are right, but it routes this Ollama/Llama app (which does not call Claude) to /claude-api build-eval, which the FAIL list names." } ] diff --git a/plugins/evals/evals/grading-method-choice/samples/grading-choice.json b/plugins/evals/evals/grading-method-choice/samples/grading-choice.json index a7d60004fa..cd2ac1ee23 100644 --- a/plugins/evals/evals/grading-method-choice/samples/grading-choice.json +++ b/plugins/evals/evals/grading-method-choice/samples/grading-choice.json @@ -5,8 +5,8 @@ "answer": "**Criterion 1, category.** Grade it with code. Parse the reply's JSON, read `category`, and compare it with the labeled value by exact match. Count a reply that doesn't parse as a fail.\n\n**Criterion 2, patient tone.** Use an LLM grader with a rubric that defines patient in observable terms: acknowledges the problem, doesn't rush the customer, no curt or blaming phrasing. Ask it for a pass or fail verdict.\n\n**Order of preference in general:** code-based grading first, whenever the output can be constrained enough to check mechanically, because it is fast, cheap and repeatable. Then an LLM grader for judgment calls. Human grading last, as the slowest and least scalable option.\n\n**Making an LLM grader trustworthy:**\n- Give it a clear, detailed rubric and a constrained output, such as pass/fail or a 1-5 scale.\n- Grade with a different model from the one that produced the replies.\n- Before relying on it, read a sample of its verdicts and compare them with your own judgment or human labels, and repeat after any rubric change." }, { - "why": "P-alt: prose, general rule first: deterministic check in code, then a model-based grader, people as last resort; category asserted in code, patience to a rubric-driven model grader; trust via fixed verdict set, different model, agreement with human-labelled replies.", - "answer": "Start with the general rule, because it shapes the other answers: pick the fastest, most reliable, most scalable method that can grade the criterion. That means a deterministic check in code where possible, a model-based grader where judgment is needed, and people only as a last resort.\n\nApplied here, the `category` field is a deterministic check. Load the reply's JSON in your harness and assert that `category` equals the label; no model is involved. Patience is a judgment, so it goes to a model-based grader working from a written rubric that describes what a patient reply does and does not do.\n\nA model grader earns trust in three ways. Its rubric is explicit and its answer is limited to a fixed set, so verdicts parse and stay consistent. It runs on a different model from the assistant, so it isn't marking its own work. And you have compared a batch of its verdicts with human-labelled replies and found that they agree before you rely on its scores." + "why": "P-alt: prose, general rule first: deterministic check in code, then a model-based grader, people as last resort; category asserted in code, patience to a rubric-driven model grader; trust via fixed verdict set, different model, agreement with human-labeled replies.", + "answer": "Start with the general rule, because it shapes the other answers: pick the fastest, most reliable, most scalable method that can grade the criterion. That means a deterministic check in code where possible, a model-based grader where judgment is needed, and people only as a last resort.\n\nApplied here, the `category` field is a deterministic check. Load the reply's JSON in your harness and assert that `category` equals the label; no model is involved. Patience is a judgment, so it goes to a model-based grader working from a written rubric that describes what a patient reply does and does not do.\n\nA model grader earns trust in three ways. Its rubric is explicit and its answer is limited to a fixed set, so verdicts parse and stay consistent. It runs on a different model from the assistant, so it isn't marking its own work. And you have compared a batch of its verdicts with human-labeled replies and found that they agree before you rely on its scores." }, { "why": "P-caveat: all three points (code for category, rubric LLM grader for tone, order code/LLM/human, trust rules); humans appear only as a calibration set and for a hypothetical future criterion, which keeps human grading last.", diff --git a/plugins/evals/evals/interval-setting-scope/graders/expected-behaviour.md b/plugins/evals/evals/interval-setting-scope/graders/expected-behavior.md similarity index 100% rename from plugins/evals/evals/interval-setting-scope/graders/expected-behaviour.md rename to plugins/evals/evals/interval-setting-scope/graders/expected-behavior.md diff --git a/plugins/evals/evals/interval-setting-scope/prompt.md b/plugins/evals/evals/interval-setting-scope/prompt.md index fb162c1bc2..525ad00ac0 100644 --- a/plugins/evals/evals/interval-setting-scope/prompt.md +++ b/plugins/evals/evals/interval-setting-scope/prompt.md @@ -1,5 +1,5 @@ --- -description: "Hard case. Why hard: The user reads a by-design behaviour as a bug; only this plugin says its interval setting changes the pass-count line alone while score intervals and the paired delta stay normal. Self-reference: the setting and the report are this plugin's." +description: "Hard case. Why hard: The user reads a by-design behavior as a bug; only this plugin says its interval setting changes the pass-count line alone while score intervals and the paired delta stay normal. Self-reference: the setting and the report are this plugin's." tags: [reading-results, local-defaults, hard, self-reference] runs: 3 max_turns: 10 diff --git a/plugins/evals/evals/interval-setting-scope/samples/expected-behaviour.json b/plugins/evals/evals/interval-setting-scope/samples/expected-behavior.json similarity index 89% rename from plugins/evals/evals/interval-setting-scope/samples/expected-behaviour.json rename to plugins/evals/evals/interval-setting-scope/samples/expected-behavior.json index 56e8ded06b..f87854f31a 100644 --- a/plugins/evals/evals/interval-setting-scope/samples/expected-behaviour.json +++ b/plugins/evals/evals/interval-setting-scope/samples/expected-behavior.json @@ -10,7 +10,7 @@ }, { "answer": "That is expected, not a bug. `interval_method` changes only the pass-count line (cases at or above the threshold); the score intervals and the delta line stay normal by design.\n\nOne thing to expect on the pass-count line itself: Wilson and normal intervals differ most when the pass count is near zero or near all cases, or the suite is small. In the middle of the range they can look almost the same, so even there a small visual difference is normal.", - "why": "P-caveat: correct scope of the setting plus a separate note on when Wilson and normal intervals visibly differ on the pass-count line; it does not call the delta behaviour a bug." + "why": "P-caveat: correct scope of the setting plus a separate note on when Wilson and normal intervals visibly differ on the pass-count line; it does not call the delta behavior a bug." }, { "answer": "It isn't ignored. By design, `interval_method` only affects the pass-count line, which counts the cases at or above the threshold. Every other interval, the delta line included, stays normal no matter what you choose, so what you're seeing is correct. The pass-count line is where the Wilson interval appears; the delta and the per-arm score intervals use the normal interval whichever method you pick. Leave the setting as it is, and look at the pass-count line to see it working.", @@ -23,7 +23,7 @@ "why": "F-wrong: says the setting is ignored, recommends restarting and re-setting, and says the delta should use Wilson." }, { - "answer": "No, this is expected behaviour, not a bug. The delta line in the noise report is always built with a normal interval, whatever `interval_method` is set to, and the per-arm score intervals are normal too. Your `wilson` setting is still being read; it just applies to a different part of the report than the delta. There's nothing to restart or change. Regenerating the report will show the delta's bounds unchanged; that is the intended output, so read the delta as usual.", + "answer": "No, this is expected behavior, not a bug. The delta line in the noise report is always built with a normal interval, whatever `interval_method` is set to, and the per-arm score intervals are normal too. Your `wilson` setting is still being read; it just applies to a different part of the report than the delta. There's nothing to restart or change. Regenerating the report will show the delta's bounds unchanged; that is the intended output, so read the delta as usual.", "why": "F-partial: says it is expected and the delta and score intervals stay normal, but never says the setting changes only the pass-count line." }, { diff --git a/plugins/evals/evals/invalid-gate-no-number/samples/no-number-from-invalid.json b/plugins/evals/evals/invalid-gate-no-number/samples/no-number-from-invalid.json index a5ee7b51bf..9b189f72aa 100644 --- a/plugins/evals/evals/invalid-gate-no-number/samples/no-number-from-invalid.json +++ b/plugins/evals/evals/invalid-gate-no-number/samples/no-number-from-invalid.json @@ -31,7 +31,7 @@ "why": "F-retract: states the no-number rule, then says to post +0.30 now with a footnote." }, { - "answer": "Not as a headline result. The INVALID comes from missing traces: the gate reads each run's trace, and without `--keep-temp` the runner removed them, so those checks couldn't run.\n\nWhat I'd do is keep +0.30 out of the summary line and put it in a clearly labelled thread reply: \"unvalidated: +0.30, gate INVALID (trace checks unchecked), rerun with --keep-temp in progress.\" That way nobody mistakes it for a validated result, and the team still sees where things stand. Then rerun with `--keep-temp`, gate the new file, and replace the unvalidated figure once it prints `verdict: VALID`.", + "answer": "Not as a headline result. The INVALID comes from missing traces: the gate reads each run's trace, and without `--keep-temp` the runner removed them, so those checks couldn't run.\n\nWhat I'd do is keep +0.30 out of the summary line and put it in a clearly labeled thread reply: \"unvalidated: +0.30, gate INVALID (trace checks unchecked), rerun with --keep-temp in progress.\" That way nobody mistakes it for a validated result, and the team still sees where things stand. Then rerun with `--keep-temp`, gate the new file, and replace the unvalidated figure once it prints `verdict: VALID`.", "why": "F-near: posts the +0.30 under an \"unvalidated\" label, which the FAIL list names, even though the rerun advice is right." } ] diff --git a/plugins/evals/evals/measurable-criterion/samples/four-properties.json b/plugins/evals/evals/measurable-criterion/samples/four-properties.json index ff0891c571..8939f07534 100644 --- a/plugins/evals/evals/measurable-criterion/samples/four-properties.json +++ b/plugins/evals/evals/measurable-criterion/samples/four-properties.json @@ -10,7 +10,7 @@ }, { "why": "P-caveat: all four properties (specific failures, rates over 2,000 conversations, targets set relative to the current bot's measured rate, intake scope), plus a caveat that grader validation checks the measurement, not the target.", - "answer": "Rewritten: \"On 2,000 synthetic intake conversations built from real triage scenarios, the share of chatbot replies that give an unverified medication dose or a diagnosis, as flagged by a rubric-based grader, falls by at least half from the current bot's rate on the same set, and the share of conversations that mention chest pain, stroke signs or suicidal thoughts that are escalated to a human in the same turn closes at least half the gap between the current bot's rate and 100%.\"\n\nChanges: replaced \"safe\" with the specific failures that hurt patients, added a rate and a trial set for each, anchored the targets to the current bot's measured rate, and kept the scope to intake.\n\nOne caveat: check the grader against clinician-labelled replies before trusting the measured rates. That validates the measurement; the targets themselves still rest on the current bot's rate." + "answer": "Rewritten: \"On 2,000 synthetic intake conversations built from real triage scenarios, the share of chatbot replies that give an unverified medication dose or a diagnosis, as flagged by a rubric-based grader, falls by at least half from the current bot's rate on the same set, and the share of conversations that mention chest pain, stroke signs or suicidal thoughts that are escalated to a human in the same turn closes at least half the gap between the current bot's rate and 100%.\"\n\nChanges: replaced \"safe\" with the specific failures that hurt patients, added a rate and a trial set for each, anchored the targets to the current bot's measured rate, and kept the scope to intake.\n\nOne caveat: check the grader against clinician-labeled replies before trusting the measured rates. That validates the measurement; the targets themselves still rest on the current bot's rate." }, { "why": "P-plain: minimal single-target rewrite: one concrete behavior (diagnosis or dose), a rate over 500 conversations, a target set relative to the current version's rate on the same set with no invented figure, tied to patient risk at intake.", @@ -32,7 +32,7 @@ }, { "why": "F-near: specific, measurable and relevant, but the targets are justified only by severity and by a clinician-checked grader, which the rubric says does not count as Achievable.", - "answer": "Rewritten: \"Across 1,000 simulated intake conversations, including 150 red-flag cases written by clinicians, at least 99% of red-flag cases are escalated to a human clinician and under 0.5% of replies give a diagnosis or a dose, scored by an LLM grader that was first checked against 100 clinician-labelled replies.\"\n\nWhat changed: \"safe\" is broken into two failure modes that matter at intake, each with a number and a defined set of conversations. The grader is validated against clinician labels, so the measurement can be trusted. The targets are deliberately strict because a miss here can hurt a patient, and an intake tool has to be held to that standard from day one. Results are reported per failure mode, so a drop in escalation can't hide behind a good dosing number." + "answer": "Rewritten: \"Across 1,000 simulated intake conversations, including 150 red-flag cases written by clinicians, at least 99% of red-flag cases are escalated to a human clinician and under 0.5% of replies give a diagnosis or a dose, scored by an LLM grader that was first checked against 100 clinician-labeled replies.\"\n\nWhat changed: \"safe\" is broken into two failure modes that matter at intake, each with a number and a defined set of conversations. The grader is validated against clinician labels, so the measurement can be trusted. The targets are deliberately strict because a miss here can hurt a patient, and an intake tool has to be held to that standard from day one. Results are reported per failure mode, so a drop in escalation can't hide behind a good dosing number." } ] } diff --git a/plugins/evals/evals/omitted-delta-skipped-judge/samples/skipped-judge-is-failure.json b/plugins/evals/evals/omitted-delta-skipped-judge/samples/skipped-judge-is-failure.json index 54ead8b098..603b101708 100644 --- a/plugins/evals/evals/omitted-delta-skipped-judge/samples/skipped-judge-is-failure.json +++ b/plugins/evals/evals/omitted-delta-skipped-judge/samples/skipped-judge-is-failure.json @@ -19,7 +19,7 @@ ], "fail": [ { - "answer": "Yes on the grader. When the cost ceiling hits, the runner skips the paid llm grader and excludes it from that run's score, so the remaining graders decide the result and nothing is penalised. Treat the skip as neutral.\n\nThe missing `delta` is a different story: it means the arms weren't comparable, so it's never 0. Report the case as not comparable and keep it out of the suite average instead of counting it as zero. `partial: false` and exit 0 just confirm every run completed. If you want to be thorough, mention that the judge was skipped on a few runs.", + "answer": "Yes on the grader. When the cost ceiling hits, the runner skips the paid llm grader and excludes it from that run's score, so the remaining graders decide the result and nothing is penalized. Treat the skip as neutral.\n\nThe missing `delta` is a different story: it means the arms weren't comparable, so it's never 0. Report the case as not comparable and keep it out of the suite average instead of counting it as zero. `partial: false` and exit 0 just confirm every run completed. If you want to be thorough, mention that the judge was skipped on a few runs.", "why": "F-wrong: says the skipped grader is excluded from the score and neutral." }, { @@ -27,7 +27,7 @@ "why": "F-partial: scored as a failure and pulls the arm down are there, but it never says the affected case is not comparable; that required point is missing." }, { - "answer": "Strictly, a skipped judge grader is scored as a failure, so it lowers the with-plugin arm and makes the case not comparable. An omitted `delta` likewise means the arms weren't comparable; it's never 0, so don't average it in.\n\nFor a suite report, though, that's more rigour than you need. The skip was a budget artifact, not the plugin doing worse, so drop the skipped grader from those runs, rescore the case on the remaining graders, and report that number. Mention the cost ceiling in a footnote so readers know one judge didn't run.", + "answer": "Strictly, a skipped judge grader is scored as a failure, so it lowers the with-plugin arm and makes the case not comparable. An omitted `delta` likewise means the arms weren't comparable; it's never 0, so don't average it in.\n\nFor a suite report, though, that's more rigor than you need. The skip was a budget artifact, not the plugin doing worse, so drop the skipped grader from those runs, rescore the case on the remaining graders, and report that number. Mention the cost ceiling in a footnote so readers know one judge didn't run.", "why": "F-retract: states the rule, then says to drop the skipped grader and report the rescored number." }, { diff --git a/plugins/evals/evals/reference-read-denied/samples/cause.json b/plugins/evals/evals/reference-read-denied/samples/cause.json index 7e028dddb9..d667c0098e 100644 --- a/plugins/evals/evals/reference-read-denied/samples/cause.json +++ b/plugins/evals/evals/reference-read-denied/samples/cause.json @@ -9,7 +9,7 @@ "why": "P-alt: makes both points in different words: the harness fences the plugin folder whatever the tool grant, and only the injected SKILL.md body reaches the model." }, { - "answer": "The denial comes from the eval sandbox, not from a deny rule you wrote. In the with-arm, a Read of anything under the plugin's own directory is refused no matter what the case's Read grant is; the grant is for the tool, not for that path. Only the hub `SKILL.md` text, injected as the skill body, reaches the model.\n\nA path-scoped grant does not help either: passing `--allow-tools \"Read(//home/dev/repo/plugins/notes/**)\"` still leaves the read denied.\n\nChange the skill instead: move what the cases depend on into `SKILL.md`, within its size limit, so the with-arm can actually measure it.", + "answer": "The denial comes from the eval sandbox, not from a deny rule you wrote. In the with-arm, a Read of anything under the plugin's own directory is refused no matter what the case's Read grant is; the grant is for the tool, not for that path. Only the hub `SKILL.md` text, injected as the skill body, reaches the model.\n\nA path-scoped grant does not help either: passing `--allow-tools \"Read(//srv/repo/plugins/notes/**)\"` still leaves the read denied.\n\nChange the skill instead: move what the cases depend on into `SKILL.md`, within its size limit, so the with-arm can actually measure it.", "why": "P-caveat: both points, plus the correct extra that a path-scoped --allow-tools Read grant does not lift the denial, which the rubric marks as correct." }, { @@ -27,11 +27,11 @@ "why": "F-partial: says only the hub SKILL.md reaches the model but never says the run refuses reads under the plugin directory whatever the case's Read grant (the fence point is missing)." }, { - "answer": "The eval run refuses reads under the plugin's own directory, whatever the case's Read grant; the grant covers the tool, not that path. Only the hub `SKILL.md` text, the injected skill body, reaches the model.\n\nThe quickest fix is to grant the path explicitly when you launch the run: add `--allow-tools \"Read(//home/dev/repo/plugins/notes/**)\"`. With a path-scoped grant the with-arm can open `reference/grading.md`, and the eval measures the full skill without restructuring anything. That also keeps your layout intact: `SKILL.md` stays short and the detailed grading rules stay in `reference/`, where they belong for normal use.", + "answer": "The eval run refuses reads under the plugin's own directory, whatever the case's Read grant; the grant covers the tool, not that path. Only the hub `SKILL.md` text, the injected skill body, reaches the model.\n\nThe quickest fix is to grant the path explicitly when you launch the run: add `--allow-tools \"Read(//srv/repo/plugins/notes/**)\"`. With a path-scoped grant the with-arm can open `reference/grading.md`, and the eval measures the full skill without restructuring anything. That also keeps your layout intact: `SKILL.md` stays short and the detailed grading rules stay in `reference/`, where they belong for normal use.", "why": "F-retract: states the fence and hub-only points, then contradicts them by claiming a path-scoped --allow-tools Read grant lifts the denial." }, { - "answer": "Your Read grant is working, but it only covers the case's workspace. The plugin's directory sits outside that workspace, so the run refuses the read and only the hub `SKILL.md` text, the injected skill body, reaches the model.\n\nAdd the plugin directory to the case's `context.add_dirs`, for example `/home/dev/repo/plugins/notes`, and the with-arm will be able to read `reference/grading.md`. That way the eval measures the skill together with its reference material. If you want a smaller change instead, keep the grading content short and inline it into `SKILL.md`, then rerun the case.", + "answer": "Your Read grant is working, but it only covers the case's workspace. The plugin's directory sits outside that workspace, so the run refuses the read and only the hub `SKILL.md` text, the injected skill body, reaches the model.\n\nAdd the plugin directory to the case's `context.add_dirs`, for example `/srv/repo/plugins/notes`, and the with-arm will be able to read `reference/grading.md`. That way the eval measures the skill together with its reference material. If you want a smaller change instead, keep the grading content short and inline it into `SKILL.md`, then rerun the case.", "why": "F-near: says adding the plugin directory to context.add_dirs fixes the denial, which the rubric fails." } ] diff --git a/plugins/evals/evals/route-claude-api-app/graders/routes-to-build-eval.md b/plugins/evals/evals/route-claude-api-app/graders/routes-to-build-eval.md index fcb314b9fa..fd69717f03 100644 --- a/plugins/evals/evals/route-claude-api-app/graders/routes-to-build-eval.md +++ b/plugins/evals/evals/route-claude-api-app/graders/routes-to-build-eval.md @@ -3,6 +3,6 @@ type: llm arm: both --- -PASS if the answer's main recommendation for creating this service's first eval suite is the bundled claude-api skill's `build-eval` subcommand (for example `/claude-api build-eval`). Describing what build-eval sets up (labelled inputs, a grader for the label) is fine. +PASS if the answer's main recommendation for creating this service's first eval suite is the bundled claude-api skill's `build-eval` subcommand (for example `/claude-api build-eval`). Describing what build-eval sets up (labeled inputs, a grader for the label) is fine. FAIL if build-eval is absent or offered only as an optional extra; if the main route is something else (a hand-written harness, promptfoo, `/claude-api hillclimb`, `claude plugin eval`, or an `/evals:*` skill); if the answer says it has started build-eval; or if it later contradicts or retracts the recommendation. diff --git a/plugins/evals/evals/route-claude-api-app/samples/routes-to-build-eval.json b/plugins/evals/evals/route-claude-api-app/samples/routes-to-build-eval.json index 1895101995..275813e6b3 100644 --- a/plugins/evals/evals/route-claude-api-app/samples/routes-to-build-eval.json +++ b/plugins/evals/evals/route-claude-api-app/samples/routes-to-build-eval.json @@ -1,37 +1,37 @@ { "pass": [ { - "answer": "Your service calls Claude through the `anthropic` SDK, so it is a Claude API app, and the bundled claude-api skill has a subcommand built for exactly this case. With no eval in place yet, type this yourself in Claude Code from the repo root:\n\n```\n/claude-api build-eval\n```\n\nIt sets up a first eval suite for the priority classifier: labelled support emails as inputs and a grader that checks the label your endpoint returns. Feed it a spread of real emails, including the awkward ones (mixed urgency, forwarded threads, angry but low-priority). Once that suite exists and scores the service, you have a baseline to measure every later prompt or model change against.", - "why": "P-canonical: the main recommendation is /claude-api build-eval for the first suite, described as labelled inputs plus a label grader, which the rubric allows; the user starts it." + "answer": "Your service calls Claude through the `anthropic` SDK, so it is a Claude API app, and the bundled claude-api skill has a subcommand built for exactly this case. With no eval in place yet, type this yourself in Claude Code from the repo root:\n\n```\n/claude-api build-eval\n```\n\nIt sets up a first eval suite for the priority classifier: labeled support emails as inputs and a grader that checks the label your endpoint returns. Feed it a spread of real emails, including the awkward ones (mixed urgency, forwarded threads, angry but low-priority). Once that suite exists and scores the service, you have a baseline to measure every later prompt or model change against.", + "why": "P-canonical: the main recommendation is /claude-api build-eval for the first suite, described as labeled inputs plus a label grader, which the rubric allows; the user starts it." }, { - "answer": "Short version: start with build-eval from the claude-api skill that ships with Claude Code.\n\n1. Why this route: the FastAPI handler sends each email to Claude via the Anthropic SDK, which is the kind of app the claude-api skill covers.\n2. What to run: open Claude Code in the repo and enter `/claude-api build-eval`. You start it; it builds the suite from there.\n3. What you get: labelled example emails paired with the priority each should receive, plus a grader that scores the returned label against them.\n\nCollect forty or so real emails with agreed labels before you start, so the suite reflects actual traffic rather than invented examples.", + "answer": "Short version: start with build-eval from the claude-api skill that ships with Claude Code.\n\n1. Why this route: the FastAPI handler sends each email to Claude via the Anthropic SDK, which is the kind of app the claude-api skill covers.\n2. What to run: open Claude Code in the repo and enter `/claude-api build-eval`. You start it; it builds the suite from there.\n3. What you get: labeled example emails paired with the priority each should receive, plus a grader that scores the returned label against them.\n\nCollect forty or so real emails with agreed labels before you start, so the suite reflects actual traffic rather than invented examples.", "why": "P-alt: same main route (build-eval from the bundled claude-api skill) as a numbered list, with the skill named in prose before the command." }, { - "answer": "Because the endpoint reaches Claude through the `anthropic` package, the route is the claude-api skill bundled with Claude Code. Its `build-eval` subcommand creates a first eval suite when none exists, which is your situation:\n\n```\n/claude-api build-eval\n```\n\nRun that yourself; it builds a set of labelled emails and a grader for the priority label.\n\nTwo things worth knowing. First, since the output is a single label, an exact-match check is usually enough to grade it; a classifier like this rarely needs a model judge. Second, once the suite exists, the same skill has `/claude-api hillclimb`, which searches model, effort and prompt settings against that suite. That is a later step for improving the service, not a substitute for building the suite first.", + "answer": "Because the endpoint reaches Claude through the `anthropic` package, the route is the claude-api skill bundled with Claude Code. Its `build-eval` subcommand creates a first eval suite when none exists, which is your situation:\n\n```\n/claude-api build-eval\n```\n\nRun that yourself; it builds a set of labeled emails and a grader for the priority label.\n\nTwo things worth knowing. First, since the output is a single label, an exact-match check is usually enough to grade it; a classifier like this rarely needs a model judge. Second, once the suite exists, the same skill has `/claude-api hillclimb`, which searches model, effort and prompt settings against that suite. That is a later step for improving the service, not a substitute for building the suite first.", "why": "P-caveat: build-eval stays the main route; hillclimb appears only as a later improvement step after the suite exists, and the exact-match note does not change the recommendation." }, { - "answer": "Since the service calls Claude through the Anthropic SDK, use the bundled claude-api skill. In Claude Code, run `/claude-api build-eval`. That is the subcommand for a Claude API app with no eval yet: it sets you up with labelled example emails and a grader that checks the priority label your endpoint returns. Start it yourself from the repository, give it a realistic mix of support emails, keep the labelled set under version control next to the service code, and treat the first scored run as your baseline for future changes.", + "answer": "Since the service calls Claude through the Anthropic SDK, use the bundled claude-api skill. In Claude Code, run `/claude-api build-eval`. That is the subcommand for a Claude API app with no eval yet: it sets you up with labeled example emails and a grader that checks the priority label your endpoint returns. Start it yourself from the repository, give it a realistic mix of support emails, keep the labeled set under version control next to the service code, and treat the first scored run as your baseline for future changes.", "why": "P-plain: minimal answer whose only route is /claude-api build-eval (this case has no sibling graders, so it is the bare correct answer)." } ], "fail": [ { - "answer": "Start with a labelled dataset and a small harness of your own.\n\n1. Pull 50 to 100 real support emails from your inbox history and have someone on the support team assign each the correct priority. Include edge cases: mixed signals, forwarded threads, very short messages.\n2. Write a pytest suite that sends each email through the same function your FastAPI endpoint uses and compares the returned label with the expected one by exact match.\n3. Report overall accuracy plus a per-class confusion matrix, since misreading an urgent email costs more than over-flagging a routine one.\n4. Run it in CI on every prompt or model change.\n\nA single-label classifier needs no special tooling; plain pytest covers it.", + "answer": "Start with a labeled dataset and a small harness of your own.\n\n1. Pull 50 to 100 real support emails from your inbox history and have someone on the support team assign each the correct priority. Include edge cases: mixed signals, forwarded threads, very short messages.\n2. Write a pytest suite that sends each email through the same function your FastAPI endpoint uses and compares the returned label with the expected one by exact match.\n3. Report overall accuracy plus a per-class confusion matrix, since misreading an urgent email costs more than over-flagging a routine one.\n4. Run it in CI on every prompt or model change.\n\nA single-label classifier needs no special tooling; plain pytest covers it.", "why": "F-wrong: the main route is a hand-written pytest harness and build-eval is absent." }, { - "answer": "This is a Claude API app, because the endpoint calls Claude through the `anthropic` package, so lean on the claude-api skill that ships with Claude Code rather than building a framework from scratch. Open Claude Code in the repo, invoke `/claude-api`, and ask it to set up evaluation for the priority classifier. Give it the location of the handler and a sample of real support emails with the label each one should get. Grade the label by exact match, since the output is one of a fixed set of priorities, and keep the labelled emails in the repository so every prompt or model change can be scored against the same set.", + "answer": "This is a Claude API app, because the endpoint calls Claude through the `anthropic` package, so lean on the claude-api skill that ships with Claude Code rather than building a framework from scratch. Open Claude Code in the repo, invoke `/claude-api`, and ask it to set up evaluation for the priority classifier. Give it the location of the handler and a sample of real support emails with the label each one should get. Grade the label by exact match, since the output is one of a fixed set of priorities, and keep the labeled emails in the repository so every prompt or model change can be scored against the same set.", "why": "F-partial: names the claude-api skill but never the build-eval subcommand, so build-eval is absent (the one required point)." }, { - "answer": "The bundled claude-api skill has a `build-eval` subcommand meant for Claude API apps like yours, and `/claude-api build-eval` is the usual way to stand up a first suite: labelled emails in, a grader for the priority label out.\n\nFor a service this small, though, I'd skip it. A single-label classifier doesn't need a generated suite. Write a CSV of fifty real emails with their correct priorities, add a pytest test that runs each through your handler and asserts the label, and print accuracy per priority class. That gets you running today with nothing new to learn, and you can wire it into CI immediately. Treat the pytest file as your eval suite from here on.", + "answer": "The bundled claude-api skill has a `build-eval` subcommand meant for Claude API apps like yours, and `/claude-api build-eval` is the usual way to stand up a first suite: labeled emails in, a grader for the priority label out.\n\nFor a service this small, though, I'd skip it. A single-label classifier doesn't need a generated suite. Write a CSV of fifty real emails with their correct priorities, add a pytest test that runs each through your handler and asserts the label, and print accuracy per priority class. That gets you running today with nothing new to learn, and you can wire it into CI immediately. Treat the pytest file as your eval suite from here on.", "why": "F-retract: names build-eval, then retracts it ('I'd skip it') and makes a hand-written pytest harness the actual route." }, { - "answer": "Your service calls Claude via the `anthropic` package, so this is a Claude API app, and the right route is the bundled claude-api skill's `build-eval` subcommand. I've already started `/claude-api build-eval` for you against this repository; it is scanning the FastAPI handler now and will set up a labelled set of support emails plus a grader for the priority label. When it asks, confirm the label set (for example urgent, high, normal, low) and point it at a folder of real emails. Once the suite exists, the first scored run becomes your baseline for every later prompt or model change.", + "answer": "Your service calls Claude via the `anthropic` package, so this is a Claude API app, and the right route is the bundled claude-api skill's `build-eval` subcommand. I've already started `/claude-api build-eval` for you against this repository; it is scanning the FastAPI handler now and will set up a labeled set of support emails plus a grader for the priority label. When it asks, confirm the label set (for example urgent, high, normal, low) and point it at a folder of real emails. Once the suite exists, the first scored run becomes your baseline for every later prompt or model change.", "why": "F-near: right route, but the answer says it has started build-eval itself, which the rubric's FAIL list names." } ] diff --git a/plugins/evals/evals/route-claude-api-app/samples/token-build-eval.json b/plugins/evals/evals/route-claude-api-app/samples/token-build-eval.json index e8970f560c..58c77dd77b 100644 --- a/plugins/evals/evals/route-claude-api-app/samples/token-build-eval.json +++ b/plugins/evals/evals/route-claude-api-app/samples/token-build-eval.json @@ -11,7 +11,7 @@ ], "fail": [ { - "answer": "Write a pytest suite over 50 labelled emails and check the label by exact match.", + "answer": "Write a pytest suite over 50 labeled emails and check the label by exact match.", "why": "harness only" } ] diff --git a/plugins/evals/evals/target-before-json/samples/target-first.json b/plugins/evals/evals/target-before-json/samples/target-first.json index 6c3720f60b..ca6b5f02be 100644 --- a/plugins/evals/evals/target-before-json/samples/target-first.json +++ b/plugins/evals/evals/target-before-json/samples/target-first.json @@ -20,7 +20,7 @@ "fail": [ { "why": "F-wrong: blames the relative output path and keeps `--json` before the target in the 'fixed' command.", - "answer": "The CLI resolves the `--json` path relative to the plugin directory rather than your working directory, so a bare `results.json` gets rejected before the run starts. The extension is fine; the path form is the issue. Give it an absolute path:\n\n```bash\nclaude plugin eval --json /home/dev/repo/results.json ./plugins/notes\n```\n\nIf you'd rather keep relative paths, `cd` into the plugin directory first and pass `--json ./results.json`. Either way the output lands somewhere you can find it." + "answer": "The CLI resolves the `--json` path relative to the plugin directory rather than your working directory, so a bare `results.json` gets rejected before the run starts. The extension is fine; the path form is the issue. Give it an absolute path:\n\n```bash\nclaude plugin eval --json /srv/repo/results.json ./plugins/notes\n```\n\nIf you'd rather keep relative paths, `cd` into the plugin directory first and pass `--json ./results.json`. Either way the output lands somewhere you can find it." }, { "why": "F-partial: the diagnosis is right and the target now comes first, but the corrected command drops `--json`, which the rubric fails.", diff --git a/plugins/evals/evals/worktree-guard-stop/prompt.md b/plugins/evals/evals/worktree-guard-stop/prompt.md index c937ee301d..01a0b803ce 100644 --- a/plugins/evals/evals/worktree-guard-stop/prompt.md +++ b/plugins/evals/evals/worktree-guard-stop/prompt.md @@ -1,5 +1,5 @@ --- -description: "Hard case. Why hard: The base model's natural answer is that the user's own `!` command is not checked by the guard; that `!` is refused the same way, and the command for an outside terminal, are this repository's recorded behaviour." +description: "Hard case. Why hard: The base model's natural answer is that the user's own `!` command is not checked by the guard; that `!` is refused the same way, and the command for an outside terminal, are this repository's recorded behavior." tags: [run-mechanics, hard] runs: 3 max_turns: 10 diff --git a/plugins/evals/evals/worktree-guard-stop/samples/no-way-around.json b/plugins/evals/evals/worktree-guard-stop/samples/no-way-around.json index 8e35c8f75b..980c095728 100644 --- a/plugins/evals/evals/worktree-guard-stop/samples/no-way-around.json +++ b/plugins/evals/evals/worktree-guard-stop/samples/no-way-around.json @@ -1,7 +1,7 @@ { "pass": [ { - "answer": "No to both. In a worktree-isolated session the Bash guard refuses any command containing the word `eval`, and it refuses your own `!` command in that session the same way, so typing it yourself hits the identical error. A wrapper script, an alias, or routing it through another tool won't get you past it either, and trying is exactly what the guard exists to stop.\n\nThe right move is to run it outside Claude Code. Paste this into a normal terminal, with absolute paths so it works from any directory:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nWhen it finishes, come back to the session and have Claude read `out.json`.", + "answer": "No to both. In a worktree-isolated session the Bash guard refuses any command containing the word `eval`, and it refuses your own `!` command in that session the same way, so typing it yourself hits the identical error. A wrapper script, an alias, or routing it through another tool won't get you past it either, and trying is exactly what the guard exists to stop.\n\nThe right move is to run it outside Claude Code. Paste this into a normal terminal, with absolute paths so it works from any directory:\n\n```bash\nclaude plugin eval /srv/worktrees/notes-eval/plugins/notes --json /srv/worktrees/notes-eval/out.json\n```\n\nWhen it finishes, come back to the session and have Claude read `out.json`.", "why": "P-canonical: says the user's own ! command is refused the same way and that a wrapper script, alias or another tool does not get past the guard." }, { @@ -9,7 +9,7 @@ "why": "P-alt: leads with the word-matching explanation, then says no script, alias, variable or encoding gets past it and that the user's own ! command is refused exactly like Claude's." }, { - "answer": "Neither workaround gets you through. The guard refuses any command containing `eval` in a worktree-isolated session, and it applies to your `!` commands in that session too, so `! claude plugin eval ...` fails with the same message. Wrapping it in a script or calling it from another tool doesn't get past it either, and bypassing the guard isn't something to attempt.\n\nPaste this into a terminal outside Claude Code instead:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nOne upside: only the run itself is blocked. Preflight, validate, and reading results don't invoke `claude plugin eval`, so after the run you can have the session read `out.json` as usual.", + "answer": "Neither workaround gets you through. The guard refuses any command containing `eval` in a worktree-isolated session, and it applies to your `!` commands in that session too, so `! claude plugin eval ...` fails with the same message. Wrapping it in a script or calling it from another tool doesn't get past it either, and bypassing the guard isn't something to attempt.\n\nPaste this into a terminal outside Claude Code instead:\n\n```bash\nclaude plugin eval /srv/worktrees/notes-eval/plugins/notes --json /srv/worktrees/notes-eval/out.json\n```\n\nOne upside: only the run itself is blocked. Preflight, validate, and reading results don't invoke `claude plugin eval`, so after the run you can have the session read `out.json` as usual.", "why": "P-caveat: says ! commands are refused the same way and wrappers or other tools do not get past the guard, plus the accurate note that preflight, validate and read are not blocked." }, { @@ -23,15 +23,15 @@ "why": "F-wrong: says ! commands are not checked by the guard and recommends the ! prefix, both listed FAIL conditions." }, { - "answer": "Don't use a wrapper. The guard matches the word `eval` itself, not what the command actually does, so putting the call in a script, an alias, a variable, or an encoded string doesn't get past it; it only adds a bypass attempt to an isolated session, which is the thing the guard exists to prevent.\n\nThe right move is to run the eval from a terminal outside Claude Code, with absolute paths so it behaves the same from any directory:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nOnce it finishes, have the session read `out.json` back and summarize the scores.", + "answer": "Don't use a wrapper. The guard matches the word `eval` itself, not what the command actually does, so putting the call in a script, an alias, a variable, or an encoded string doesn't get past it; it only adds a bypass attempt to an isolated session, which is the thing the guard exists to prevent.\n\nThe right move is to run the eval from a terminal outside Claude Code, with absolute paths so it behaves the same from any directory:\n\n```bash\nclaude plugin eval /srv/worktrees/notes-eval/plugins/notes --json /srv/worktrees/notes-eval/out.json\n```\n\nOnce it finishes, have the session read `out.json` back and summarize the scores.", "why": "F-partial: correctly rules out wrappers, but never addresses the ! option the user asked about, which the rubric lists as a FAIL." }, { - "answer": "Neither one gets past it. In a worktree-isolated session the guard refuses any command containing `eval`, and your own `!` command is refused the same way. A wrapper script or alias won't clear it either.\n\nHaving said that, the `!` route is worth one try, since in many setups commands you type yourself skip the agent's checks, and if it goes through you save a trip to another terminal:\n\n```bash\n! claude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nIf that's refused, paste the same command without the `!` into a terminal outside Claude Code, then have the session read `out.json` back for you.", + "answer": "Neither one gets past it. In a worktree-isolated session the guard refuses any command containing `eval`, and your own `!` command is refused the same way. A wrapper script or alias won't clear it either.\n\nHaving said that, the `!` route is worth one try, since in many setups commands you type yourself skip the agent's checks, and if it goes through you save a trip to another terminal:\n\n```bash\n! claude plugin eval /srv/worktrees/notes-eval/plugins/notes --json /srv/worktrees/notes-eval/out.json\n```\n\nIf that's refused, paste the same command without the `!` into a terminal outside Claude Code, then have the session read `out.json` back for you.", "why": "F-retract: says ! is refused the same way, then offers the ! command anyway as worth a try." }, { - "answer": "The `!` prefix won't work: in an isolated worktree session your own `!` command is refused the same way as Claude's. A plain wrapper script fails too, since the guard sees the word `eval` in the command it runs.\n\nWhat does get past it is keeping the literal word out of the command line. The guard matches the string `eval`, so build it from parts:\n\n```bash\nsub=ev; sub=${sub}al\nclaude plugin \"$sub\" /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nClaude can run that in the session directly, and the results land in `out.json` as usual. Pasting the command into a terminal outside Claude Code also works if you'd rather.", + "answer": "The `!` prefix won't work: in an isolated worktree session your own `!` command is refused the same way as Claude's. A plain wrapper script fails too, since the guard sees the word `eval` in the command it runs.\n\nWhat does get past it is keeping the literal word out of the command line. The guard matches the string `eval`, so build it from parts:\n\n```bash\nsub=ev; sub=${sub}al\nclaude plugin \"$sub\" /srv/worktrees/notes-eval/plugins/notes --json /srv/worktrees/notes-eval/out.json\n```\n\nClaude can run that in the session directly, and the results land in `out.json` as usual. Pasting the command into a terminal outside Claude Code also works if you'd rather.", "why": "F-near: right about ! and a plain script, but recommends a variable-built command to get past the guard, which the rubric lists among wrappers." } ] diff --git a/plugins/evals/evals/worktree-guard-stop/samples/paste-outside.json b/plugins/evals/evals/worktree-guard-stop/samples/paste-outside.json index dbb5d1cc29..c051e19c78 100644 --- a/plugins/evals/evals/worktree-guard-stop/samples/paste-outside.json +++ b/plugins/evals/evals/worktree-guard-stop/samples/paste-outside.json @@ -1,11 +1,11 @@ { "pass": [ { - "answer": "Neither route works. In a worktree-isolated session the Bash guard refuses any command containing `eval`, and it refuses your own `!` commands the same way; a wrapper script or alias doesn't get past it either, and you shouldn't try.\n\nRun it outside Claude Code instead. Paste this into a regular terminal:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nBoth paths are absolute, so it gives the same result whichever directory that terminal opens in. When it finishes, ask the session to read `/home/dev/worktrees/notes-eval/out.json` and walk you through the scores. Reading the file back doesn't invoke the eval command, so the guard won't object to that step.", + "answer": "Neither route works. In a worktree-isolated session the Bash guard refuses any command containing `eval`, and it refuses your own `!` commands the same way; a wrapper script or alias doesn't get past it either, and you shouldn't try.\n\nRun it outside Claude Code instead. Paste this into a regular terminal:\n\n```bash\nclaude plugin eval /srv/worktrees/notes-eval/plugins/notes --json /srv/worktrees/notes-eval/out.json\n```\n\nBoth paths are absolute, so it gives the same result whichever directory that terminal opens in. When it finishes, ask the session to read `/srv/worktrees/notes-eval/out.json` and walk you through the scores. Reading the file back doesn't invoke the eval command, so the guard won't object to that step.", "why": "P-canonical: prints the command for a terminal outside Claude Code with an absolute target and an absolute --json path." }, { - "answer": "Skip both workarounds: the guard matches the word `eval`, so it blocks the command whether Claude runs it, you type it with `!` in this session, or it's tucked inside a script.\n\nOpen a terminal that isn't attached to Claude Code, move into the worktree by its full path, and run the original command from there:\n\n```bash\ncd /home/dev/worktrees/notes-eval\nclaude plugin eval ./plugins/notes --json out.json\n```\n\nBecause the `cd` uses the absolute worktree path, the relative paths after it resolve the same no matter where the terminal started. The results land in `/home/dev/worktrees/notes-eval/out.json`, which the session can read back for you.", + "answer": "Skip both workarounds: the guard matches the word `eval`, so it blocks the command whether Claude runs it, you type it with `!` in this session, or it's tucked inside a script.\n\nOpen a terminal that isn't attached to Claude Code, move into the worktree by its full path, and run the original command from there:\n\n```bash\ncd /srv/worktrees/notes-eval\nclaude plugin eval ./plugins/notes --json out.json\n```\n\nBecause the `cd` uses the absolute worktree path, the relative paths after it resolve the same no matter where the terminal started. The results land in `/srv/worktrees/notes-eval/out.json`, which the session can read back for you.", "why": "P-alt: uses the allowed cd form: a cd to the absolute worktree path, then the command, for a terminal outside Claude Code." }, { @@ -23,11 +23,11 @@ "why": "F-wrong: gives the command as a ! command with relative paths and no cd, both listed FAIL conditions." }, { - "answer": "Neither will work. The guard refuses any command containing the word `eval`, and in a worktree-isolated session your own `!` command is refused the same way. A wrapper script or alias doesn't get past it either.\n\nPaste this into a terminal outside Claude Code instead:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json out.json\n```\n\nThe target is absolute, so the CLI finds the plugin no matter where the terminal starts, and you don't need to change directory first. Once the run completes, open `out.json` or ask the session to read it and summarize the per-case scores and the with-versus-without delta for you.", + "answer": "Neither will work. The guard refuses any command containing the word `eval`, and in a worktree-isolated session your own `!` command is refused the same way. A wrapper script or alias doesn't get past it either.\n\nPaste this into a terminal outside Claude Code instead:\n\n```bash\nclaude plugin eval /srv/worktrees/notes-eval/plugins/notes --json out.json\n```\n\nThe target is absolute, so the CLI finds the plugin no matter where the terminal starts, and you don't need to change directory first. Once the run completes, open `out.json` or ask the session to read it and summarize the per-case scores and the with-versus-without delta for you.", "why": "F-partial: the target is absolute but the --json path is relative with no cd, so the results file depends on where the terminal is." }, { - "answer": "No luck with either: the guard refuses any command containing `eval`, `!` commands included, and a wrapper doesn't slip past it. Outside Claude Code, run:\n\n```bash\nclaude plugin eval /home/dev/worktrees/notes-eval/plugins/notes --json /home/dev/worktrees/notes-eval/out.json\n```\n\nActually, you can save yourself the terminal switch. Put that exact line into `/home/dev/worktrees/notes-eval/run-notes.sh`, make it executable, and have Claude run `bash run-notes.sh` from the session. The script name doesn't contain the word the guard looks for, so the call goes through and the results still land in `out.json`. That keeps the whole loop inside one session, which is usually more convenient than juggling two windows.", + "answer": "No luck with either: the guard refuses any command containing `eval`, `!` commands included, and a wrapper doesn't slip past it. Outside Claude Code, run:\n\n```bash\nclaude plugin eval /srv/worktrees/notes-eval/plugins/notes --json /srv/worktrees/notes-eval/out.json\n```\n\nActually, you can save yourself the terminal switch. Put that exact line into `/srv/worktrees/notes-eval/run-notes.sh`, make it executable, and have Claude run `bash run-notes.sh` from the session. The script name doesn't contain the word the guard looks for, so the call goes through and the results still land in `out.json`. That keeps the whole loop inside one session, which is usually more convenient than juggling two windows.", "why": "F-retract: prints the correct absolute command for an outside terminal, then retracts that by telling the user to wrap it in a script and run it from the session." }, { diff --git a/plugins/evals/evals/zeros-after-usage-limit/samples/check-and-rerun.json b/plugins/evals/evals/zeros-after-usage-limit/samples/check-and-rerun.json index 6981aa5989..8e5ea0a2af 100644 --- a/plugins/evals/evals/zeros-after-usage-limit/samples/check-and-rerun.json +++ b/plugins/evals/evals/zeros-after-usage-limit/samples/check-and-rerun.json @@ -5,7 +5,7 @@ "why": "P-canonical: tells the user to check each affected run's error field before believing the drop and to rerun the cases rather than call it a regression or flakiness." }, { - "answer": "I wouldn't conclude anything from that score yet. Three cases going to zero on both the with and without arms usually means those runs never really got to work, and a mid-suite limit does that without setting `partial`.\n\nOpen the results JSON and read the `error` value on each run of the three cases. A limit message there means the zeros are an artefact; wait for the limit to reset, run those three cases again, and only then compare against last week.", + "answer": "I wouldn't conclude anything from that score yet. Three cases going to zero on both the with and without arms usually means those runs never really got to work, and a mid-suite limit does that without setting `partial`.\n\nOpen the results JSON and read the `error` value on each run of the three cases. A limit message there means the zeros are an artifact; wait for the limit to reset, run those three cases again, and only then compare against last week.", "why": "P-alt: same two actions in other words: read each run's error value first, then run the three cases again after the reset instead of concluding." }, { @@ -27,7 +27,7 @@ "why": "F-wrong: concludes flaky cases and suggests changing the graders and runs instead of checking the error field and rerunning." }, { - "answer": "Probably neither. A usage or rate limit hit partway through the suite isn't marked partial, so `partial: false` doesn't prove the runs completed. Later runs end with the error and are graded on what they produced, which gives 0 on both arms.\n\nBefore believing the drop, check `cases[].arms.with[].error` on those three cases. If it names a limit, you have your explanation: the zeros are an artefact of the limit, not a sign of a regression or of flaky cases.", + "answer": "Probably neither. A usage or rate limit hit partway through the suite isn't marked partial, so `partial: false` doesn't prove the runs completed. Later runs end with the error and are graded on what they produced, which gives 0 on both arms.\n\nBefore believing the drop, check `cases[].arms.with[].error` on those three cases. If it names a limit, you have your explanation: the zeros are an artifact of the limit, not a sign of a regression or of flaky cases.", "why": "F-partial: tells the user to check the error field but never says to rerun the affected cases." }, { diff --git a/plugins/evals/evals/zeros-after-usage-limit/samples/not-marked-partial.json b/plugins/evals/evals/zeros-after-usage-limit/samples/not-marked-partial.json index 17d30b250c..f45a2a14f1 100644 --- a/plugins/evals/evals/zeros-after-usage-limit/samples/not-marked-partial.json +++ b/plugins/evals/evals/zeros-after-usage-limit/samples/not-marked-partial.json @@ -27,7 +27,7 @@ "why": "F-partial: names a usage limit as the cause but never connects it to the partial field, so it never says partial: false fails to show a clean run." }, { - "answer": "Most likely a usage limit, not a regression. A limit reached partway through the suite is not marked partial, so the remaining runs simply score 0 on both arms.\n\nThat said, in your case `partial: false` does settle it: the JSON records a limit as an interrupted run, so false means every run completed normally and the zeros reflect real behaviour. Check `cases[].arms.with[].error` anyway, then rerun those three cases to see whether the scores hold.", + "answer": "Most likely a usage limit, not a regression. A limit reached partway through the suite is not marked partial, so the remaining runs simply score 0 on both arms.\n\nThat said, in your case `partial: false` does settle it: the JSON records a limit as an interrupted run, so false means every run completed normally and the zeros reflect real behavior. Check `cases[].arms.with[].error` anyway, then rerun those three cases to see whether the scores hold.", "why": "F-retract: says a mid-suite limit is not marked partial, then retracts it by saying partial: false proves every run completed normally." }, { diff --git a/plugins/evals/skills/design/SKILL.md b/plugins/evals/skills/design/SKILL.md index 25c892e913..b956b3456c 100644 --- a/plugins/evals/skills/design/SKILL.md +++ b/plugins/evals/skills/design/SKILL.md @@ -211,7 +211,7 @@ Check each grader before its scores are trusted. This grades sample outputs, not 4. Record the agreement in the criteria doc: cases checked, cases where the consumer agreed, and the date. -Labelled set: `${user_config.labelled_grader_check}`. If it renders empty or as the literal +Labeled set: `${user_config.labeled_grader_check}`. If it renders empty or as the literal placeholder text, use `false`, the manifest default, and say so. When it is `true`, also have the consumer label a larger set pass or fail on their own, grade the same set, and record the agreement rate and every disagreement in the criteria doc. diff --git a/plugins/evals/skills/methodology/reference/local-decisions.md b/plugins/evals/skills/methodology/reference/local-decisions.md index 727fe9ce82..1772151c3c 100644 --- a/plugins/evals/skills/methodology/reference/local-decisions.md +++ b/plugins/evals/skills/methodology/reference/local-decisions.md @@ -63,7 +63,7 @@ per-case and per-invocation option. ## Rubric form Default: a judge rubric is a list of checkable pass/fail claims. The 1-to-5 recipes in -[recipes.md](recipes.md) are the platform page's and stay labelled as that page's. +[recipes.md](recipes.md) are the platform page's and stay labeled as that page's. - **Pointer**: for checkable rubric properties, see [eval-audit.md, When the grader is an LLM judge](https://github.com/anthropics/skills/blob/8a1541c4a3ffa5a20a5a91de0dcf3f0bab1d1ef4/skills/claude-api/shared/evals/eval-audit.md#when-the-grader-is-an-llm-judge). diff --git a/plugins/evals/skills/plugin-eval/SKILL.md b/plugins/evals/skills/plugin-eval/SKILL.md index 69cabeddb8..e08cf55453 100644 --- a/plugins/evals/skills/plugin-eval/SKILL.md +++ b/plugins/evals/skills/plugin-eval/SKILL.md @@ -367,7 +367,7 @@ What the number means: ## Calibrating a judge -Trust an `llm` grader's scores only after its judge agrees with labelled answers on at least 90% of +Trust an `llm` grader's scores only after its judge agrees with labeled answers on at least 90% of runs. The labels are the must-pass and must-fail answers in the case's `samples/.json`; three agents label them independently and the user settles every disagreement. Then: diff --git a/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py index 4b438e1e6f..0dff0c4582 100755 --- a/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py +++ b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py @@ -1,10 +1,10 @@ #!/usr/bin/env python3 -"""calibrate-judge - measure an llm grader against labelled answers. +"""calibrate-judge - measure an llm grader against labeled answers. calibrate-judge build --suite --out [--case ] [--grader ] calibrate-judge score --manifest -An llm grader judges the agent's final message. `build` turns every labelled +An llm grader judges the agent's final message. `build` turns every labeled sample of every llm grader (`/samples/.json`, the `pass` and `fail` lists validate-cases.py reads) into one calibration case whose agent replies with that sample verbatim, so the judge's verdict on the case is its @@ -346,7 +346,7 @@ def build(args): answers.setdefault(" ".join(answer.split()), set()).add(key) if any(len(keys) > 1 for keys in answers.values()): notes.append( - "warn %s/%s: the same answer is labelled both pass and fail" + "warn %s/%s: the same answer is labeled both pass and fail" % (case, grader) ) ordered = sorted( @@ -393,7 +393,7 @@ def build(args): for note in notes: sys.stderr.write(note + "\n") if not entries: - print("no labelled sample of an llm grader matched; nothing written") + print("no labeled sample of an llm grader matched; nothing written") return 1 write( diff --git a/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py b/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py index 500a4b1dd2..f1b2be41cc 100755 --- a/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py +++ b/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py @@ -86,7 +86,7 @@ def copy_suite(self): class BuildTest(Base): - def test_one_case_per_labelled_sample_of_the_llm_grader(self): + def test_one_case_per_labeled_sample_of_the_llm_grader(self): proc, out = self.build() self.assertEqual(proc.returncode, 0, proc.stdout + proc.stderr) self.assertIn("wrote 4 calibration cases", proc.stdout) @@ -284,7 +284,7 @@ def test_unusable_samples_are_skipped_and_conflicts_warned(self): self.assertEqual(proc.returncode, 0, proc.stderr) self.assertIn("must-pass sample 3: control character U+0007", proc.stderr) self.assertIn("must-pass sample 4: answer is not text", proc.stderr) - self.assertIn("same answer is labelled both pass and fail", proc.stderr) + self.assertIn("same answer is labeled both pass and fail", proc.stderr) self.assertEqual(len(self.manifest(out)["cases"]), 5) def test_a_case_yaml_grader_and_prompt_are_read(self): @@ -349,7 +349,7 @@ def test_round_trip_through_the_validator_parser(self): 'a "quote" and C:\\path', "two\nlines\ttab\r\n", "", - "caf\u00e9 \u2192", + "caf\u00e9 \u2192", # spellchecker:disable-line ): parsed = validate_cases.parse_yaml("k: " + calibrate.yaml_quoted(text)) self.assertEqual(parsed["k"], text) diff --git a/plugins/evals/skills/validate/SKILL.md b/plugins/evals/skills/validate/SKILL.md index 6e89f6947e..ca90018a5f 100644 --- a/plugins/evals/skills/validate/SKILL.md +++ b/plugins/evals/skills/validate/SKILL.md @@ -107,7 +107,7 @@ it must pass, near misses it must reject, an empty answer, and an answer to a di | A setting the script cannot reproduce offline, such as the `y` or `v` regex flag | WARN | | An `llm` or `baseline` grader with samples: they need a paid judge calibration run | WARN | -The script never calls a judge. Samples on an `llm` grader are the labelled answers a calibration +The script never calls a judge. Samples on an `llm` grader are the labeled answers a calibration run feeds the judge, and that run is the operator's to start. **Claim:** the sample check grades as the binary does. **Basis:** the grader code in Claude Code From 805510938fb6d92655ed480578adad1567cc46b0 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Fri, 2 Oct 2026 17:21:00 -0400 Subject: [PATCH 10/11] fix(evals): leave unchecked runs out of judge calibration agreement calibrate-judge recorded a run with neither a kept trace nor judge evidence as unchecked, then still counted it as judged and possibly agreeing, so a run never shown to reproduce the labeled sample could lift a grader over the trust bar. It now stops at the unchecked record; a sample whose runs are all unchecked reports as untested. render-review's md_cell docstring now states the contract the CHANGELOG already records: no image and no link hidden behind other text, while a bare URL may autolink showing its own address. A test pins that. Co-Authored-By: Claude Opus 5.5 --- .../skills/design/scripts/render-review.py | 3 +- .../design/scripts/test_render_review.py | 7 +++++ plugins/evals/skills/plugin-eval/SKILL.md | 6 ++-- .../plugin-eval/scripts/calibrate-judge.py | 3 +- .../fixtures/calibrate-judge/result.json | 4 +++ .../scripts/test_calibrate_judge.py | 29 +++++++++++++++++-- 6 files changed, 45 insertions(+), 7 deletions(-) diff --git a/plugins/evals/skills/design/scripts/render-review.py b/plugins/evals/skills/design/scripts/render-review.py index 10e787137f..b38c859346 100755 --- a/plugins/evals/skills/design/scripts/render-review.py +++ b/plugins/evals/skills/design/scripts/render-review.py @@ -56,7 +56,8 @@ def as_text(value): def md_cell(value): - """One table cell: a single line, HTML-inert, with no live link or image.""" + """One table cell: a single line, HTML-inert, with no image and no link hidden + behind other text. A bare URL may still autolink, but it shows its own address.""" flat = re.sub(r"([\\\[\]|])", r"\\\1", " ".join(as_text(value).split())) return html.escape(flat, quote=False) diff --git a/plugins/evals/skills/design/scripts/test_render_review.py b/plugins/evals/skills/design/scripts/test_render_review.py index 705eb6bd02..82b6c6606c 100755 --- a/plugins/evals/skills/design/scripts/test_render_review.py +++ b/plugins/evals/skills/design/scripts/test_render_review.py @@ -121,6 +121,13 @@ def test_link_and_image_syntax_in_cells_is_inert(self): self.assertIn(r"!\[\](https://x.example/p.png) \[ok\](https://y.example)", row) self.assertIn(r"a\\\|b", row) + def test_a_bare_url_stays_its_own_visible_text(self): + cases = [{"id": 1, "name": "see https://x.example/a and www.y.example"}] + code, text, _ = run(cases, "markdown") + self.assertEqual(code, 0) + _, _, row = table_rows(text) + self.assertIn("see https://x.example/a and www.y.example", row) + def test_one_fenced_block_per_case_longer_than_any_backtick_run(self): tricky = "before\n`````\n\nafter" cases = [{"id": 1, "prompt": tricky}, {"id": 2, "input": "plain"}] diff --git a/plugins/evals/skills/plugin-eval/SKILL.md b/plugins/evals/skills/plugin-eval/SKILL.md index e08cf55453..c7c160d935 100644 --- a/plugins/evals/skills/plugin-eval/SKILL.md +++ b/plugins/evals/skills/plugin-eval/SKILL.md @@ -392,9 +392,9 @@ about another. `score` prints a `FAIL grader` line for each grader under 90% and exits 1; fix that rubric, or move to a stronger judge, and calibrate again before reading its scores. Its false positives and negatives name the samples to read first. A run whose reply was not the sample is left out of the -agreement; one with neither a kept trace nor judge evidence is reported unchecked. A sample with no -reproduced run is listed as `untested`, counts toward no agreement, and shows in the verdict line; -raise `--runs` or tighten the prompt before reading the grader's score. +agreement, and so is one with neither a kept trace nor judge evidence, which is also reported +unchecked. A sample with no reproduced run is listed as `untested`, counts toward no agreement, +and shows in the verdict line; raise `--runs` or tighten the prompt before reading the grader's score. - **Pointer**: what a judge reads for each `focus`, see ; the 90% bar, see diff --git a/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py index 0dff0c4582..91b3178702 100755 --- a/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py +++ b/plugins/evals/skills/plugin-eval/scripts/calibrate-judge.py @@ -574,7 +574,8 @@ def score_run(tally, entry, run, where, base): tally.checked[source] += 1 if reply is None: tally.unchecked.append(where) - if reply is not None and not same_text(reply, entry["answer"]): + return + if not same_text(reply, entry["answer"]): tally.unreproduced.append( '%s: reply begins "%s" (%s)' % (where, first_line(reply), source) ) diff --git a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/result.json b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/result.json index f5b5b2cccc..18fa6863e3 100644 --- a/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/result.json +++ b/plugins/evals/skills/plugin-eval/scripts/fixtures/calibrate-judge/result.json @@ -80,6 +80,10 @@ "score": 0, "error": null, "skippedPaidGraders": false, "graders": [{"name": "names-paris", "passed": false, "explanation": "judge votes: FAIL FAIL FAIL", "judgeVotes": [false, false, false], "evidence": "The capital of France is Marseille."}] }, + { + "score": 0, "error": null, "skippedPaidGraders": false, + "graders": [{"name": "names-paris", "passed": false, "explanation": "judge votes: FAIL FAIL FAIL", "judgeVotes": [false, false, false], "evidence": "The capital of France is Marseille."}] + }, { "score": 0, "error": null, "skippedPaidGraders": false, "graders": [{"name": "names-paris", "passed": false, "explanation": "judge votes: FAIL FAIL FAIL", "judgeVotes": [false, false, false], "evidence": "The capital of France is Marseille."}] diff --git a/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py b/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py index f1b2be41cc..1d6ec4277c 100755 --- a/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py +++ b/plugins/evals/skills/plugin-eval/scripts/test_calibrate_judge.py @@ -416,7 +416,7 @@ def test_fixture_meets_the_target_at_exactly_ninety_percent(self): out, ) self.assertIn( - "reproduction: 2 checked against traces, 8 against judge evidence, 1 unchecked", + "reproduction: 2 checked against traces, 9 against judge evidence, 1 unchecked", out, ) self.assertNotIn("FAIL grader", out) @@ -446,6 +446,30 @@ def test_a_sample_with_no_reproduced_run_is_untested_and_not_counted(self): "verdict: PASS (every grader at or above 90%; 1 untested)", ) + def test_an_unchecked_run_is_excluded_from_the_agreement(self): + result = copy.deepcopy(self.result) + del self.rows(result, 4)[3]["graders"][0]["evidence"] + proc = self.score(result) + out = proc.stdout + self.assertIn("agreement 8/9 runs (88.9%) over 4 samples", out) + self.assertIn("8 against judge evidence, 2 unchecked", out) + self.assertIn( + "reproduction unchecked, no trace or evidence: capital-city--names-paris--04 " + "with-arm run 4", + out, + ) + + def test_a_sample_with_only_unchecked_runs_is_untested(self): + result = copy.deepcopy(self.result) + for row in self.rows(result, 1): + row.pop("tracePath", None) + for grader in row["graders"]: + grader.pop("evidence", None) + proc = self.score(result) + out = proc.stdout + self.assertIn("samples with no reproduced run: 1 of 4", out) + self.assertIn("untested: capital-city--names-paris--01", out) + def test_a_false_positive_drops_the_grader_under_the_target(self): result = copy.deepcopy(self.result) grader = self.rows(result, 4)[0]["graders"][0] @@ -508,9 +532,10 @@ def test_a_trace_focus_never_reads_the_evidence(self): path.write_text(json.dumps(manifest)) proc = self.score(manifest=path) self.assertIn( - "reproduction: 2 checked against traces, 0 against judge evidence, 9 unchecked", + "reproduction: 2 checked against traces, 0 against judge evidence, 10 unchecked", proc.stdout, ) + self.assertIn("agreement 1/1 runs (100.0%) over 1 samples", proc.stdout) def test_a_case_missing_from_the_result_is_named(self): result = copy.deepcopy(self.result) From 1479903f99bf8d97515b87c849a966c9cff467ec Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Fri, 2 Oct 2026 17:21:06 -0400 Subject: [PATCH 11/11] fix(skill-quality): pair probes when computing the compare interval measure-invocation compare scored baseline and treatment on the same probes but built its 95% interval as if the two trigger rates were independent, which widens it and can call a consistent gain noise. The interval now comes from per-probe deltas over the positive probes both reports share, and is null when the reports cannot be paired. Co-Authored-By: Claude Opus 5.5 --- .../reference/invocation-probes.md | 13 ++--- .../scripts/measure-invocation.sh | 32 ++++++++---- .../scripts/measure-invocation.test.sh | 51 ++++++++++++------- 3 files changed, 63 insertions(+), 33 deletions(-) diff --git a/plugins/skill-quality/reference/invocation-probes.md b/plugins/skill-quality/reference/invocation-probes.md index bb5975eff7..5d2d27afb9 100644 --- a/plugins/skill-quality/reference/invocation-probes.md +++ b/plugins/skill-quality/reference/invocation-probes.md @@ -117,12 +117,13 @@ selected). `compare` prints treatment minus baseline for both rates on both splits so a rewrite cannot hide a validation drop behind a train gain. Each trigger-rate delta also carries `trigger_rate_delta_interval`, a 95% -normal-approximation interval over the baseline and treatment `n_positive` for that split, -clamped to [-1, 1] and null when either rate falls outside [0, 1], and -`trigger_rate_within_noise`, true when that interval contains 0; stderr repeats it -as one INFO line per split. With about 4 to 6 positives per split, only a large -delta clears noise. When both rates are 0 or 1 the interval has zero width, so -read the probe count before trusting it. +paired normal-approximation interval over the per-probe hit changes of the positive +probes, matched by id between the two reports and clamped to [-1, 1]. It is null when +fewer than 2 probes pair or the positive probe ids differ between the reports. +`trigger_rate_within_noise` is true when that interval contains 0; stderr repeats it +as one INFO line per split. With about 4 to 6 positives per split, only a consistent +shift clears noise. When no probe changes, or every probe changes the same way, the +interval has zero width, so read the probe count before trusting it. `emit-plugin-eval` writes `runs: 3` per case, the CLI's default; `--runs N` changes it. diff --git a/plugins/skill-quality/scripts/measure-invocation.sh b/plugins/skill-quality/scripts/measure-invocation.sh index 7a0116ea35..de2f96213c 100755 --- a/plugins/skill-quality/scripts/measure-invocation.sh +++ b/plugins/skill-quality/scripts/measure-invocation.sh @@ -20,9 +20,10 @@ # # validate WARNs when a should-trigger probe shares N or more consecutive # words (default 4) with the target listing. emit-plugin-eval writes N runs -# per case (default 3, the CLI's own default). compare adds a 95% -# normal-approximation interval on each trigger-rate delta and an INFO line -# that says "within noise" when the interval contains 0. +# per case (default 3, the CLI's own default). compare adds a 95% paired +# normal-approximation interval on each trigger-rate delta, from per-probe +# outcomes matched by id, and an INFO line that says "within noise" when the +# interval contains 0. # # Probe files: /*.json (not baselines/). Each file is one skill: # skill, plugin, skill_dir (repo-relative), competitors[], queries[] @@ -459,12 +460,23 @@ cmd_compare() { report="$(jq -n --slurpfile b "$base" --slurpfile t "$treat" ' def delta($t; $b): if $t == null or $b == null then null else $t - $b end; def r3: . * 1000 | round / 1000; - # 95% normal-approximation interval on the difference of two trigger rates. - def interval($t; $b; $nt; $nb): - if $t == null or $b == null or ($nt // 0) == 0 or ($nb // 0) == 0 - or $t < 0 or $t > 1 or $b < 0 or $b > 1 then null - else (($t * (1 - $t) / $nt) + ($b * (1 - $b) / $nb) | sqrt * 1.96) as $h - | [([$t - $b - $h, -1] | max | r3), ([$t - $b + $h, 1] | min | r3)] end; + # 95% paired normal-approximation interval on the trigger-rate delta: both + # reports score the same positive probes, so the per-probe deltas + # (treatment hit - baseline hit, hit = predicted) are the sample. Null when + # fewer than 2 probes pair or the positive probe ids differ between reports. + def interval($bs; $ts; $s): + def pos($r): [$r.cases[]? | select(.split == $s and .expect_trigger == true)]; + def hit: if .predicted == true then 1 else 0 end; + pos($bs) as $bc | pos($ts) as $tc + | ($bc | map(.id) | sort) as $bi + | if $bi != ($tc | map(.id) | sort) or ($bi | length) < 2 or ($bi | unique | length) != ($bi | length) + then null + else ($bc | map({key: .id, value: hit}) | from_entries) as $bh + | [$tc[] | hit - $bh[.id]] as $d + | ($d | length) as $n + | ($d | add / $n) as $m + | ((($d | map(. - $m | . * .) | add) / ($n - 1)) | sqrt * 1.96 / ($n | sqrt)) as $h + | [([$m - $h, -1] | max | r3), ([$m + $h, 1] | min | r3)] end; def noise($i): if $i == null then null else ($i[0] <= 0 and $i[1] >= 0) end; ($b[0].skills) as $bs | ($t[0].skills) as $ts | { @@ -475,7 +487,7 @@ cmd_compare() { ($bs[] | select(.skill == $t.skill)) as $bb | def split($s): $bb.splits[$s] as $x | $t.splits[$s] as $y - | interval($y.trigger_rate; $x.trigger_rate; $y.n_positive; $x.n_positive) as $i + | interval($bb; $t; $s) as $i | { trigger_rate_baseline: $x.trigger_rate, trigger_rate_treatment: $y.trigger_rate, diff --git a/plugins/skill-quality/scripts/measure-invocation.test.sh b/plugins/skill-quality/scripts/measure-invocation.test.sh index cd0c1e0447..6bf5d5d78b 100755 --- a/plugins/skill-quality/scripts/measure-invocation.test.sh +++ b/plugins/skill-quality/scripts/measure-invocation.test.sh @@ -327,8 +327,9 @@ else fail "emit-plugin-eval --runs 0 should exit 2 (rc=$rc): $out" fi -# compare carries a normal-approximation interval on each trigger-rate delta -# and says "within noise" when the interval contains 0. +# compare carries a paired normal-approximation interval on each trigger-rate +# delta, from per-probe outcomes matched by id, and says "within noise" when +# the interval contains 0. cmp_out="$(run compare "$TMP/score.json" "$TMP/score.json" 2>"$TMP/cmp.err")" if jq -e '.skills[0].validation.trigger_rate_within_noise == true and (.skills[0].validation.trigger_rate_delta_interval | length == 2)' <<<"$cmp_out" >/dev/null && @@ -337,27 +338,43 @@ if jq -e '.skills[0].validation.trigger_rate_within_noise == true else fail "self-compare should report an interval and within noise: $cmp_out $(cat "$TMP/cmp.err")" fi -jq '.skills[0].splits.validation |= (.n_positive = 100 | .trigger_rate = 0.2)' "$TMP/score.json" >"$TMP/big-base.json" -jq '.skills[0].splits.validation |= (.n_positive = 100 | .trigger_rate = 0.8)' "$TMP/score.json" >"$TMP/big-treat.json" -cmp_out="$(run compare "$TMP/big-base.json" "$TMP/big-treat.json" 2>"$TMP/cmp.err")" +# Rewrites the validation cases of the fixture report into $2 positive probes +# named with prefix $3, of which the first $4 are hits. Output file: $1. +paired_report() { + jq --argjson n "$2" --arg prefix "$3" --argjson hits "$4" \ + '.skills[0].cases |= (map(select(.split != "validation")) + + [range(0; $n) | {id: ($prefix + tostring), split: "validation", expect_trigger: true, + predicted: (. < $hits), correct: (. < $hits)}])' "$TMP/score.json" >"$1" +} +paired_report "$TMP/pair-base.json" 100 p 0 +paired_report "$TMP/pair-treat.json" 100 p 10 +cmp_out="$(run compare "$TMP/pair-base.json" "$TMP/pair-treat.json" 2>"$TMP/cmp.err")" if jq -e '.skills[0].validation.trigger_rate_within_noise == false and .skills[0].validation.trigger_rate_delta_interval[0] > 0' <<<"$cmp_out" >/dev/null && ! grep -q 'validation trigger_rate delta .* within noise' "$TMP/cmp.err"; then - pass "compare does not call a large delta over 100 probes within noise" + pass "compare does not call 10 of 100 probes flipping miss to hit within noise" +else + fail "10 of 100 positive probes flipping miss to hit should clear noise: $cmp_out $(cat "$TMP/cmp.err")" +fi +paired_report "$TMP/other-treat.json" 100 q 10 +cmp_out="$(run compare "$TMP/pair-base.json" "$TMP/other-treat.json" 2>/dev/null)" +if jq -e '.skills[0].validation.trigger_rate_delta_interval == null + and .skills[0].validation.trigger_rate_within_noise == null' <<<"$cmp_out" >/dev/null; then + pass "compare reports no interval when the probe ids differ between reports" else - fail "a 0.2 -> 0.8 delta over 100 probes should clear noise: $cmp_out $(cat "$TMP/cmp.err")" + fail "differing probe ids should give a null interval: $cmp_out" fi -jq '.skills[0].splits.validation |= (.n_positive = 4 | .trigger_rate = 0.25)' "$TMP/score.json" >"$TMP/small-base.json" -jq '.skills[0].splits.validation |= (.n_positive = 4 | .trigger_rate = 1)' "$TMP/score.json" >"$TMP/small-treat.json" -jq '.skills[0].splits.validation.trigger_rate = 1.25' "$TMP/score.json" >"$TMP/bad-treat.json" -cmp_small="$(run compare "$TMP/small-base.json" "$TMP/small-treat.json" 2>/dev/null)" -cmp_bad="$(run compare "$TMP/score.json" "$TMP/bad-treat.json" 2>/dev/null)" -if jq -e '.skills[0].validation.trigger_rate_delta_interval[1] == 1' <<<"$cmp_small" >/dev/null && - jq -e '.skills[0].validation.trigger_rate_delta_interval == null - and .skills[0].validation.trigger_rate_within_noise == null' <<<"$cmp_bad" >/dev/null; then - pass "compare clamps the interval to 1 and reports no interval for a rate above 1" +paired_report "$TMP/tiny-base.json" 4 p 0 +paired_report "$TMP/tiny-treat.json" 4 p 3 +paired_report "$TMP/one-base.json" 1 p 0 +paired_report "$TMP/one-treat.json" 1 p 1 +cmp_tiny="$(run compare "$TMP/tiny-base.json" "$TMP/tiny-treat.json" 2>/dev/null)" +cmp_one="$(run compare "$TMP/one-base.json" "$TMP/one-treat.json" 2>/dev/null)" +if jq -e '.skills[0].validation.trigger_rate_delta_interval[1] == 1' <<<"$cmp_tiny" >/dev/null && + jq -e '.skills[0].validation.trigger_rate_delta_interval == null' <<<"$cmp_one" >/dev/null; then + pass "compare clamps the interval to 1 and reports no interval for a single probe" else - fail "interval should clamp at 1 and be null for an out-of-range rate: $cmp_small $cmp_bad" + fail "interval should clamp at 1 and be null for one probe: $cmp_tiny $cmp_one" fi # validate warns when a should-trigger probe copies 4+ consecutive words of