diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 8951d5b198..10367db727 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -690,17 +690,6 @@ "skill" ] }, - { - "name": "provenance", - "source": "./plugins/provenance", - "description": "Deprecated: renamed to attribution. Install attribution@melodic-software; this shim only points at /attribution:audit and /attribution:setup and is removed in a later release.", - "category": "quality", - "tags": [ - "deprecated", - "provenance", - "attribution" - ] - }, { "name": "writing", "source": "./plugins/writing", diff --git a/.claude/ai-slop.json b/.claude/ai-slop.json index bbf90a712d..5f299ff2a2 100644 --- a/.claude/ai-slop.json +++ b/.claude/ai-slop.json @@ -1,5 +1,5 @@ { - "_comment": "Em dashes are not this repository's house style: rule-em-dash runs at its shipped zero-tolerance default everywhere except em_dash_allowed_paths. Two groups are in em_dash_allowed_paths. The per-source ledgers named under docs/upstream are first-party prose, this repository's own verdicts and decision tables about upstream material; they are not snapshots and not vendored text. Their entries are a temporary deferral of the purge, not a judgment that em dashes belong there: each is listed by file, deleted in the change that purges that file, and added then to scripts/em-dash-purged-paths.txt, and every other file under docs/upstream stays at zero tolerance. The research memos and decision verdicts under .claude/unhobble/*/evidence are the frozen record of one experiment run and must not be rewritten, which is the one exemption here that rests on something other than purge cost; the experiment's live ledger and README are not exempt. Vendored upstream material is the prose this repository does not author, so plugins/*/skills/*/vendor/** is excluded. Two rules stay disabled, each on a specific owner ruling rather than a count: (1) rule-curly-artifacts, because punctuation glyphs were ruled out of scope twice in writing (plugins/songwriting/context/pat-pattison/research/book-references.md 'Do not sweep, measure, audit, or open work items on punctuation glyphs', and the songwriting CHANGELOG entry 'PUNCTUATION GLYPHS ARE NOT A FIDELITY AXIS'), and the rule's yield is almost entirely the quoted book text and the pasted external review those rulings protect; (2) rule-emoji-formatting, because every measured hit is a semantic marker (wrong/right teaching pairs, coaching pairs, warning headings, a severity legend) and none is decoration; this repository uses those glyphs as vocabulary. Both stay whole-rule disables because each is a repo-wide ruling, not a path-scoped one; revisit either if genuine LLM residue of that kind lands here. skills/audit/evals/fixtures/** is excluded because those files are the eval suite's committed slop samples: an excluded_paths glob is a layer of THIS repo's config, so the detector's own empty-HOME isolation lifts it and a fixture still reports its real findings, where an in-file ignore marker would silence it everywhere. The catalog is not excluded: it documents tells inside quotes and code spans, which the quotation exemption already declines, and its own prose is subject to every rule like any other authored file. Both density thresholds sit at 2.0 per 1000 words, below the shipped 3.0 and 4.0, so a regression surfaces as soon as it lands: a full-corpus scan on 2026-09-28 found the densest file at 1.6 for rule-ai-vocabulary and 0.2 for rule-copulative-avoidance (files with at least three hits), so 2.0 fires on nothing today. Re-measure before lowering further. The rule_allowed_paths entry is still load-bearing: without it the deepening research's vocabulary.md and interface-design.md fire rule-ai-vocabulary, because they discuss those words.", + "_comment": "Em dashes are not this repository's house style: rule-em-dash runs at its shipped zero-tolerance default everywhere except em_dash_allowed_paths. The per-source ledgers named under docs/upstream are first-party prose, this repository's own verdicts and decision tables about upstream material; they are not snapshots and not vendored text. Their entries are a temporary deferral of the purge, not a judgment that em dashes belong there: each is listed by file, deleted in the change that purges that file, and added then to scripts/em-dash-purged-paths.txt, and every other file under docs/upstream stays at zero tolerance. Vendored upstream material is the prose this repository does not author, so plugins/*/skills/*/vendor/** is excluded. Two rules stay disabled, each on a specific owner ruling rather than a count: (1) rule-curly-artifacts, because punctuation glyphs were ruled out of scope twice in writing (plugins/songwriting/context/pat-pattison/research/book-references.md 'Do not sweep, measure, audit, or open work items on punctuation glyphs', and the songwriting CHANGELOG entry 'PUNCTUATION GLYPHS ARE NOT A FIDELITY AXIS'), and the rule's yield is almost entirely the quoted book text and the pasted external review those rulings protect; (2) rule-emoji-formatting, because every measured hit is a semantic marker (wrong/right teaching pairs, coaching pairs, warning headings, a severity legend) and none is decoration; this repository uses those glyphs as vocabulary. Both stay whole-rule disables because each is a repo-wide ruling, not a path-scoped one; revisit either if genuine LLM residue of that kind lands here. skills/audit/evals/fixtures/** is excluded because those files are the eval suite's committed slop samples: an excluded_paths glob is a layer of THIS repo's config, so the detector's own empty-HOME isolation lifts it and a fixture still reports its real findings, where an in-file ignore marker would silence it everywhere. The catalog is not excluded: it documents tells inside quotes and code spans, which the quotation exemption already declines, and its own prose is subject to every rule like any other authored file. Both density thresholds sit at 2.0 per 1000 words, below the shipped 3.0 and 4.0, so a regression surfaces as soon as it lands: a full-corpus scan on 2026-09-28 found the densest file at 1.6 for rule-ai-vocabulary and 0.2 for rule-copulative-avoidance (files with at least three hits), so 2.0 fires on nothing today. Re-measure before lowering further. The rule_allowed_paths entry is still load-bearing: without it the deepening research's vocabulary.md and interface-design.md fire rule-ai-vocabulary, because they discuss those words.", "thresholds": { "ai_vocabulary": 2.0, "copulative_avoidance": 2.0 @@ -10,8 +10,7 @@ "docs/upstream/cursor-pstack.md", "docs/upstream/humanlayer-skills.md", "docs/upstream/mattpocock-skills.md", - "docs/upstream/mattpocock-skills-v12-map.md", - ".claude/unhobble/*/evidence/**" + "docs/upstream/mattpocock-skills-v12-map.md" ], "rule_allowed_paths": { "rule-ai-vocabulary": ["plugins/architecture/skills/improve/research/deepening/**"] diff --git a/.claude/settings.json b/.claude/settings.json index 4b04b81073..75407d08c2 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -41,7 +41,6 @@ "harness-ops@melodic-software": true, "pixel-art@melodic-software": false, "playgrounds@melodic-software": false, - "provenance@melodic-software": false, "retro-audio@melodic-software": false }, "extraKnownMarketplaces": { diff --git a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/README.md b/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/README.md deleted file mode 100644 index 3548233c59..0000000000 --- a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/README.md +++ /dev/null @@ -1,38 +0,0 @@ -# unhobble experiment: Fable 5.1 bare baseline - -Durable mirror for one run of `/claude-config:unhobble` (experiment id -`claude-code-plugins-fable-5-1-20260908-15744de6`). The snapshot ran in an ephemeral cloud -container; that container goes away when the session that opened pull request #4090 ends, and -the plugin-data copy of the manifest goes with it, so from then on this directory is the -canonical experiment record, not a copy. It lives under `.claude/` rather than the topic-docs -contract slice because the contract-slice gate red-lines any pull request that carries a slice, -and this record has to outlive the whole observe window. - -| File | Role | -|---|---| -| `stumbles.md` | The observation ledger. Append a row whenever the bare model stumbles, or does something better. | -| `manifest-mirror.json` | The Phase 1 manifest with absolute paths tokenized: every surface, its class, what was stripped, how to restore it, and the four resolved decisions. | -| `evidence/research-D*.md` | Source-backed research memos behind decisions D1 to D3 (official docs fetched 2026-09-11). | -| `evidence/decision-*.md` | Two independent decision agents' verdicts, one on Fable 5.1 and one on Opus 5, which agreed on all four decisions. | - -Observe phase: on 2026-09-12 the operator decided to merge #4090 so the window runs on main; -from that merge on, the bare state is the state of main (the manifest records this under -`checkout.merged_to_main`). Work normally on real tasks from fresh sessions on main and append a -row here whenever the model stumbles or does something better without an instruction. - -Restoring the canonical state before readd: the skill compares `checkout.worktree_path` with the -resolved absolute checkout before every phase command and aborts on a mismatch, so the mirror -cannot be copied back verbatim. Copy `manifest-mirror.json` to -`${CLAUDE_PLUGIN_DATA}/unhobble//manifest.json` with the `` token -replaced by the output of `git rev-parse --show-toplevel` and the `` token -replaced by the resolved plugin data directory, copy `stumbles.md` alongside it, and create an -empty `backups/` directory next to them. Keep the tokenized copy here unchanged. - -Readd phase: start a fresh branch off main, restore `.claude/rules/pr-body-contract.md` from -`git show c41c6422:.claude/rules/pr-body-contract.md` regardless of the ledger (register hold, -contested external-publication), restore only what the ledger defends with repeated same-cause -rows, re-render the AGENTS.md rules index with -`plugins/instruction-placement/scripts/render-index.sh write --file AGENTS.md`, set -`"phase": "closed"` in the manifest, and close the experiment. Every stripped surface is -recoverable from commit `c41c6422` (the pre-strip base); the manifest names each one. Whether this -directory is kept as the experiment's record or removed at close is the readd phase's call. diff --git a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/decision-fable.md b/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/decision-fable.md deleted file mode 100644 index 4d64ed9f2f..0000000000 --- a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/decision-fable.md +++ /dev/null @@ -1,191 +0,0 @@ -# Unhobble decisions (agent: Fable, independent) - -Experiment `claude-code-plugins-fable-5-1-20260908-15744de6`, branch `claude/unhobble-config-oe7yfx`, -HEAD c41c6422. Default strip plan in the manifest is accepted as-is except where D4 flags it. -Evidence tags: D1/D2/D3 = memo finding numbers; repo paths are relative to ``. - -## D1. User-scope opt-in - -**VERDICT:** Do not opt user scope in. The project-scope `enabledPlugins: false` overlay IS a -contract-consistent Phase 2 mechanism, but this run disables zero plugins through it: all 20 -hook-wiring plugins classify as keep, and the 53 skill-only plugins need a classification pass first. - -**CONFIDENCE:** high on "do not opt in" and on the 20-plugin classification; medium on whether a project -`false` would even take effect in-cloud (rests on D1 Unverified #1 plus a bootstrap gap found below). - -**EVIDENCE** - -- The in-container `~/.claude/settings.json` is environment-synthesized, not a human's standing - instructions: D1 §3 (cloud-environments "fresh VM", "user-scoped enabledPlugins ... not read"), - D1 §4 (`docs/cloud-sessions.md` ~372-388; `.claude/cloud-bootstrap.sh` 205-210). Ablating it measures - provisioning, not instructions, and edits there do not survive reclaim (D3 §1.3-1.4, U1/U2). -- Project `false` is documented per-project opt-out: D1 §2 (plugins-reference "Synced plugins"; - "`enabledPlugins` still honors project and local settings"). The catalog gate passes an explicit `false` - (`scripts/check-plugin-catalog-enablement.sh` header: "A key set to `false` PASSES"; keys must stay byte-sorted). - This is exactly the contract's Phase 2 "disable the ones classified behavioral for this project" - (SKILL.md Phase 2, plugins bullet), so it stays inside project scope. -- Gap that caps confidence: the bootstrap's overlay only drives INSTALLS (`cloud-bootstrap.sh` ~236-241, - `select(.value == true)`); the snapshot has already installed and enabled all 73 at user scope, and the - only uninstall path (~340-360) is the same-version refresh. So a project `false` on a fleet-enabled - plugin leaves user `true` in place and relies on Claude Code precedence (project over user), which no - doc states for this exact pair (D1 Unverified #1). Registry is built at process start and not re-read - (`docs/cloud-sessions.md` 287), so any effect lands in the next session only (D1 Unverified #3). -- Classification of the 20 hook plugins, from `plugins//hooks/hooks.json` + hook headers, against - `docs/plugin-philosophy.md` "Classifying a hook" (782-815): - - policy / keep whole: `guardrails` (secret-pattern, hardcoded-path, block-no-verify, block-dangerous-git - = secret-handling / irreversible-action; cli-flag-verify, skill-reference-verify, stale-path-verify are - the rubric's named ground-truth-oracle keeps), `source-control` (pr-body-linkage-gate, worktree gates), - `disk-hygiene` (destructive_guard = irreversible-action), `context-budget` (settings_write_ask = - agent-authority), `instruction-placement` (index-drift: machine oracle; manifest already depends on it - staying green), `actionlint` (relays measured linter findings = policy per rubric example). - - notification/infra, no model payload / keep: `claude-ops` (audit emitters, event log; default off), - `desktop-notification`, `rate-limit-guard`, `session-flow` (observer-arm; default off), - `autonomy` (lane-stop-gate; default OFF, honored from user/managed settings only, so inert here). - - deterministic-transform tooling / keep (operator's formatter carve-out): `bash-format`, `biome-format`, - `go-format`, `markdown-format`, `powershell-format`, `ruff-format`, `typos-format`, `eol-normalizer`. - - hybrid / keep whole, record `unstripped-hybrid-hook`: `context-guard`. zone-crossing-inject carries a - measured zone (oracle) plus an inline counter-steer prose payload into model context - (`hooks/zone-crossing-inject.sh` header lines 5-20 = behavioral surface); zone-gate is advisory-inert - by default. Its only kill switch `context_guard_hooks_enabled` is a master switch for all three hooks - and lives in `pluginConfigs`, which project settings cannot set (D1 §2), so no project-scope partial strip exists. -- 53 skill-only plugins: neither memo classifies them; descriptions load per turn (D1 §2, skills page) but - bodies only on invocation, and `skillListingBudgetFraction: 0.05` already caps the listing. Per the task - rule, no strip without a classification pass. Note `claude-config` must stay enabled regardless: it hosts - `/claude-config:unhobble` for Phases 3-4. Candidates the pass should look at first, on their own - descriptions: `discipline`, `playbooks`, `adhd`, `education` (pure coaching / doctrine); not stripped now. - -**WHAT WOULD CHANGE THIS VERDICT** - -1. A doc sentence or an in-cloud `claude plugin list --json` reading in a fresh session proving project - `false` beats user `true` for a snapshot-installed plugin; then option C becomes operable and the - classification pass can strip skill-only behavioral plugins. -2. The operator declaring the cloud-synthesized user file in scope as an environment surface (a different - experiment than the contract's; would need its own manifest entries). - -**PHASE 2 MECHANICAL STEPS (D1)** - -1. Add to `manifest.json` a `user_scope_plugins` block: per-plugin `class` and `action: keep` for the 20 - above, `context-guard` with `unstripped-hybrid-hook`, and the 53 skill-only ids as - `classification: pending`. No edit to `.claude/settings.json#enabledPlugins` this run. -2. In the first bare session, run `claude plugin list --json` and `/context`; append the loaded set to the - manifest as `bare_session_loaded` (D1 option F) so the observe phase knows its confounds. - -## D2. The four convention units - -**VERDICT:** Strip (a) "Validate a change" and (c) `pr-body-contract.md`; keep (b) "Open a PR as a draft" -and (d) `ruff-pin.md`. - -**CONFIDENCE:** high for (a) and (d); medium for (b) and (c) (Gate 0 class contested, D2 Unverified #1). - -**EVIDENCE** - -- The consensus row is the cut test with no exception (D2 consensus table rows 1, 3): the "conventions are - exempt" reading is repo-only and the repo's own `criteria.md` 1887 concedes it appears on no official - page. So the experiment, not the class label, decides; strip where a live oracle catches the miss and - keep where the miss is silent or matches a keep rubric. -- (a) strip: gating oracle `scripts/affected-tests.sh` runs in `ci.yml` `test-linux` under `ci-status` - (D2 §4a); the script fails loud on a zero-suite mapping itself; README "Validate a change" owns the - contract. Gate 0: none. Suppressed only while the PR is a draft, which the flip-to-ready resolves. -- (c) strip: four surviving oracles: the user-scope `source-control` hook - `plugins/source-control/hooks/pr-body-linkage-gate.sh` blocks a bare `gh pr create` whose body misses - the keyword or any of the four sections (header lines 12-25); `.claude/source-control.md` - `pr_body_required_sections`; the `/source-control:pull-request` pre-create gate; and the CI composite's - advisory comment + label (D2 §4c). Gate 0 contested-weak (D2 §3.5); handled at Phase 4 below. -- (b) keep: no oracle of any kind (D2 §4b: no CI job, no hook, no skill default); a miss is silent spend - on the test and two AI-review lanes; `external-publication` names "opening PRs" literally - (`plugins/instruction-placement/context/routing-rubric.md` 38). Thin evidence, so prefer the reversible, - contract-intact default (kept). -- (d) keep: non-derivable ground-truth carve-out (`docs/plugin-philosophy.md` 801-805, D2 §3.3): a bare - `ruff` returns a clean-looking wrong answer; `scripts/run-ruff.sh` exits 2/127 only when invoked; CI lint - is one subtree and fail-open (D2 §4d). Path-scoped to `**/*.py`, so its context cost is near zero and a - strip measures little. -- Phase 4 for (c): restore on two same-cause ledger rows as usual; if the readd pass affirms - `external-publication` for it, restore as a register hold instead (register README line 56), never left - deleted on silence. Always-loaded count after this plan: 50 - 6 - 25 = 19 lines. - -**WHAT WOULD CHANGE THIS VERDICT** - -1. A stated adjudication (rubric or register) that reversible publication-cost rules are inside - `external-publication`: then (c) is stripped only under a pre-declared register hold, and (b) may be - stripped under the same hold (D2 option 5). -2. Evidence the `source-control` hook is not loaded in the bare session (D1 confound reading): (c) then - loses its in-session oracle and should stay kept. - -## D3. Where the state lives - -**VERDICT:** Canonical location stays the plugin data dir per the contract; a durable mirror of -`manifest.json` and `stumbles.md` is committed on the experiment branch at -`docs/topics/unhobble-fable-5-1/`, synced each session; `backups/` never leaves the plugin data dir. - -**CONFIDENCE:** high that hand-carry alone is unsafe; medium-high on the path choice. - -**EVIDENCE** - -- The state must outlive the VM and the observe phase is days of sessions (SKILL.md Phase 3). Data dir is - lost at reclaim: D3 §1.1-1.4 (fresh VM, only conversation history restored, cache is the setup-script - snapshot), repo-normative matrix D3 §3.2 (`${CLAUDE_PLUGIN_DATA}` "invisible" to a cloud clone), sibling - precedent D3 §3.4 ("reclaimed container ... kept — it is tracked"). Only U1 is inference; a sentinel - under `~/.claude/plugins/data/` in the next session settles it. -- `docs/topics//` is the named contract tier: "committed on the task branch only; pruned before - merge" (`docs/conventions/topic-docs/README.md` 65, 656-679); sole entry in `scripts/docs-only-paths.txt`; - not gitignored (D3 §5.5); outside the changelog-parity arms `plugins/*/*` and - `docs/conventions/*/CHANGELOG.md` (`scripts/check-changelog-parity.sh` 480, 484), so no version bump per - ledger write (D3 §5.2). Committing is unaddressed, not forbidden, by the contract (D3 §4). -- Frictions: F1 absolute worktree path vs the machine-specific-paths gate (`ci.yml` 486-507, exclude list is - hand-maintained); F2 `backups/` vs gitleaks; F3 Phase 2 already commits to the branch. A fourth found here: - `scripts/check-contract-slice-prune.sh --check-diff` (`ci.yml` 982) red-lines any PR whose diff adds a - path under `docs/topics/` until the prune commit. That is the convention's designed state and is the - gate guaranteeing run state never reaches `main`. -- F2 dissolves for this run: no non-tracked file is modified (settings hooks kept, no `enabledPlugins` - edit per D1), so `backups/` is empty; the data dir currently holds `backups/`, `manifest.json`, `stumbles.md`. - -**WHAT WOULD CHANGE THIS VERDICT** - -1. The sentinel survives into the next session (U1 false): the mirror becomes optional insurance. -2. A later phase must edit an untracked settings file: `backups/` then stays data-dir-only and the - manifest records the backup as non-durable (operator hand-carry for that one artifact). - -**PHASE 2 MECHANICAL STEPS (D3)** - -1. `mkdir -p docs/topics/unhobble-fable-5-1/`; copy `manifest.json` and `stumbles.md` there. In the committed - copy drop `checkout.worktree_path` (keep `origin_url`, `branch`, `base_commit`); add - `state_location: {canonical: "${CLAUDE_PLUGIN_DATA}/unhobble//", durable_mirror: - "docs/topics/unhobble-fable-5-1/", deviation: "committed copy omits the absolute worktree path; - sync re-derives it"}` to both copies. Ensure final newline and a valid table for markdownlint/editorconfig; - typos will spell-check ledger prose. -2. Commit separately from the strip commit: `experiment: carry unhobble state on the branch`. Push. -3. Session-start sync (every fresh session): `mkdir -p "$CLAUDE_PLUGIN_DATA/unhobble/"`, copy both files - in, `jq --arg p "$(git rev-parse --show-toplevel)" '.checkout.worktree_path=$p'` into the data-dir copy; - the contract's identity check then compares as written. Session-end: copy `stumbles.md` and - `manifest.json` (minus the path) back, commit, push. -4. Phase 4 close: final commit prunes `docs/topics/unhobble-fable-5-1/` with the pre-prune SHA named in the - PR body (topic-docs lifecycle step 4); the contract-slice gate goes green; `backups/` needs no action. - -## D4. Sanity check of the default plan - -**VERDICT:** No class changes to any strip/trim/regenerate unit; four annotations to the manifest. - -**CONFIDENCE:** high. - -**EVIDENCE** - -- `agents-draft-pr` (b): add `register_class_contested: external-publication` (D2 §3.5; rubric line 38). - Kept, so no Phase 2 effect, but Phase 4 must not treat a future strip of it as silence-deletable. -- `rule-pr-body-contract` (c): same annotation, marked weaker; add `surviving_oracles` naming the - source-control hook, `.claude/source-control.md`, and the CI composite (D2 §4c), since that is why it may - be stripped at all. -- `rule-ruff-pin` (d): reason should cite the ground-truth-oracle carve-out (PLUGIN-PHILOSOPHY 801-805) - rather than only "tooling convention"; class stays `convention`, `operator_optional_strip` may stay but - the carve-out makes a strip low-value. -- Manifest `strip_plan_summary` counts: recompute `always_loaded_lines_after` to 19 for this plan and add a - `user_scope_plugins` block (D1 step 1). `ephemeral_note` is superseded by `state_location` (D3). -- Units checked and unchanged: `rule-worktree-base-ref` (consequence is a reviewable settings key, git - history is the oracle; no Gate 0 class), `nested-autonomy` (CI typos gate is the oracle; it is a - contributor workaround, not a rail), `rule-vendor-docs`, `rule-catalog-taxonomy`, `rule-hook-budget` - (pointer-only; owner docs stay), `rule-skill-bodies` and `nested-provenance` trims (residues kept), - `nested-model-adaptation` (legal-compliance hold, correct), `settings-permissions-deny` (secret-handling). - -**WHAT WOULD CHANGE THIS VERDICT** - -1. A Gate 0 adjudication for (b)/(c) (see D2): the annotations become `register_class`. -2. The bare-session load reading shows the `source-control` hook absent: (c) moves back to keep. diff --git a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/decision-opus.md b/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/decision-opus.md deleted file mode 100644 index cc84411a8d..0000000000 --- a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/decision-opus.md +++ /dev/null @@ -1,269 +0,0 @@ -# Unhobble run decisions (independent agent: Opus 5) - -Experiment: `claude-code-plugins-fable-5-1-20260908-15744de6`. Operator standing instruction: -accept/resolve where consensus research from authoritative sources backs it; where evidence is -thin or inference-only, prefer reversibility and contract intactness, and say so. - ---- - -## D1. User-scope opt-in - -**VERDICT: No. Do not ablate user scope this run, and do not ablate the 73 plugins via the -project-scope `false` overlay either; record all 73 as confounds. The overlay IS the correct -mechanism for a future plugin ablation, but it is gated on a Phase 1 classification pass that has -not run.** - -CONFIDENCE: **high** (not opting user scope in) / **medium-high** (overlay is the right future -mechanism). - -EVIDENCE - -- Contract, `SKILL.md` "Scope and safety rails": user-global surfaces "are included only when the - operator explicitly opts in per phase-1 prompt, never by default." No opt-in was given. -- D1 finding 3 + consensus row: a real user's `~/.claude/settings.json` never reaches a cloud - session (code.claude.com/docs/en/cloud-environments "What carries over"; - code.claude.com/docs/en/settings "Settings in cloud sessions"). The in-container - `~/.claude/settings.json` is environment-synthesized by the snapshot plus - `.claude/cloud-bootstrap.sh`, not a human's standing instruction. Ablating it measures the - provisioning layer, not the operator's instruction surface. -- D1 finding 3 + D3 findings 1.1/1.3/1.4: every session gets a fresh VM; the only persisted - filesystem is the setup-script snapshot (~7-day expiry). A user-scope edit made in this session - is **self-reverting** at the next session, so it cannot hold across an observe window measured in - days. It buys no durable arm. -- Contract Phase 1: classification is required for "**every surface the strip plan will touch** - ... and project-enabled plugins alike." No per-plugin classification exists for the 73. The - manifest records only names (`confounds[0].with_hooks`, 20 entries). Per the task's own rule, - unclassifiable plugins stay kept. -- Contract Phase 2 plugin rule: a plugin with any `policy` surface alongside behavioral components - is **hybrid, kept whole**. Every one of the 20 hook-wiring plugins wires at least one gate-shaped - hook, so the prior for each is "hybrid → keep whole", with only per-hook kill switches available. - Named policy keeps the task lists (guardrails, source-control, rate-limit-guard) and the - deterministic tooling lane (actionlint, bash-format, biome-format, eol-normalizer, go-format, - markdown-format, powershell-format, ruff-format, typos-format) all fall on the keep side of that - rule on the evidence available; the remaining eight (autonomy, claude-ops, context-budget, - context-guard, desktop-notification, disk-hygiene, instruction-placement, session-flow) cannot be - classified from the memos or the manifest and therefore stay kept pending the pass. -- The overlay mechanism is real and durable: `scripts/check-plugin-catalog-enablement.sh` header — - "A key set to `false` PASSES. An explicit `false` is a recorded decision"; and - `.claude/cloud-bootstrap.sh` (~line 205) computes its install set as fleet overlaid with - `.claude/settings.json`, selecting `.value == true`, "so a repo entry set to false opts out of a - fleet entry." Committed, it survives every fresh VM. This is the contract's own Phase 2 spelling - ("disable the ones classified `behavioral` **for this project**"). -- But the outcome of `true` at user + `false` at project is **inference, not documentation** (D1 - Unverified #1: no page states that pair), and mid-session enablement changes are unverified - (D1 Unverified #3; `docs/cloud-sessions.md` ~line 286 records the registry is built at process - start and not re-read). Thin evidence → keep the contract intact. - -WHAT WOULD CHANGE THIS VERDICT - -1. A completed per-plugin classification pass (hooks.json + README per plugin) plus an official doc - or an empirical check confirming project `false` beats user `true` — then disable the - behavioral-classified subset via the committed overlay. -2. Operator explicitly opting user scope in at the Phase 1 prompt, accepting that the strip must be - re-applied every session and that the arm measures provisioning, not instructions. - -PHASE 2 MECHANICAL STEPS (D1) - -1. Touch **no** plugin enablement: `.claude/settings.json#enabledPlugins` stays `{fleet: true, - playgrounds: false}`. Do not edit `~/.claude/settings.json`. -2. Leave `manifest.scope.user_global_surfaces = false`; append to `scope.user_global_note`: "Opt-in - declined for this run; user-scope edits do not survive a fresh VM, and no classification pass - exists for the 73." -3. Promote the confound to a first-class experiment limitation: in `confounds[0]`, add - `"classification_pass": "required before any of these may be stripped"` and - `"durable_mechanism": ".claude/settings.json enabledPlugins : false (committed; bootstrap - honors settings-wins; catalog-enablement gate passes on an explicit false, keys must stay byte- - sorted)"`. -4. After the strip, in the fresh session, run `/context` and `claude plugin list --json` and record - what actually loaded in the manifest (D1 option F). Do not call any arm "stripped" on intent. - ---- - -## D2. The four convention units - -**VERDICT: Strip (a) "Validate a change" and (c) `pr-body-contract.md`; keep (b) "Open a pull -request as a draft" and (d) `ruff-pin.md`.** - -CONFIDENCE: **high** for (a) and (d); **medium-high** for (b); **medium** for (c). - -EVIDENCE - -- (a) **strip.** D2 §4(a): `scripts/affected-tests.sh` is run by `ci.yml` job `test-linux`, which is - in `ci-status.needs`, the single required check — a **gating** oracle. `README.md` "Validate a - change" owns the full contract and survives the strip; the script self-documents (`--explain`, - "FAIL LOUD, NOT OPEN"). D2 §3.5 Gate 0: no class (cost is wall clock). This is the exact - derivable-with-a-live-oracle shape the experiment exists to measure. -- (c) **strip.** D2 §4(c): the `pr-contract` composite leaves an advisory comment plus a - `needs-issue-linkage` label within one CI run — a real, fast correction signal the bare model - receives. The section list survives in `.claude/source-control.md` - (`pr_body_required_sections`), which is **not** in the strip plan, so the strip tests whether the - model finds the surviving owner. Gate 0 is contested-weak: by the rubric's own "recognition is by - consequence, not by phrasing" (`plugins/instruction-placement/context/routing-rubric.md` line - ~42), a removable comment and a removable label is a bounded, reversible consequence. -- (b) **keep.** D2 §4(b): **no oracle at all** — no CI job, no hook, no skill default. A violation - is silent and self-inflicted CI spend, and this is a cloud session that opens PRs. Gate 0 is - contested toward `external-publication` (the rubric's literal example is "opening PRs"), which - means that under contract Phase 4 step 4 it would be **restored regardless of the ledger** as a - register hold. Stripping it therefore purchases cost with no evidentiary return. Thin/contested - evidence → keep. -- (d) **keep.** D2 §3.3, `docs/plugin-philosophy.md` "Classifying a hook": "a hook with a - behavioral purpose but a non-derivable ground-truth oracle ... is a keep, not an ablation - candidate." D2 §4(d): `scripts/run-ruff.sh` exits 2 on drift but **127/SKIP when the pin is - unavailable** (fail-open), CI's only ruff lane covers one subtree - (`plugins/source-control/skills/babysit-prs/**`), and nothing detects a local bare `ruff`. A bare - model typing `ruff check` gets a clean-looking wrong answer with no error. The pin fact (a - release moving rules into defaults) is machine ground truth no model can know unaided. -- Cross-check on the carve-out the keeps rest on: D2 §1.8 and the consensus table record that - "team conventions in git" is **not** an officially stated exemption from the cut test — the - repo's own `criteria.md` line 1887 concedes it. So (b) and (d) are kept on their own evidence - (no oracle; non-derivable oracle), not on the convention class alone. (a) and (c) are stripped - precisely because the convention label was doing the work for them. - -WHAT WOULD CHANGE THIS VERDICT - -1. (b): a detector landing (a `source-control` skill default or a CI check that flags a non-draft - new PR) would make it strippable with a real feedback loop; or an adjudication that a reversible - publication-cost rule is outside `external-publication`. -2. (c): evidence that the `pr-contract` composite at SHA `5776760…` actually **gates** rather than - advises on body/linkage (D2 Unverified #2) would flip it to keep, since the gate would then - already carry the correction and the strip would measure nothing new. -3. (d): a repo-wide, fail-closed ruff lane in `ci.yml` would remove the non-derivable-oracle carve- - out and make it strippable. - ---- - -## D3. Where experiment state lives - -**VERDICT: Split. Keep `${CLAUDE_PLUGIN_DATA}/unhobble//` as the contract-canonical home, and -commit a durable mirror on the experiment branch at `docs/topics/unhobble-fable-5-1/` — -`stumbles.md` (the primary ledger) and `manifest-mirror.json` (identity by `origin_url` + `branch` + -`base_commit`, absolute path replaced by the literal token `${CLAUDE_PROJECT_DIR}`). `backups/` -never leaves the plugin data dir.** - -CONFIDENCE: **medium-high**. - -EVIDENCE - -- Durability is not optional here: D3 findings 1.1/1.3/1.4 (fresh VM per session; reclaim restores - conversation history, not disk; the only persisted filesystem is the post-setup-script snapshot, - ~7-day expiry) and the repo's own **normative** matrix, `docs/conventions/topic-docs/README.md` - "Visibility across execution contexts": cloud clone sees `${CLAUDE_PLUGIN_DATA}` as **invisible** - and carries "pushed commits only". `SKILL.md` Phase 3 calls `stumbles.md` "the experiment's - entire evidentiary output"; option (i) makes that output depend on an ungated manual hand-carry, - twice per session, over days. -- Committing is unaddressed, not forbidden: D3 finding 4 — `SKILL.md` "What this skill does NOT do" - lists four items, none about state location. Keeping the plugin-data dir canonical means no - contract clause is bent; the mirror is an additive deviation, recorded in the manifest. -- `docs/topics//` is the repo's **contract tier** — "Committed on the task branch only; - pruned before merge" (`docs/conventions/topic-docs/README.md`), with a documented prune-with- - pointer lifecycle. It is deliberately **not** the memory tier (`.work/`, self-ignoring, lost to a - reclaimed container — `.gitignore` carries `.work/`), which is the tier that would lose the - ledger. `scripts/docs-only-paths.txt` describes this exact prefix as "Operator precedent- - codification & session working docs." -- **Changelog-parity clearance:** `scripts/check-changelog-parity.sh` case arms are `plugins/*/*` - and `docs/conventions/*/CHANGELOG.md`. `docs/topics/` matches neither, so no version bump and no - release entry is demanded per state write. State under `plugins/claude-config/` would demand both - (D3 finding 5.2) — that path is rejected on this ground. -- **Friction F1 (machine-specific-paths gate vs the recorded absolute worktree path):** resolved by - keeping the absolute path only in the plugin-data manifest, which is never committed, and - tokenizing it in the mirror. No ci.yml exception-list edit is needed. Identity still holds: - `origin_url` + `branch` + `base_commit` identify the checkout, and the canonical manifest the - later phases verify against is still the plugin-data one. -- **Friction F2 (gitleaks vs `backups/`):** does not bite this run — the confirmed strip plan - modifies **no** non-tracked file (`strip_whole`, `trim` and `regenerate` are all tracked paths; - every `settings.json` surface is `action: keep`), so `backups/` is empty. The rule stands anyway: - `backups/` stays out of git, since it would hold `.claude/settings.local.json`-class content the - repo gitignores. -- **Friction F3 (Phase 2 already commits):** an advantage, not a cost. The branch is already the - carrier for the stripped surfaces; the mirror rides the same flow, and D3 finding 3.4 - (`instruction-placement`, `overengineering`) establishes the repo's pattern that anything which - must survive a reclaimed container becomes a tracked file. -- Gate profile accepted for the mirror: always-on markdownlint, typos, editorconfig and gitleaks - still run on `docs/topics/**` (D3 5.1). `stumbles.md` must be ATX-headed, dash-bulleted, final- - newline; the mirror JSON 2-space indented with a final newline. The em-dash gate is an allowlist - and `docs/topics/` is not on it (D3 5.3), so free prose there is unenforced. - -WHAT WOULD CHANGE THIS VERDICT - -1. An empirical sentinel check (D3 U1: write a file under `~/.claude/plugins/data/`, look for - it next session) showing the plugin data dir **does** survive reclaim for this environment — - then option (i) alone suffices and the mirror becomes optional. -2. The operator electing to run observe from a durable local checkout, which moves the durability - problem out of scope (and pins later phases to that machine's path, per F1). - -PHASE 2 MECHANICAL STEPS (D3) - -1. `mkdir -p docs/topics/unhobble-fable-5-1/`. -2. Write `docs/topics/unhobble-fable-5-1/manifest-mirror.json`: a copy of `manifest.json` with - `checkout.worktree_path` set to the literal string `${CLAUDE_PROJECT_DIR}` and a new - `"mirror_note"` stating that the canonical manifest with the resolved absolute path lives at - `${CLAUDE_PLUGIN_DATA}/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/manifest.json` - and that this mirror exists because a cloud clone cannot see that directory. -3. Write `docs/topics/unhobble-fable-5-1/stumbles.md` with the Phase 3 header row - (`| Date | Task | What happened | Expected | Suspected missing instruction | Severity |`) and a - one-paragraph preamble naming the experiment id and branch. This tracked file is the **primary** - ledger; the plugin-data copy is derived. -4. Add to the canonical `manifest.json` a `"state_location_deviation"` object recording: the mirror - path, the reason (cloud ephemerality; `topic-docs` visibility matrix), that `backups/` is - excluded from the mirror, and that the mirror is contract-tier and pruned before merge. -5. Include both files in the Phase 2 commit (`experiment: strip instruction surfaces for unhobble - baseline`) and push to `claude/unhobble-config-oe7yfx`. -6. At the start of every later session: `git pull`, then rehydrate - `${CLAUDE_PLUGIN_DATA}/unhobble//` from the mirror (restoring the absolute - `checkout.worktree_path` for the current checkout) before running any phase command. -7. At Phase 4 close: fold the ledger's conclusions into the restoring commits, then prune - `docs/topics/unhobble-fable-5-1/` per the contract tier's prune-with-pointer lifecycle before - merge. - ---- - -## D4. Sanity check of the default strip plan - -**VERDICT: Three class/flag changes — `rule-ruff-pin` becomes a ground-truth-oracle keep (drop -`operator_optional_strip`), `agents-draft-pr` gains a contested `external-publication` register- -class candidate and drops `operator_optional_strip` for this run, and `rule-pr-body-contract` is -stripped under a **named register hold** rather than as a plain convention strip. No other unit's -class changes.** - -CONFIDENCE: **medium-high**. - -EVIDENCE - -- `rule-ruff-pin` — manifest has `class: convention, action: keep, operator_optional_strip: true`, - reason "Repo tooling convention". The memo shows it is stronger than that: D2 §3.3's ground- - truth-oracle carve-out (`docs/plugin-philosophy.md` "Classifying a hook") plus D2 §4(d)'s - fail-open wrapper, single-subtree CI lane, and non-derivable pin fact make it a **keep on the - rule, not a keep the operator may casually override**. Change: keep `class: convention`, set - `"operator_optional_strip": false`, and add `"keep_basis": "ground-truth-oracle carve-out - (PLUGIN-PHILOSOPHY 'Classifying a hook'); run-ruff.sh is fail-open outside one subtree"`. -- `agents-draft-pr` — manifest has `class: convention, operator_optional_strip: true`. D2 §3 Gate 0 - determination records it as **contested `external-publication`** (the rubric's literal example is - "opening PRs"; `plugins/instruction-placement/context/routing-rubric.md` line ~38). Under - contract Phase 4 step 4 that class is restored regardless of the ledger. Change: add - `"register_class_candidate": "external-publication (contested; rubric example vs reversible - consequence)"` and set `"operator_optional_strip": false` for this run, with the reason recorded - as "no oracle exists (D2 §4(b)) and a Gate 0 candidate returns on a register hold, so the strip - buys cost without evidence." -- `rule-pr-body-contract` — manifest has `class: convention, action: keep, - operator_optional_strip: true`. D2 §3 records it as **contested-weak `external-publication`**. - Per D1/D2 the strip is elected (D2 above), so the manifest must carry the hold rather than - strip it silently as a convention. Change: `action: strip`, `mechanism: git rm`, `restore: git`, - plus `"register_class_candidate": "external-publication (contested, weak)"` and - `"phase4_precommit": "register hold — restored at Phase 4 regardless of ledger rows if the - contested reading is adopted; not counted in the ledger's defence tally"`. -- `agents-validate` — **no class change.** It stays `convention`; only `action` flips from `keep` - to `strip` via the existing `operator_optional_strip` flag (D2 above). Gate 0: no class. -- Units the memos do not touch (`rule-vendor-docs`, `rule-catalog-taxonomy`, `rule-hook-budget`, - `rule-worktree-base-ref`, `rule-skill-bodies`, the four `nested-*`, every `settings-*`, - `plugin-fleet`, `plugin-playgrounds`, `plugin-config-files`) are **not re-litigated** and stand - as the manifest classifies them. `nested-model-adaptation` (policy / `legal-compliance`) and - `settings-permissions-deny` (policy / `secret-handling`) already carry their register classes - correctly. - -WHAT WOULD CHANGE THIS VERDICT - -1. An adjudication (an added register entry, or a rubric amendment) placing a reversible - publication-cost rule definitively inside or outside `external-publication` — that settles - `agents-draft-pr` and `rule-pr-body-contract` in one move and removes both hedges. -2. Reading the `ci-workflows` `pr-contract` composite at SHA `5776760…` and finding the - body/linkage half gating — `rule-pr-body-contract` then reverts to `action: keep`. diff --git a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D1-user-scope.md b/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D1-user-scope.md deleted file mode 100644 index bacd0bb39a..0000000000 --- a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D1-user-scope.md +++ /dev/null @@ -1,285 +0,0 @@ -# Research memo D1: should user scope be opted in for the unhobble run? - -All web sources fetched live on **2026-09-11** (WebFetch, plus `curl` of the Mintlify raw -`.md` for pages whose HTML render exceeded the fetch budget). No claim below rests on training -memory. Repo evidence is second-tier and labeled as such. - -## Question - -In a Claude Code on the web (cloud) session on ``, the -user-global `~/.claude/settings.json` enables all 73 plugins of the repo's own marketplace (20 -wiring hooks), and that file is synthesized by the cloud environment snapshot plus -`.claude/cloud-bootstrap.sh` rather than by a human editing personal settings. The unhobble -contract strips PROJECT-scope surfaces by default and treats user-global surfaces as opt-in. -Gather the evidence needed to decide whether to opt user scope in for this run. **No decision is -made here.** - -## Findings - -### 1. Settings scopes and precedence; where `enabledPlugins` lives - -Four settings files plus managed sources; precedence is managed > command line > project local > -shared project > user. - -> Claude Code reads settings from four files, and an organization can also deliver managed -> settings from the claude.ai console. -> — code.claude.com/docs/en/settings, "Settings files and who they affect", fetched 2026-09-11 - -> When the same key appears in more than one place, Claude Code uses the value from the highest -> level that sets it. -> — same page, "Settings precedence" (ordered list: 1 Managed, ... 5 User settings -> `~/.claude/settings.json`), fetched 2026-09-11 - -`enabledPlugins` is settable in **any** of the four files: - -> Turn individual plugins on or off, keyed by `plugin-name@marketplace-name`. A plugin with no -> entry at any scope falls back to its `defaultEnabled` value. -> — code.claude.com/docs/en/settings-reference, "`enabledPlugins`" (Scope: `Any file`; Type: -> object mapping to Boolean), fetched 2026-09-11 - -**Which scope wins.** The docs state the project-over-user direction explicitly, and only that -direction: - -> Project settings take precedence over user settings, so setting a plugin to `false` in -> `~/.claude/settings.json` doesn't disable a plugin that the project's `.claude/settings.json` -> enables. -> — same entry, fetched 2026-09-11 - -For the case asked about (true at user, **false** at project), the general precedence stack -implies project wins, but no sentence states it for `enabledPlugins` specifically. For -**absent** at project, the user value stands: `defaultEnabled` applies only when there is no -entry "at any scope", and an entry at any scope takes precedence over it: - -> `defaultEnabled` is the fallback when nothing else has decided the plugin's state. Two things -> take precedence over it: **The user's setting**: an entry for the plugin in `enabledPlugins` -> at any settings scope. -> — code.claude.com/docs/en/plugins-reference, "Default enablement", fetched 2026-09-11 - -### 2. How an enabled plugin's hooks and skills load; per-project disable - -Hooks: a plugin's `hooks/hooks.json` is a first-class hook source, active on enablement. - -> | Plugin `hooks/hooks.json` | When plugin is enabled | Yes, bundled with the plugin | -> — code.claude.com/docs/en/hooks, hook-source table, fetched 2026-09-11 - -> When a plugin is enabled, its hooks merge with your user and project hooks. -> — same page, "Plugin scripts" tab, fetched 2026-09-11 - -Skills: plugin skills load wherever the plugin is enabled, namespaced, and their **descriptions** -occupy context every turn while bodies load on invocation. - -> | Plugin | `/skills//SKILL.md` | Wherever the plugin is enabled, as -> `/plugin-name:skill-name` | -> — code.claude.com/docs/en/skills, "Where skills live", fetched 2026-09-11 - -> In a regular session, skill descriptions are loaded into context so Claude knows what's -> available, but full skill content only loads when invoked. -> — same page, frontmatter/invocation section, fetched 2026-09-11 - -**Per-project disable of a user-enabled plugin is possible** — `false` at project scope is a -documented, supported spelling: - -> To keep a plugin out of one project's synced sessions in every environment, set -> `"@synced": false` under `enabledPlugins` in that project's committed -> `.claude/settings.json`. -> — code.claude.com/docs/en/plugins-reference, "Synced plugins", fetched 2026-09-11 - -> The restriction is specific to `pluginConfigs`: `enabledPlugins` still honors project and local -> settings. -> — same page, `pluginConfigs` section, fetched 2026-09-11 - -The repo's own bootstrap already relies on this overlay direction: a repo `false` opts out of a -fleet entry (see finding 4). - -### 3. Cloud provisioning; is `~/.claude` durable per-user state? - -Fresh VM per session, repo cloned in: - -> In Anthropic-hosted environments, each session gets a fresh virtual machine (VM) running -> Ubuntu 24.04 on x86_64 ... with your repository cloned and common toolchains pre-installed. -> — code.claude.com/docs/en/cloud-environments, "Cloud environments", fetched 2026-09-11 - -> Cloud sessions start from a fresh clone of your repository. Anything you commit to the repo is -> available. Anything you've installed or configured only on your own machine isn't available in -> the session. -> — same page, "What carries over from your setup", fetched 2026-09-11 - -The **user's own machine** `~/.claude` never reaches the cloud — and the docs name -`enabledPlugins` at user scope by name as a thing that does **not** carry over: - -> | Plugins enabled only in your user settings | No | User-scoped `enabledPlugins` lives in -> `~/.claude/settings.json`. Declare them in the repo's `.claude/settings.json` instead ... | -> — same table, fetched 2026-09-11 - -> **User and project local settings** (`~/.claude/settings.json` and -> `.claude/settings.local.json`): not read. Both stay on your machine, and the local file isn't -> in the clone. -> — code.claude.com/docs/en/settings, "Settings in cloud sessions", fetched 2026-09-11 - -> Cloud sessions on Claude Code on the web don't read your local `~/.claude/settings.json`; -> hooks there come from the repo and from your organization's server-managed settings. -> — code.claude.com/docs/en/hooks, hook-sources note, fetched 2026-09-11 - -What *does* persist across sessions is the **environment cache**, a filesystem snapshot built by -the setup script: - -> The cache is a filesystem snapshot, so it keeps what the setup script writes to disk and loses -> anything that was only running. Packages you install, Docker images you pull, and files you -> write all carry over. -> — code.claude.com/docs/en/cloud-environments, "Environment caching", fetched 2026-09-11 - -> The setup script runs again to rebuild the cache when you change the environment's setup script -> or allowed network hosts, and when the cache reaches its expiry after roughly seven days. -> Resuming an existing session never re-runs the setup script. -> — same section, fetched 2026-09-11 - -> You can also ask Claude to install packages mid-session, but those installs don't carry over to -> other sessions. -> — same page, "Run tests, start services, and add packages", fetched 2026-09-11 - -**Net:** the in-container `~/.claude/settings.json` is not the operator's personal file and is not -per-user durable state in the sense the docs use "user settings". It is per-environment -synthesized state, rebuilt by the setup script into the environment cache (≈7-day expiry) and -re-repaired by the repo's `SessionStart` hook. Docs are **silent** on whether the cache snapshot -covers `$HOME` specifically; that specific durability claim is unverified (see Unverified). - -### 4. Repo's own account (second-tier): why the whole catalog is enabled at user scope - -> The whole catalog is installed here, so this repo dogfoods everything it publishes and a -> regression in any plugin surfaces here first — bar what a repo delta opts out of. -> — `/docs/cloud-sessions.md` (~line 372), read 2026-09-11 - -> ... which the shared environment fetches at cache build, writes into the snapshot at -> `/opt/melodic-fleet-plugins.json`, and installs at user scope; `cloud-bootstrap.sh` reads that -> snapshot copy overlaid with `.claude/settings.json`, so the committed `enabledPlugins` block -> carries only this repo's deltas (an explicit `false` opts out of a fleet entry ...). -> — same file (~line 379), read 2026-09-11 - -> The trade is context: every enabled plugin adds per-turn cost, so a *consumer* repo should opt -> out of what it does not need rather than copying anything wholesale. -> — same file (~line 388), read 2026-09-11 - -Bootstrap comment confirming user scope is the mechanism, not a preference: - -> Enabled set: the fleet list the shared environment baked into the snapshot ... overlaid with -> the tracked settings file, so a repo entry set to false opts out of a fleet entry ... so a -> settings block reduced to deltas still dogfoods the whole catalog from the current branch. -> — `/.claude/cloud-bootstrap.sh` (~lines 205-210), read 2026-09-11 - -CI holds the dogfooding claim to the files: - -> docs/cloud-sessions.md promises `enabledPlugins` "turns on the whole catalog, so this repo -> dogfoods everything it publishes"; nothing enforced it ... The failure is silent by -> construction: .claude/cloud-bootstrap.sh computes its install set from that same -> enabledPlugins map. -> — `/.github/workflows/ci.yml` (~line 761), read 2026-09-11 -> (gate: `scripts/check-plugin-catalog-enablement.sh`) - -Also relevant to how a strip would behave in-session: the repo records that a `SessionStart` -install is never visible to the session that ran it ("The command/skill registry is built when the -Claude Code process starts ... and is not re-read afterwards", CLOUD-SESSIONS.md ~line 286), -verified by that doc's own 2026-08-15 observation, not by an Anthropic page. - -### 5. Official guidance on instruction hygiene; hooks that enforce vs. correct - -Trimming guidance exists and is scope-agnostic — it names CLAUDE.md size and pruning without ever -distinguishing user-scope from project-scope instructions: - -> Keep it concise. For each line, ask: *"Would removing this cause Claude to make mistakes?"* If -> not, cut it. Bloated CLAUDE.md files cause Claude to ignore your actual instructions! -> — code.claude.com/docs/en/best-practices, "Write an effective CLAUDE.md", fetched 2026-09-11 - -> **Fix**: Ruthlessly prune. If Claude already does something correctly without the instruction, -> delete it or convert it to a hook. -> — same page, "The over-specified CLAUDE.md", fetched 2026-09-11 - -> **Size**: target under 200 lines per CLAUDE.md file. Longer files consume more context and -> reduce adherence. -> — code.claude.com/docs/en/memory, "How CLAUDE.md affects behavior", fetched 2026-09-11 - -The memory page lists user scope (`~/.claude/CLAUDE.md`) and project scope in one load-order -table but attaches no different ablation or trimming advice to either. - -Hooks: instructions are advisory context; hooks are the deterministic layer, and the docs -separate "enforce a gate" from "nudge behavior": - -> Both are loaded at the start of every conversation. Claude treats them as context, not enforced -> configuration. To block an action regardless of what Claude decides, use a PreToolUse hook -> instead. -> — code.claude.com/docs/en/memory, "Two memory systems", fetched 2026-09-11 - -> Unlike CLAUDE.md instructions which are advisory, hooks are deterministic and guarantee the -> action happens. -> — code.claude.com/docs/en/best-practices, "Set up hooks", fetched 2026-09-11 - -> Because the `if` filter is best-effort, use the permission system rather than a hook to enforce -> a hard allow or deny. -> — code.claude.com/docs/en/hooks, matcher/`if` section, fetched 2026-09-11 - -> If your hook is meant to enforce a policy, use `exit 2`. -> — same page, exit-code note, fetched 2026-09-11 - -This matches, but does not originate, the unhobble contract's own carve-out: "Hooks that enforce -policy (secrets gates, PR-body contracts, permission guards) are classified `policy` at snapshot -time and are NOT stripped by default" -(`plugins/claude-config/skills/unhobble/SKILL.md`, "Scope and safety rails", read 2026-09-11). - -## Consensus table - -| Claim | Sources agreeing | Sources disagreeing / silent | Confidence | -|---|---|---|---| -| Four scopes: managed > command line > project local > shared project > user | settings (precedence list + graphic) | none | High | -| `enabledPlugins` may be set in any of the four files | settings-reference (`Scope: Any file`); plugins-reference | none | High | -| A project `false` can disable a plugin enabled elsewhere (per-project opt-out is supported) | plugins-reference (synced-plugin opt-out); repo bootstrap overlay | settings-reference states only the reverse (project `true` beats user `false`) for `enabledPlugins` | Medium-High | -| Plugin absent at project + `true` at user ⇒ enabled | plugins-reference "Default enablement" (an entry at any scope decides) | no page states the user-only case in those words | Medium-High | -| Enabling a plugin activates its hooks and loads its skills (descriptions each turn) | hooks (source table + "merge with your user and project hooks"); skills ("wherever the plugin is enabled"; descriptions in context) | none | High | -| A real user's `~/.claude/settings.json` never reaches a cloud session | cloud-environments "What carries over"; settings "Settings in cloud sessions"; hooks note | none | High | -| Each cloud session gets a fresh VM; only the environment cache (setup-script filesystem snapshot, ≈7-day expiry) persists between sessions | cloud-environments (fresh VM, caching); claude-code-on-the-web (expiry reclaims the VM) | no page says whether the snapshot covers `$HOME` | High for the mechanism, Low for `$HOME` | -| The repo's user-scope enablement is environment/bootstrap-synthesized dogfooding, not a human preference | CLOUD-SESSIONS.md; cloud-bootstrap.sh; ci.yml gate comment | no official Anthropic source describes this pattern | High (repo-internal), second-tier | -| Official trimming guidance does not distinguish user-scope from project-scope instructions | best-practices; memory | both are silent on the distinction rather than contradicting it | High (as a silence claim) | -| Hooks are the deterministic/enforcement layer; instructions are advisory | best-practices; memory; hooks | none | High | - -## Unverified claims - -1. **`enabledPlugins` `true` at user + `false` at project.** No page states the outcome for this - exact pair. Inference from the general precedence stack (project over user) plus the - synced-plugin opt-out sentence points to project winning; treat as inference, not doc. -2. **Whether the environment cache snapshot includes `$HOME/.claude`.** Docs say the cache "keeps - what the setup script writes to disk" but never scope it to a directory. The repo's bootstrap - behaves as if it does; unverified against Anthropic docs. -3. **In-session effect of disabling plugins mid-cloud-session.** Docs describe `/reload-plugins` - and note `/plugin` is unavailable in cloud sessions; the repo records that the registry is - built at process start and not re-read. No official page states what happens when - `enabledPlugins` is edited on disk mid-cloud-session. Unverified. -4. **Anthropic engineering blog posts on Claude Code best practices** were not fetched; only - code.claude.com docs were. Any claim attributed to anthropic.com/engineering would be - unverified here. -5. **Per-turn context cost of 73 enabled plugins.** Only qualitative sources exist (skills page: - descriptions load every session; discover-plugins: a per-plugin "Context cost" estimate in - `/plugin`). No measured number was collected. - -## Options the evidence supports - -Listed, not ranked; no recommendation. - -- **A. Leave user scope out (contract default).** Rests on: the in-container user file is - environment-synthesized, so ablating it measures the environment's provisioning rather than - standing human instructions; and the unhobble contract already reserves user-global surfaces - for explicit opt-in. -- **B. Opt user scope in wholesale.** Rests on: 73 plugins' skill descriptions load into every - turn and 20 plugins' hooks merge into every session (hooks + skills docs), so a project-only - strip leaves the dominant instruction surface in place and the "bare baseline" is not bare. -- **C. Ablate at project scope instead of touching user scope** — add `"@": - false` entries to the repo's committed `.claude/settings.json`. Supported by the documented - per-project opt-out and by the bootstrap's existing overlay semantics (repo `false` opts out of - a fleet entry). Stays inside the contract's PROJECT-scope default and is reversible by git. -- **D. Partial opt-in: hooks only.** Strip the 20 hook-wiring plugins, keep the rest enabled, - since hooks are the enforcement layer while skill descriptions are advisory context. Note the - contract's `policy` carve-out (secrets gates, PR-body contracts, permission guards) would still - exclude some of them. -- **E. Defer to the environment layer.** Change the fleet list / setup script rather than the - session, accepting the ≈7-day cache rebuild latency and the fact that the change would affect - every repo served by that environment, not just this one. -- **F. Run the experiment, then verify what actually loaded** with `/context` and - `claude plugin list --json` before treating any arm as "stripped", given the unverified - mid-session reload behavior (Unverified #3). diff --git a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D2-conventions.md b/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D2-conventions.md deleted file mode 100644 index 00551cb20d..0000000000 --- a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D2-conventions.md +++ /dev/null @@ -1,273 +0,0 @@ -# Research memo D2 — should the four "convention" units be stripped for this unhobble run? - -Research only. No recommendation. All web sources fetched live 2026-09-11 via WebFetch. - -## Question - -For each of four units classified `convention` by `/claude-config:unhobble` Phase 1 and kept by -default under the official carve-out, gather the evidence an operator needs to decide whether to -opt each one into the strip: - -- **(a)** `AGENTS.md` § "Validate a change" -- **(b)** `AGENTS.md` § "Open a pull request as a draft" -- **(c)** `.claude/rules/pr-body-contract.md` -- **(d)** `.claude/rules/ruff-pin.md` - -All four are always-loaded or near it: `CLAUDE.md` is `@AGENTS.md`; `pr-body-contract.md` carries no -`paths:` frontmatter (rules without `paths` "are loaded unconditionally"); `ruff-pin.md` is -path-scoped to `**/*.py`, so it loads only on reading a Python file, and not inside subagents. - -## Findings - -### 1. Official Claude Code docs — what belongs in CLAUDE.md, the cut test, the carve-outs - -**1.1** `https://code.claude.com/docs/en/best-practices`, § "Write an effective CLAUDE.md" -(fetched 2026-09-11) — the cut test, verbatim: - -> "Keep it concise. For each line, ask: *'Would removing this cause Claude to make mistakes?'* If -> not, cut it. Bloated CLAUDE.md files cause Claude to ignore your actual instructions!" - -**1.2** Same page, same section — the include/exclude table. Its Include column contains -**"Bash commands Claude can't guess"**, **"Testing instructions and preferred test runners"**, -**"Repository etiquette (branch naming, PR conventions)"**, and **"Developer environment quirks -(required env vars)"**. Its Exclude column contains **"Anything Claude can figure out by reading -code"** and **"Standard language conventions Claude already knows"**. All four units under review -sit inside named Include rows: (a) is a test runner instruction, (b) and (c) are PR conventions / -repository etiquette, (d) is a bash command plus an environment quirk. - -**1.3** Same page, § "Avoid common failure patterns": - -> "**The over-specified CLAUDE.md.** If your CLAUDE.md is too long, Claude ignores half of it -> because important rules get lost in the noise. **Fix**: Ruthlessly prune. If Claude already -> does something correctly without the instruction, delete it or convert it to a hook." - -**1.4** Same page, same section — the team-convention posture: "Check CLAUDE.md into git so your -team can contribute. The file compounds in value over time." - -**1.5** `https://code.claude.com/docs/en/memory`, § "When to add to CLAUDE.md" (fetched -2026-09-11): - -> "Keep it to facts Claude should hold in every session: build commands, conventions, project -> layout, 'always do X' rules." - -and the trigger list: "Claude makes the same mistake a second time"; "A new teammate would need the -same context to be productive". - -**1.6** `memory`, § "My CLAUDE.md is too large" — the only official statement of what an automated -trim cuts and what it keeps: - -> "it cuts content Claude can derive from the codebase, such as directory layouts, dependency -> lists, and architecture overviews, and keeps pitfalls, rationale, and conventions that differ -> from tool defaults." - -This is the closest official wording to a conventions carve-out. Note its qualifier: conventions -**that differ from tool defaults**. All four units differ from a tool default — (a) run a selector -not the whole corpus; (b) draft not ready; (c) a body shape `gh` does not impose; (d) a wrapper not -bare `ruff`. - -**1.7** `memory`, § "CLAUDE.md vs auto memory": "Use for: Coding standards, workflows, project -architecture". And the enforcement caveat, twice: "Claude treats them as context, not enforced -configuration. To block an action regardless of what Claude decides, use a PreToolUse hook". - -**1.8** No official page states a carve-out for "team conventions" **as a class exempt from the cut -test**. The cut test in 1.1 is stated without exception; the conventions language in 1.2 / 1.5 / 1.6 -is an inclusion list, not an exemption from "would removing this cause mistakes". - -### 2. Official Anthropic prompting guidance — over-constraining and judgment - -**2.1** `https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices` -(fetched 2026-09-11; `docs.claude.com` 302-redirects here): - -> "**Prefer general instructions over prescriptive steps.** A prompt like 'think thoroughly' often -> produces better reasoning than a hand-written step-by-step plan. Claude's reasoning frequently -> exceeds what a human would prescribe." - -**2.2** `https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/prompting-claude-opus-5`, -§ "Task scope and over-verification" (fetched 2026-09-11) — the removal directive is scoped to -*verification scaffolding*, not to conventions: - -> "If your prompt contains explicit verification instructions ... remove them: instructions like -> these cause over-verification on Claude Opus 5 ... The same applies to legacy harness scaffolding -> that adds separate verification steps." - -Same page, same section, the counter-direction: "For narrow tasks, constrain scope explicitly." -And § "Capability improvements": "If your review prompt says 'only report high-severity issues' ... -the model may follow that instruction literally" — current models follow standing text **literally**, -which cuts both ways for a convention line. - -**2.3** The "highly important areas" carve-out. Verbatim, from -`https://claude.com/blog/the-new-rules-of-context-engineering-for-claude-5-generation-models` -(fetched 2026-09-11), under its **"Skills"** heading: - -> "Avoid making them overconstrained, except in highly important areas." - -Two load-bearing qualifications, both verified: (i) it is stated about **skills**, not about -CLAUDE.md or rules files; (ii) "highly important areas" is **never defined on the page**. The repo -reaches the same conclusion independently — `plugins/claude-config/skills/audit-instructions/reference/criteria.md` -line 1887: "**Source:** none — the 'except in highly important areas' carve-out appears on no -official page". Same blog, on team conventions: "It's best when skills encode particular opinions, -knowledge, or best practices that are particular to you, your team, or product." - -**2.4** Same blog, precedent for aggressive subtraction, as cited by this repo's own philosophy -doc: Anthropic removed "over 80%" of Claude Code's system prompt for the Opus 5 / Fable 5 -generation with no measurable loss on coding evals (`docs/plugin-philosophy.md` § "Instruction -economy", which cites the blog, verified there 2026-08-08). - -### 3. Repo doctrine (second tier) - -**3.1** `docs/plugin-philosophy.md` § "Instruction economy" — the **durable-tier exemption**, which -is where the unhobble `convention` class comes from: - -> "**The durable tier is exempt.** Deterministic policy hooks (gates that enforce team or safety -> policy regardless of model capability) and team conventions checked into git are the officially -> carved-out durable instruction tiers." - -Note the claim "officially carved-out" is the repo's reading of 2.3; per 2.3(i)/(ii) and the repo's -own criteria.md line 1887, the upstream wording is about skills and leaves the term undefined. - -**3.2** Same section — **evidence-gated additions** (governs re-add, symmetrically informs strip): - -> "A new standing instruction requires observed, repeated stumble evidence against the current -> model — the same failure seen more than once — never anticipation of a failure a past model had." - -**3.3** Same doc, § "Classifying a hook" — the **ground-truth-oracle carve-out**, the rubric's -sharpest instrument and directly applicable to (d): - -> "a hook with a *behavioral purpose* but a *non-derivable ground-truth oracle* ... is a keep, not -> an ablation candidate. It corrects hallucination with machine ground truth no model can know -> unaided, so 'it corrects the model' alone is never the delete criterion; 'the model could derive -> this itself' is." - -**3.4** `docs/conventions/instruction-exception-register/README.md` — the register adopts Gate 0's -six classes by reference and applies them to **deletion**: - -> "A candidate matching any Gate 0 class is **not deletable** by an instruction-audit trim." - -On unhobble specifically, the register's consumer table: the experiment "may strip a protected rule -during the run, since the strip is reversible and branch-local, but Phase 4 restores it regardless -of whether the ledger logged a stumble against it". Also: "**Omission from this register is not -licence to delete.**" - -**3.5** `plugins/instruction-placement/context/routing-rubric.md` § "Gate 0 — the hard-deny -classes". Relevant row verbatim: - -> | `external-publication` | Governs anything that leaves the machine | opening PRs, deploying, -> posting, sending mail, publishing packages | - -and the recognition rule: - -> "**Recognition is by consequence, not by phrasing.** ... Ask what breaks when the instruction is -> absent at the moment it was needed, not how the sentence is worded." - -Also rung 1 of the ladder: "**Does removing it cause a mistake?** No → **delete**." - -**Gate 0 determination, by consequence:** - -- **(a)** No class. A skipped or over-broad local test run costs wall clock; CI decides. -- **(b)** *Contested.* The literal `external-publication` example is "opening PRs", and the rule - governs the act of opening one. By consequence the miss is bounded and reversible: a PR opened - ready can be returned to draft (`gh pr ready --undo`); the cost is spent CI minutes and one - AI-review pass, not an unrecoverable state. Evidence supports either reading; both recorded. -- **(c)** *Contested, weaker than (b).* It governs PR body text — content that leaves the machine — - but the enforcing composite is explicitly advisory (§4), so the failure mode is a bot comment and - a label, both removable. -- **(d)** No class. Tooling-version fidelity; no irreversible, secret, data, legal, or authority - consequence. - -### 4. Remaining deterministic oracles - -**(a) `scripts/affected-tests.sh`.** CI runs the same selector: `.github/workflows/ci.yml` -line ~1460, job `test-linux`, step "Run plugin contract tests": -`scripts/affected-tests.sh --run --shard "$LEG/$LEGS" --base "origin/$BASE_REF"`. `test-linux` is in -`ci-status.needs`, and `ci-status` is the single required check — so **gating**. The script's own -header states the contract the AGENTS.md line summarizes: "FAIL LOUD, NOT OPEN. A changed file that -maps to NO suite is an ERROR"; and "--run IS A LINUX GATE". `README.md` § "Validate a change" owns -the full contract. Caveat from `ci.yml` line 133: `run_tests` carries -`github.event.pull_request.draft != true`, so the gate does not fire while the PR is a draft. - -**(b) draft PRs.** **No oracle at all.** No CI job, no local hook (`.claude/hooks/` holds only -`cloud-bootstrap-plugins.test.sh` and `hook-telemetry-sink.sh`), and no default in the -`source-control:pull-request` skill (`reference/create.md` mentions `draft` only as an optional REST -field). The `draft` flag only *gates other lanes*: `ci.yml` lines 133-137 disable the test lanes on -a draft; `claude-review.yml` line 46 and `claude-security-review.yml` lines 59/103 skip on -`draft == true`. The convention is a cost-control etiquette rule with no detector. - -**(c) PR body contract.** `ci.yml` job `ci-status`, step "Check the pull-request contract", the -composite `melodic-software/ci-workflows/.github/actions/pr-contract@5776760…` (v0.22.2). Per the -rule file itself and `.claude/source-control.md`: the linkage/section half is **advisory** — "a body -that misses a closing keyword or a required section gets a warning, an upserted comment and the -`needs-issue-linkage` label, and `ci-status` still passes on that account". The gating part of the -same composite is the Conventional-Commits title check and the `do-not-merge` label. A second -surviving derivation path: `.claude/source-control.md` (not one of the four units, and not stripped -with them) carries `pr_body_required_sections` = Summary / Fix / Verification / Related. - -**(d) ruff pin.** `scripts/run-ruff.sh` resolves the pin from `.github/requirements-ci.txt` -(`ruff==0.16.5`) and **exits 2** when a PATH ruff drifts and `uvx` is absent, printing "ruff on PATH -is ${got}, but CI pins ruff==${pin}". Rationale is owned by `docs/ci-runner-routing.md` § "Local / -workstation ruff": "Do **not** trust a bare `ruff` on `PATH` for verification in this repository." -CI reaches ruff only through `plugins/source-control/skills/babysit-prs/scripts/engine.test.sh` -(lines 42-53), which lints `. tests` **in that one subtree** and **SKIPs on exit 127** ("pinned ruff -not available … lint pass omitted"). So the oracle is gating only for that subtree, fail-open when -the pin is unavailable, and there is **no repo-wide ruff lane** in `ci.yml` (the only other `ruff` -hit is `ruff.toml` in the `changes` path filter). Nothing detects a *local* bare-`ruff` run. - -## Per-unit evidence table - -| Unit | Gate 0 class match? | Remaining oracle | Gating or advisory | What the model could derive without the line | -|---|---|---|---|---| -| (a) AGENTS.md "Validate a change" | No | `scripts/affected-tests.sh` run by `ci.yml` job `test-linux`; script header + `README.md` § "Validate a change" | **Gating** (via `ci-status.needs`), but suppressed while the PR is a draft | The script exists and self-documents (usage block, `--explain`, `--run` Linux-gate note); README owns the contract. Derivation requires the model to look — nothing prompts it, and a bare model may run the whole corpus, one suite, or none. Selector existence is not inferable from source layout alone. | -| (b) AGENTS.md "Open a pull request as a draft" | **Contested**: `external-publication` names "opening PRs" literally; by consequence the miss is bounded and reversible (`gh pr ready --undo`) | **None** | n/a — no detector | Inferable only by reading `ci.yml` lines 133-137 plus both `claude-*-review.yml` draft guards and reasoning backwards about cost. No repo file states the rule outside AGENTS.md. Low derivability, zero feedback: a violation is silent and self-inflicted spend. | -| (c) `.claude/rules/pr-body-contract.md` | **Contested (weak)**: governs PR body content leaving the machine; failure is a comment + `needs-issue-linkage` label | `pr-contract` composite step inside `ci-status`; plus `.claude/source-control.md` `pr_body_required_sections` (survives this strip) | **Advisory** for body/linkage (the rule says so); the composite's title and `do-not-merge` checks are gating | Section list is recoverable from `.claude/source-control.md` and from any recent merged PR body; the composite is named in `ci.yml`. The advisory comment is a real feedback loop — a violation announces itself on the PR within one CI run, so the bare model gets a correction signal the other three units lack. | -| (d) `.claude/rules/ruff-pin.md` | No | `scripts/run-ruff.sh` (exit 2 on drift, 127 when unavailable); `engine.test.sh` lints one subtree in CI; `docs/ci-runner-routing.md` owns rationale | Wrapper exit code is deterministic but **only when invoked**; the CI lint is gating for `plugins/source-control/skills/babysit-prs/**` only and **fail-open** (SKIP) otherwise | The wrapper exists at `scripts/run-ruff.sh` with a full rationale header, and `ruff.toml` sits at the root. But the pin's existence and the "bare ruff disagrees in both directions" fact are **non-derivable ground truth** (a release moving 18 E/F rules into defaults) — the exact shape §3.3 calls a keep. A bare model typing `ruff check` gets a clean-looking wrong answer with no error. | - -## Consensus table - -| Claim | Official CC docs | Anthropic prompting guidance | Repo doctrine | Verdict | -|---|---|---|---|---| -| Cut any line whose removal would not cause mistakes | Yes, verbatim (1.1) | Consistent (2.1, 2.2) | Adopted verbatim (3.1 quotes it) | **Consensus** | -| Team conventions / PR etiquette / test-runner instructions belong in CLAUDE.md | Yes, explicit Include rows (1.2, 1.5) | Analogous for skills (2.3) | Yes (3.1) | **Consensus** | -| "Team conventions in git" are a class *exempt from the cut test* | **Not stated anywhere** (1.8); the trim "keeps … conventions that differ from tool defaults" (1.6) is the nearest wording | Blog carve-out is about **skills** and is undefined (2.3) | Repo asserts an "officially carved-out durable tier" (3.1) | **Disagreement**: repo doctrine is stronger than any source supports; the repo's own criteria.md line 1887 concedes the carve-out "appears on no official page" | -| Deletion floor is the six Gate 0 consequence classes | Not addressed | Not addressed | Owned by register + rubric (3.4, 3.5) | **Repo-only**, uncontradicted upstream | -| A rule with a non-derivable machine oracle is a keep | Not addressed | Not addressed | Explicit (3.3) | **Repo-only**, uncontradicted | -| Instructions are disposable per model generation; ablate at release | Implied ("prune regularly") | Strongly (2.2, 2.4) | Explicit (3.1 "Generation-triggered ablation") | **Consensus** | - -## Unverified claims - -- **The Gate 0 class of (b) and (c).** No source adjudicates whether a reversible publication-cost - rule is inside `external-publication`. The rubric's own instruction ("by consequence, not by - phrasing") points away from inclusion; its example list points toward it. Unresolved by evidence. -- **The `ci-workflows` `pr-contract` composite's internals** were not fetched (it lives in - `melodic-software/ci-workflows` at SHA `5776760…`). The advisory-vs-gating split is taken from - `.claude/rules/pr-body-contract.md` and `.claude/source-control.md`, both of which state it; - neither was checked against the composite at that SHA. -- **`docs.claude.com` Claude-4/5 prompting pages**: `claude-5-best-practices` returned **404**; the - Claude-4 URL redirects to the consolidated `claude-prompting-best-practices` page, which is what - was read. No separate "Claude 4.x best practices" wording was verified. -- **Whether any of the four has prior stumble evidence.** No ledger, issue, or PR search was run for - observed violations of these four lines; 3.2's symmetric question ("what evidence earned this - line?") is unanswered for all four. -- Whether `test-windows` is intentionally absent from `ci-status.needs` (observed, not investigated). - -## Options the evidence supports - -1. **Keep all four** (unhobble default). Grounds: 1.2's Include rows name all four categories; - 3.1's durable-tier exemption; the strip would measure nothing new for (c), whose violation is - already self-announcing. -2. **Strip all four.** Grounds: 1.1's cut test is stated without exception; 1.8 and 2.3 show the - conventions carve-out is undefined and upstream-unstated; 2.4's 80% precedent; the point of the - experiment is to replace a judgment with a measurement, and all four are reversible and - branch-local (3.4). -3. **Strip the units with a live oracle, keep the ones without.** Strip (a) — CI is gating and will - catch a bad selection — and (c) — the advisory comment is a real, fast feedback loop. Keep (b), - which has no detector at all, and (d), whose oracle is fail-open outside one subtree and whose - ground truth is non-derivable per 3.3. -4. **Invert on derivability rather than on oracle.** Strip (a) and (d), whose rationale is fully - written down in files the model can read (`README.md` § "Validate a change"; `run-ruff.sh` - header + `docs/ci-runner-routing.md`), on the theory that the experiment tests whether the model - *finds* them. Keep (b) and (c), whose content is stated nowhere outside the instruction surface. -5. **Strip (b) under a named Gate 0 hold.** Treat (b) as `external-publication`, strip it anyway - (3.4 permits it), and pre-commit to restoring it at Phase 4 as a register hold regardless of the - ledger. Yields an observation without accepting the silence-as-evidence inference. -6. **Split (a) rather than strip it whole.** The unit is two claims: *use the selector* (derivable - from README) and *zero suites is an error* (a non-obvious failure-mode fact). SKILL.md Phase 1 - already licenses section-granular splitting of mixed files; the same mechanic applies here. diff --git a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D3-state-durability.md b/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D3-state-durability.md deleted file mode 100644 index a44bdaefbc..0000000000 --- a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D3-state-durability.md +++ /dev/null @@ -1,348 +0,0 @@ -# D3 — Where unhobble state should live for a cloud-session run - -Research memo. Evidence only; no recommendation. All web sources fetched **2026-09-11** with -WebFetch. Repo paths read at HEAD on 2026-09-11. - -## Question - -The unhobble contract puts `manifest.json`, `stumbles.md`, and `backups/` under -`${CLAUDE_PLUGIN_DATA}/unhobble//` (resolving to -`/unhobble//`). This run is a Claude -Code on the web session with an ephemeral container, and the experiment must survive days across -multiple sessions (observe phase, then readd). Where should the state live: (i) plugin data dir -plus operator hand-carry, (ii) committed on the experiment branch, (iii) some other durable home? - -## Findings - -### 1. Cloud-session container lifecycle and what persists - -**1.1 Each session gets a fresh VM.** -`https://code.claude.com/docs/en/cloud-environments` § "What's available in cloud sessions": -> "In Anthropic-hosted environments, each session gets a fresh virtual machine (VM) running -> Ubuntu 24.04 on x86_64 ... with your repository cloned and common toolchains pre-installed." - -**1.2 Only committed repo content carries over; `~/.claude` explicitly does not.** -Same page, § "What carries over from your setup": -> "Cloud sessions start from a fresh clone of your repository. Anything you commit to the repo is -> available. Anything you've installed or configured only on your own machine isn't available in -> the session." - -The same section's table marks `Your user ~/.claude/CLAUDE.md` — **Available in cloud sessions: No** -— "Lives on your machine, not in the repo", and the same for `~/.claude/skills/`, `~/.claude/agents/`, -`~/.claude/commands/`. The section closes: -> "To make your own configuration available in cloud sessions, commit it to the repo." - -This is about *your machine's* `~/.claude` reaching the VM. It does not directly state whether the -VM's own `~/.claude` survives between sessions — see Finding 1.4 and Unverified claim U1. - -**1.3 The VM is reclaimed on inactivity; a reopened session gets a fresh VM.** -`https://code.claude.com/docs/en/claude-code-on-the-web` § "Environment expired": -> "Cloud sessions stop after a period of inactivity and the session's VM is reclaimed." -> "Reopen the session from claude.ai/code to provision a fresh VM with your conversation history -> restored. Background work that was still running when the VM was reclaimed, such as subagents and -> shell commands, isn't restored." - -What is named as restored is **conversation history**. The filesystem is not named as restored. - -**1.4 The one filesystem that does persist is the environment cache — and it is a snapshot of the -setup script's output, not of session work.** -`cloud-environments` § "Environment caching": -> "The setup script runs the first time you start a session in an environment. After it completes, -> Anthropic snapshots the filesystem and reuses that snapshot as the starting point for later -> sessions." -> "The cache is a filesystem snapshot, so it keeps what the setup script writes to disk and loses -> anything that was only running." -> "The setup script runs again to rebuild the cache when you change the environment's setup script -> or allowed network hosts, and when the cache reaches its expiry after roughly seven days. -> Resuming an existing session never re-runs the setup script." - -So the snapshot is taken **after the setup script completes**, before Claude starts working. Nothing -the session itself writes is documented as entering the snapshot, and the snapshot expires in about -seven days regardless. - -**1.5 Resource ceiling, for completeness.** Same page § "Resource limits": "4 vCPUs", "16 GB of -RAM", "30 GB of disk". Not a constraint at this state size. - -**1.6 Isolation.** `claude-code-on-the-web` § "Security and isolation": "each session runs in an -isolated, Anthropic-managed VM." Sessions are isolated from each other, so one session's disk is -not a channel to the next. - -### 2. Plugin data directory: purpose, location, persistence, per-machine - -**2.1 Location and resolution.** -`https://code.claude.com/docs/en/plugins-reference` § "Persistent data directory": -> "The `${CLAUDE_PLUGIN_DATA}` directory resolves to `~/.claude/plugins/data/{id}/`, where `{id}` is -> the plugin identifier with characters outside `a-z`, `A-Z`, `0-9`, `_`, and `-` replaced by `-`." - -**2.2 What it promises.** Same page, env-var table: -> "`${CLAUDE_PLUGIN_DATA}` | Persistent directory that survives plugin updates, created on first -> reference | Installed dependencies such as `node_modules` or Python virtual environments, -> generated code, and caches" - -The documented guarantee is scoped to **plugin updates**, not to machines or containers. The stated -use case is dependency/cache reuse: -> "A common use is installing language dependencies once and reusing them across sessions and plugin -> updates." - -**2.3 It is deleted on uninstall.** Same section: -> "The data directory is deleted automatically when you uninstall the plugin from the last scope -> where it is installed." - -**2.4 Per-machine by construction.** It sits under `~/.claude/` (Finding 2.1), and Finding 1.2 -establishes `~/.claude` on your own machine does not reach a cloud session. No official page states -any cross-machine or cross-container replication of the data directory. - -**2.5 The `plugins` overview page does not mention it at all.** Fetched -`https://code.claude.com/docs/en/plugins` (2026-09-11): the page covers manifests, skills, agents, -hooks, LSP, monitors, `settings.json`, `--plugin-dir`, marketplaces. It contains no mention of -`CLAUDE_PLUGIN_DATA`, a plugin data directory, or plugin state persistence. `plugins-reference` is -the sole official source for this. - -**2.6 This repo already records the per-machine property.** -`/plugins/claude-config/skills/unhobble/SKILL.md` § "State": -> "`${CLAUDE_PLUGIN_DATA}` is machine-global, so two checkouts sharing a basename ... would -> otherwise resolve to one directory". -Same framing in -`/plugins/claude-config/skills/audit-pass/reference/run-state-and-resumability.md:15`: -> "`${CLAUDE_PLUGIN_DATA}` is machine-global, not per-project". - -### 3. Repo conventions (second tier) - -**3.1 The memory tier is never committed, and is explicitly lost to a reclaimed container.** -`/docs/conventions/topic-docs/README.md` § "The two tiers (and their -neighbors)" — the tier table rows: -> "| Memory | `.work//` | Never committed (self-ignoring) | ... |" -> "| Contract | `docs/topics//` | Committed **on the task branch only**; pruned before merge | ... |" -> "| Machine state | `${CLAUDE_PLUGIN_DATA}`; `.claude/observability/` | Never committed | telemetry; -> caches; durable machine-scoped state a later session reopens across projects |" - -`/.gitignore` carries the matching entry: -> "# Topic memory tier — never committed (docs/conventions/topic-docs/README.md \"Memory\")" -> ".work/" - -**3.2 The convention's own visibility matrix marks machine state invisible to a cloud clone.** -Same README, § "Visibility across execution contexts", "Context × tier visibility matrix": -> "| Cloud clone / CI checkout | invisible | pushed commits only | pushed state only | invisible |" - -The four columns are memory tier, contract tier, durable, and `${CLAUDE_PLUGIN_DATA}`. So the -repo's own normative contract already states: **in a cloud checkout, `${CLAUDE_PLUGIN_DATA}` is -invisible and only pushed commits carry.** The section is marked "This section is normative". - -**3.3 `docs/cloud-sessions.md` restates the same rule for this repo.** -`/docs/cloud-sessions.md` § "What this is": -> "runs each session in a fresh, isolated cloud VM with your repository cloned into it" -> "repo-committed `.claude/` config reaches cloud sessions; user-level `~/.claude` config never -> does." - -**3.4 Sibling plugins in this marketplace treat "must survive a reclaimed container" as the trigger -for a tracked file.** `plugins/instruction-placement/reference/topic-docs.md` § "What survives what": -> "| A deleted memory root, a reclaimed container | lost | lost | kept — it is tracked, not memory tier |" -> "| A fresh clone | absent | absent | **present** |" -and: -> "The third and fifth rows are the whole reason the suppression surface exists. Git is the -> mechanism: a tracked file reaches another checkout because git moves it, and no `memory_dir` -> setting makes a memory-tier file do the same". - -The durable half is a **tracked file under `.claude/`**: `.claude/instruction-placement.md` -(`plugins/instruction-placement/reference/consumer-config.md` § "`suppressions`": "persisted so it -survives the branch switches, other worktrees, removed memory roots, and reclaimed containers that -lose the memory-tier findings artifact"). `overengineering` uses the same split — -`plugins/overengineering/reference/topic-docs.md` § "What this plugin writes": -> "Both artifacts are therefore lane-local and **ephemeral by design** — a branch switch, a removed -> worktree, or a reclaimed container loses them. That is acceptable for evidence, verdicts, and a -> comparison baseline, all of which a run recomputes or recaptures, and is exactly why operator -> judgments are not kept here." -Its durable half is the tracked `.claude/overengineering.md`. - -**3.5 No sibling skill offers a branch-committed state option today.** Grep across -`plugins/claude-config/skills/*/SKILL.md`: `audit-instructions`, `audit-prompting-postures`, -`audit-pass`, and `unhobble` all persist to `${CLAUDE_PLUGIN_DATA}`; -`plugins/claude-config/skills/audit/SKILL.md:108-113` writes `.work/claude-config-audit/findings.json` -(memory tier). No skill in `claude-config`, `overengineering`, or `instruction-placement` documents a -committed-state or repo-path alternative for its *run state*. The only tracked writes any of them make -are consumer **config/judgment** files (`.claude/overengineering.md`, `.claude/instruction-placement.md`), -and both are ask-gated. - -**3.6 Committed-on-the-task-branch is a named tier, with a lifecycle.** The contract tier -(Finding 3.1) is exactly "committed on the task branch only; pruned before merge"; the README's -"Contract-slice lifecycle (prune with pointer)" section owns the pruning rule. `docs/topics/` does -not currently exist in the repo, but it is the documented default `contract_dir`, and -`scripts/docs-only-paths.txt` contains exactly one entry: `docs/topics/`. - -### 4. What the unhobble "State" section promises — and what it does not say - -`plugins/claude-config/skills/unhobble/SKILL.md` lines 52-71. It promises, verbatim: - -- The location: "`${CLAUDE_PLUGIN_DATA}/unhobble//`". -- Identity discipline: "The manifest therefore records the canonical checkout identity, the resolved - absolute worktree path and, when a remote exists, the origin URL, and every later phase verifies - it matches the current checkout before acting; a mismatch aborts with the conflicting path named." -- No reuse: "`snapshot` never reuses an existing experiment directory: a fresh run mints a fresh id, - and resuming an open experiment means passing its phase commands from inside the same checkout its - manifest names." -- The three artifacts: `manifest.json`, `stumbles.md`, `backups/` ("pre-strip copies of any - non-git-tracked file modified"). - -**It does not say state must not be committed.** There is no prohibition on a repo path anywhere in -the file. `## What this skill does NOT do` lists four items, none about state location. So -committing is **unaddressed, not forbidden** — with three contract-adjacent frictions: - -- **F1.** The manifest records "the resolved absolute worktree path" and "every later phase verifies - it matches the current checkout". A committed manifest carrying `` - aborts any later phase run from a different path (a local checkout, a differently-rooted container). -- **F2.** `backups/` holds pre-strip copies of non-tracked files, explicitly "e.g. settings hook - entries" — i.e. `.claude/settings.local.json`-class content, which this repo gitignores - (`.gitignore`: `.claude/settings.local.json`, `.claude/**/*.local.*`). Committing backups commits - content the repo has decided not to track. -- **F3.** Phase 2 already commits: "One commit, message `experiment: strip instruction surfaces for - unhobble baseline`." The branch is already the carrier for the *stripped surfaces*; the state dir - is the only part the contract routes elsewhere. - -### 5. CI gates that would act on a committed state file - -From `/.github/workflows/ci.yml`, the `lint` job (lines 269-560) plus -the changelog-parity steps (lines 885-930). - -**5.1 Always-on, diff-scoped, no path allowlist in this repo's configs.** The ci.yml comment at -line 269 states markdownlint / typos / editorconfig / gitleaks / eol-renormalize / comment-hygiene / -machine-specific-paths "read the changed docs themselves and run on every diff". Their configs carry -no repo-path scoping: - -- `.markdownlint-cli2.jsonc` — `"ignores"` is only `**/node_modules/**`, `**/.venv/**`, `**/bin/**`, - `**/obj/**`; the file's own comment says "Globs are passed by the caller (CLI / CI), not declared - here." So a `.md` anywhere (including `.claude/**` and `docs/**`) is linted. MD013 is off, MD024 - siblings-only, MD004 dash bullets, MD003 ATX. A `stumbles.md` table is fine; watch MD012/MD047-class - defaults (all rules on unless listed). -- `_typos.toml` — no path scoping; "typos already respects .gitignore and skips binary files". Free - prose in `stumbles.md` is spell-checked. -- `.editorconfig-checker.json` — `Exclude` is build/dependency dirs only. So `.editorconfig` applies: - `[*]` requires `insert_final_newline`, `trim_trailing_whitespace`, `charset = utf-8`; - `[*.{json,jsonc,yml,yaml,toml}]` sets `indent_size = 2` (but `IndentSize` is **disabled** in - `.editorconfig-checker.json`, so indent width is not gated); `[*.{md,markdown}]` sets - `trim_trailing_whitespace = false`. -- `.gitleaks.toml` — `[extend] useDefault = true`, no repo-specific rules. CI runs it with - `scan-mode: git`, `redact: true`. This is the gate `backups/` (F2) collides with: pre-strip copies - of settings files are exactly the class of content the default ruleset scans for. -- **machine-specific-paths** (`ci.yml:484-486`) — a dedicated gate for absolute host paths. The - manifest's "resolved absolute worktree path" (F1) is precisely its target; the step's `with:` - block is a hand-maintained exception list for fixtures that must carry such paths. - -**5.2 changelog-parity fires on any path under `plugins//`.** -`scripts/check-changelog-parity.sh` case arms include `plugins/*/*` (line 480), and the error text at -line 639 reads: "still carries $head_version while this change set modifies files under -plugins/$name/ — bump the manifest and add a new '## [$head_version]' release entry instead of -editing in place." So state committed under `plugins/claude-config/` would demand a version bump plus -a changelog entry **per experiment write**. State under `.claude/…` or `docs/…` does not (the other -in-scope arm is `docs/conventions/*/CHANGELOG.md`, line 484). - -**5.3 The em-dash gate is an allowlist, so a new path is unenforced.** -`scripts/em-dash-purged-paths.txt` lists specific plugin READMEs and `SKILL.md` globs. Neither -`.claude/unhobble/**` nor a `docs/` state path appears, so `check-purged-em-dashes.sh` would not gate -free prose there. (Adding an entry is optional and would then require the file to stay purged.) - -**5.4 `docs/topics/` is the only docs-only allowlist entry.** `scripts/docs-only-paths.txt` contains -exactly `docs/topics/`. A diff confined there lets the path-scoped linters (actionlint, the four -check-jsonschema steps, the manifest duplicate-key detector) report an honest not-applicable success -(`ci.yml:269-278`, `ci.yml:1227`). ShellCheck, exec-bit, hook-wiring-liveness and purged-em-dashes -stay unconditional. The always-on lints in 5.1 still run. - -**5.5 Ignore status of candidate paths** (`git check-ignore`, run 2026-09-11 at repo root): -`.claude/unhobble/manifest.json` → not ignored (would be tracked); -`docs/unhobble/manifest.json` → not ignored; `docs/topics/foo/PLAN.md` → not ignored; -`.work/x/y.md` → **ignored**. The repo has no `.claude/topic-docs.yaml`, so `memory_dir` is the -default `.work/`. - -## Consensus table - -| Claim | Official docs | Repo conventions | Sibling skills | Verdict | -|---|---|---|---|---| -| Each cloud session gets a fresh VM | Yes (1.1) | Yes (3.3) | — | Consensus | -| VM is reclaimed on inactivity; reopen provisions a fresh VM | Yes (1.3) | implied (3.2) | "reclaimed container loses them" (3.4) | Consensus | -| Only committed repo content reaches a cloud session | Yes (1.2) | Yes (3.2, 3.3) | — | Consensus | -| `${CLAUDE_PLUGIN_DATA}` is `~/.claude/plugins/data/{id}/` | Yes (2.1) | Yes (2.6) | Yes (2.6) | Consensus | -| Its persistence guarantee is "survives plugin updates", not machines | Yes (2.2) | — | — | Consensus (docs make no cross-machine claim) | -| `${CLAUDE_PLUGIN_DATA}` is invisible to a cloud clone | not stated | **Yes, normative** (3.2) | Yes (3.4) | Repo-only; official docs silent | -| A state need that must outlive a container goes to a tracked file | — | Yes (3.1 tiers) | Yes (3.4) | Consensus within repo | -| unhobble state may be committed | — | unaddressed | no precedent for run state (3.5) | No source forbids it; no source sanctions it | -| Conversation history survives VM reclamation | Yes (1.3) | — | — | Consensus | -| Session **filesystem** survives VM reclamation | not stated; "fresh VM" (1.3) | "reclaimed container loses them" (3.4) | — | Disagreement is absent, but official wording is indirect — see U1 | - -## Unverified claims - -- **U1.** No official page says in so many words "the session's filesystem is discarded when the VM - is reclaimed" or "`~/.claude` on the VM does not persist between sessions". The conclusion - rests on three indirect statements: "each session gets a fresh virtual machine" (1.1), "provision a - fresh VM with your conversation history restored" (1.3, which enumerates history and not disk), and - the caching section's statement that the reused snapshot is the one taken after the **setup - script** completes (1.4). Treat "cloud-session writes to `~/.claude` are lost" as **strongly - supported by inference, not a verbatim official guarantee.** A cheap empirical check exists: - write a sentinel file under `~/.claude/plugins/data/` and look for it in the next session. -- **U2.** Whether the ~7-day environment-cache snapshot could incidentally carry a session's writes - is not addressed either way; the wording ("keeps what the setup script writes to disk") points - against it, and the ~7-day expiry caps it regardless. Unverified. -- **U3.** The exact glob each `melodic-software/ci-workflows` composite action (markdown, typos, - gitleaks, editorconfig, machine-specific-paths, comment-hygiene) passes was not read — that repo - is external and was not fetched. Path-scope conclusions in 5.1 come from this repo's configs and - the ci.yml comments, not from the composites' source. -- **U4.** Whether `` is a stable absolute path across successive cloud - VMs for this repository is not documented; F1's severity depends on it. - -## Options the evidence supports, with the constraints each carries - -### (i) Plugin data dir per the contract, operator hand-carries a copy - -- Contract-conformant: no deviation to record, no manifest/identity rules bent (Finding 4). -- The state is lost at VM reclamation on the evidence of 1.1/1.3/1.4 plus 3.2/3.4 (subject to U1). - Every cross-session continuity step is a manual operator action with no gate behind it: a missed - copy is a silently lost `stumbles.md`, which SKILL.md Phase 3 calls "the experiment's entire - evidentiary output". -- The observe phase spans days and multiple sessions (SKILL.md Phase 3: "days of real work"), so the - hand-carry is not one-off; it is per-session, in both directions (restore before, capture after). -- No CI gate touches it. No changelog, no lint, no secret scan. -- F1 still bites if the operator carries the state to a machine whose checkout path differs from the - recorded absolute worktree path: "a mismatch aborts with the conflicting path named". - -### (ii) Committed on the experiment branch under a repo path - -- Matches the repo's contract tier verbatim — "Committed **on the task branch only**; pruned before - merge" (3.1) — and is the only mechanism the repo's own normative matrix says reaches a cloud - clone ("pushed commits only", 3.2). Phase 2 already commits to this branch (F3). -- Deviates from the skill's documented State location. Not forbidden (Finding 4), but no sibling - skill has a committed run-state precedent (3.5); the tracked-file precedents that do exist are - ask-gated consumer *config*, not run state. -- Gates, by path: - - under `plugins/claude-config/…` — adds changelog-parity (5.2): a version bump plus a changelog - entry on every state write. Highest friction. - - under `.claude/unhobble/…` or `docs/…` — no changelog-parity; markdownlint, typos, editorconfig, - gitleaks, machine-specific-paths all still apply (5.1, 5.5). - - under `docs/topics//` — additionally the sole docs-only allowlist entry (5.4), so - path-scoped linters report not-applicable; and the tier already has a documented prune-before-merge - lifecycle (3.6). -- Two specific collisions: **machine-specific-paths** vs the manifest's recorded absolute worktree - path (F1, 5.1), and **gitleaks** vs `backups/` holding pre-strip copies of settings files the repo - otherwise gitignores (F2, 5.1). Both are properties of what the contract says the state *contains*, - not of the choice to commit per se; a split (commit `stumbles.md` + a path-free manifest, leave - `backups/` in the plugin data dir) would address both but is itself a contract deviation. -- `stumbles.md` free prose enters the spell-check and markdown lanes on every push (5.1); the em-dash - gate does not apply unless a path is added to the allowlist (5.3). - -### (iii) Another durable home - -The evidence surfaced these, unranked: - -- **A tracked file under `.claude/`**, the shape both `instruction-placement` and `overengineering` - already use for the judgment half of their state (3.4): `.claude/instruction-placement.md`, - `.claude/overengineering.md`. Precedent is for durable *operator judgments*, not raw run state, - and both are ask-gated writes. Same gate profile as the `.claude/…` row above; not ignored (5.5). -- **The forge as the store**: the ledger in a PR body or a tracker issue on the experiment branch. - The topic-docs README rejects this shape generally — § "Native mechanisms": "Markdown-in-tickets as - a primary artifact store is rejected: ticket bodies are not diffable, carry no review gate, and - drift from code." No CI gate touches it; no repo precedent endorses it. -- **Run the observe phase off cloud** (a local or self-hosted checkout), leaving the contract - location intact and the durability problem out of scope. SKILL.md Phase 2 already requires "a - fresh session after stripping" and Phase 3 requires days of real work, neither of which binds the - work to this surface. Constraint: F1's identity check then pins every later phase to that one - machine's checkout path. -- **Environment setup script / cache**: ruled out by 1.4 — the snapshot is built from the setup - script's writes and expires in roughly seven days, which is shorter than the observe window the - skill describes. diff --git a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/manifest-mirror.json b/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/manifest-mirror.json deleted file mode 100644 index ef8a60e272..0000000000 --- a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/manifest-mirror.json +++ /dev/null @@ -1,89 +0,0 @@ -{ - "experiment_id": "claude-code-plugins-fable-5-1-20260908-15744de6", - "phase": "observe", - "phase_timestamps": { - "snapshot": "2026-09-08T17:20:00Z", - "bare": "2026-09-11T00:00:00Z", - "observe": "2026-09-12T01:30:00Z" - }, - "phase_note": "observe opens with the merge of the experiment branch into main (checkout.merged_to_main); the window runs from fresh sessions on main and closes when readd runs.", - "decisions": { - "method": "Three source-backed research memos (official Claude Code docs fetched live 2026-09-11, repo doctrine second-tier), then two independent decision agents (Fable 5.1 and Opus 5) blind to each other; all four verdicts agreed. Memos and verdicts are mirrored under .claude/unhobble// on the experiment branch.", - "D1_user_scope": "Not opted in. No plugin ablated this run: every hook-wiring plugin classified keep (policy, tooling, or ground-truth oracle; context-guard is unstripped-hybrid-hook), and the 53 skill-only plugins need a classification pass before any project-scope enabledPlugins:false overlay is applied. The overlay is the documented, CI-passing mechanism for a future plugin ablation (settings-reference: project beats user; check-plugin-catalog-enablement.sh passes an explicit false).", - "D2_conventions": "Strip (a) AGENTS.md 'Validate a change' (gating CI oracle remains: affected-tests.sh in test-linux) and (c) pr-body-contract.md (advisory CI oracle plus .claude/source-control.md remain; stripped under a register hold, contested external-publication, so Phase 4 restores it regardless of the ledger). Keep (b) draft-PR rule (no oracle at all, contested external-publication) and (d) ruff-pin.md (non-derivable ground-truth oracle per PLUGIN-PHILOSOPHY 'Classifying a hook').", - "D3_state": "Canonical state stays in the plugin data dir per the contract. A durable mirror (stumbles.md as the primary ledger, manifest-mirror.json with absolute paths tokenized) is committed on the experiment branch at .claude/unhobble//. Amended 2026-09-11 after the first CI run: the decision agents chose docs/topics// (contract tier), but scripts/check-contract-slice-prune.sh --check-diff fails any pull request that leaves a path under docs/topics/, so a mirror there keeps ci-status red for the whole observe window; .claude/ is the home the instruction-placement and overengineering plugins already use for durable tracked state, is not ignored, and trips no contract-slice or changelog-parity gate. backups/ never leaves the data dir and is empty this run.", - "D4_classification": "No class changes. Annotations: agents-draft-pr and rule-pr-body-contract carry register_class_contested external-publication; rule-ruff-pin cites the ground-truth-oracle carve-out. Nested plugin AGENTS.md units left unstripped under the skill's two-hats rule (files under plugins// ship to consumers and trip changelog-parity, which has no exemption list)." - }, - "checkout": { - "worktree_path": "", - "origin_url": "https://github.com/melodic-software/claude-code-plugins", - "base_commit": "c41c6422", - "base_commit_note": "Snapshot taken at 2dfaaa40 on 2026-09-08; branch fast-forwarded to origin/main c41c6422 on 2026-09-11 before the strip. Delta re-inventoried: settings.json gained permissions.deny (policy, kept); skill-bodies rule grew a stamped deviation paragraph inside the section already planned for trim; .claude/audit-pass.md, .claude/bugs.md, .claude/code-metrics.yaml are plugin config (kept).", - "branch": "claude/unhobble-config-oe7yfx", - "strip_commit": "4f6180f2", - "pull_request": "https://github.com/melodic-software/claude-code-plugins/pull/4090", - "branch_note": "Session-designated branch used as the experiment branch; the suggested experiment/unhobble-fable-5-1 name was not created because this session may push only to its designated branch.", - "merged_to_main": "Decision recorded 2026-09-12, before the merge landed: the operator asked for the experiment branch (pull request #4090) to merge into main so the observe window runs on main rather than on a draft branch. From that merge on, the bare state (every strip in this manifest, the register-hold strip of rule-pr-body-contract included) is the state of main and the observe phase runs from fresh sessions on main. The readd phase starts from a fresh branch off main, restores rule-pr-body-contract regardless of the ledger, and restores only what stumbles.md defends with repeated same-cause rows. Until #4090 is merged, main still carries every surface and this record describes the intended state.", - "ephemeral_container": true, - "ephemeral_note": "The snapshot ran in an ephemeral cloud container. That container is reclaimed once the session that opened #4090 ends, and the plugin-data copy of this manifest goes with it; from then on this mirror is the canonical record. Readd must rehydrate it into the plugin data path (see README, 'Restoring the canonical state before readd') or re-snapshot from the pre-strip commit c41c6422 before it runs from a durable checkout.", - "state_dir": "/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6", - "base_merges": [ - "2026-09-11: origin/main merged into the experiment branch (7 commits past c41c6422); checked with git log over CLAUDE.md, AGENTS.md, .claude/rules, .claude/settings.json and every nested AGENTS.md: no instruction surface changed, so the bare state is unaffected. Brought in the re-enabled rule-em-dash in .claude/ai-slop.json (PR #3988)." - ] - }, - "target_model": { - "id": "claude-fable-5-1", - "cli_version": "2.1.263" - }, - "scope": { - "project_surfaces": true, - "user_global_surfaces": false, - "managed_settings": "none present (/etc/claude-code absent)", - "user_global_note": "~/.claude/settings.json enables all 73 marketplace plugins at user scope (cloud environment snapshot + cloud-bootstrap.sh); 20 of them wire hooks. Out of scope by default; recorded as confounds. Operator may opt in explicitly." - }, - "surfaces": [ - { "id": "claude-md-shim", "path": "CLAUDE.md", "lines": 1, "loads": "always", "class": "convention", "action": "keep", "reason": "Import shim (@AGENTS.md); stays while AGENTS.md exists." }, - { "id": "agents-validate", "path": "AGENTS.md#validate-a-change", "lines": 6, "loads": "always", "class": "convention", "action": "strip", "mechanism": "edit: section removed from AGENTS.md", "restore": "git", "reason": "Which command validates a change; full contract owned by README.md 'Validate a change' and the script's own exit codes. Stripped per D2: the CI selector in test-linux is a gating oracle." }, - { "id": "agents-draft-pr", "path": "AGENTS.md#open-a-pull-request-as-a-draft", "lines": 4, "loads": "always", "class": "convention", "register_class_contested": "external-publication", "action": "keep", "reason": "Team PR flow with a CI-cost consequence; no oracle at all. Kept per D2." }, - { "id": "agents-rules-index", "path": "AGENTS.md#conventions-that-load-on-demand", "lines": 23, "loads": "always", "class": "behavioral", "action": "regenerate", "restore": "git", "reason": "Generated discoverability scaffolding for deferred surfaces. Re-rendered with plugins/instruction-placement/scripts/render-index.sh after the strip so it lists only surviving deferred surfaces." }, - { "id": "rule-vendor-docs", "path": ".claude/rules/vendor-docs-are-not-style.md", "lines": 11, "loads": "always", "class": "behavioral", "action": "strip", "mechanism": "git rm", "restore": "git", "reason": "Style coaching (no em dashes from vendor docs; run /ai-slop:audit). Partial oracle exists: scripts/check-purged-em-dashes.sh in CI. Cross-surface conflict noted: .claude/ai-slop.json disables rule-em-dash as house style." }, - { "id": "rule-pr-body-contract", "path": ".claude/rules/pr-body-contract.md", "lines": 25, "loads": "always", "class": "convention", "register_class_contested": "external-publication", "register_hold": true, "action": "strip", "mechanism": "git rm", "restore": "git", "reason": "PR body contract; the ci-status pr-contract composite is the authority and reports (advisory label), not gates; .claude/source-control.md keeps the section list. Stripped per D2 under a register hold: Phase 4 restores it regardless of the ledger." }, - { "id": "rule-catalog-taxonomy", "path": ".claude/rules/catalog-taxonomy.md", "lines": 12, "loads": "on-read .claude-plugin/marketplace.json", "class": "behavioral", "action": "strip", "mechanism": "git rm", "restore": "git", "reason": "Pointer-only rule; docs/CATALOG-TAXONOMY.md owns the taxonomy and stays. Measures whether the bare model finds the owner doc." }, - { "id": "rule-hook-budget", "path": ".claude/rules/hook-budget.md", "lines": 13, "loads": "on-read plugins/*/hooks/**", "class": "behavioral", "action": "strip", "mechanism": "git rm", "restore": "git", "reason": "Pointer-only rule; docs/conventions/hook-budget/README.md owns the budget and stays." }, - { "id": "rule-ruff-pin", "path": ".claude/rules/ruff-pin.md", "lines": 13, "loads": "on-read **/*.py", "class": "convention", "action": "keep", "reason": "Repo tooling convention (pinned wrapper vs bare ruff); docs/CI-RUNNER-ROUTING.md owns the rationale. Kept per D2 under the ground-truth-oracle carve-out: a bare ruff disagrees with the CI pin silently, which no model can derive unaided." }, - { "id": "rule-skill-bodies", "path": ".claude/rules/skill-bodies-state-current-rules.md", "lines": 61, "loads": "on-read plugins/*/skills/**, plugins/*/agents/**", "class": "hybrid", "action": "trim", "mechanism": "edit: strip the 'keep out of a body' coaching list, its derivation from the bundled prompt-audit guide, and the history-placement paragraphs; keep the '## Next' successor-section contract and the four-part verification-record requirement", "restore": "git", "reason": "Writing coaching (behavioral, derivable from the bundled guide) sits beside two structural conventions other tooling depends on." }, - { "id": "rule-worktree-base-ref", "path": ".claude/rules/worktree-base-ref.md", "lines": 21, "loads": "on-read .claude/settings.json, .claude/settings.local.json", "class": "behavioral", "action": "strip", "mechanism": "git rm", "restore": "git", "reason": "Corrects a specific recurring misread (restoring a key a review bot calls dropped). Ground truth is git history, which the model can query; consequence is a reviewable settings key, not Gate 0." }, - { "id": "nested-autonomy", "path": "plugins/autonomy/AGENTS.md (+CLAUDE.md shim)", "lines": 9, "loads": "on-read plugins/autonomy/**", "class": "behavioral", "action": "unstripped-product-surface", "reason": "Pre-empts the CI typos gate on coined hyphenated compounds; the gate is the deterministic oracle and stays. Left in place: it ships inside plugins/autonomy/ (two-hats rule) and any edit there trips changelog-parity, which has no exemption list. Confound for the observe phase, path-scoped so it loads only when that subtree is read." }, - { "id": "nested-machine-health", "path": "plugins/machine-health/skills/audit/AGENTS.md (+CLAUDE.md shim)", "lines": 27, "loads": "on-read plugins/machine-health/skills/audit/**", "class": "behavioral", "action": "unstripped-product-surface", "reason": "Pointer-only to co-located README sections. Left in place for the same two-hats and changelog-parity reasons as nested-autonomy; confound noted." }, - { "id": "nested-model-adaptation", "path": "plugins/playbooks/reference/model-adaptation/AGENTS.md", "lines": 17, "loads": "on-read plugins/playbooks/reference/model-adaptation/**", "class": "policy", "register_class": "legal-compliance", "action": "keep", "reason": "Verbatim-quotation and attribution rail in a public repo; protected class, never left deleted on silence." }, - { "id": "nested-provenance", "path": "plugins/attribution/skills/audit/AGENTS.md", "lines": 23, "loads": "on-read plugins/attribution/skills/audit/**", "class": "hybrid", "action": "unstripped-product-surface", "planned_trim": "remove the 'affected-tests.sh --run exits 3' section (restates README); keep the 'measurement results out of rubric.md' judge-contamination invariant", "reason": "One derivable restatement beside one repo-specific data-integrity invariant. Left in place: ships inside plugins/provenance/ (two-hats rule) and an edit trips changelog-parity; confound noted." }, - { "id": "nested-work-loop", "path": "plugins/work-items/skills/work-loop/AGENTS.md", "lines": 10, "loads": "on-read plugins/work-items/skills/work-loop/**", "class": "convention", "action": "keep", "reason": "Manual test procedure for an LLM-executed gate with no automated surface; not derivable." }, - { "id": "hook-sessionstart-bootstrap", "path": ".claude/settings.json#hooks.SessionStart[0]", "loads": "startup|resume", "mechanism_axis": "notification/infra", "class": "policy", "action": "keep", "reason": "Provisions the cloud VM toolchain and plugin registry; no model-facing payload." }, - { "id": "settings-permissions-deny", "path": ".claude/settings.json#permissions.deny", "class": "policy", "register_class": "secret-handling", "action": "keep", "reason": "Read-deny list for .env, secrets/, keys and the local settings file; protected class, never stripped." }, - { "id": "settings-env-telemetry", "path": ".claude/settings.json#env.HOOK_TELEMETRY_SINK", "class": "policy", "action": "keep", "reason": "Observability wiring, not an instruction." }, - { "id": "settings-auto-memory-off", "path": ".claude/settings.json#autoMemoryEnabled", "class": "policy", "action": "keep", "reason": "Operator memory choice; also keeps the bare baseline free of self-written notes." }, - { "id": "settings-skill-budget", "path": ".claude/settings.json#skillListingBudgetFraction", "class": "convention", "action": "keep", "reason": "Skill-listing budget, not an instruction surface." }, - { "id": "plugin-fleet", "path": ".claude/settings.json#enabledPlugins.fleet", "class": "convention", "action": "keep", "reason": "Skill-only capability plugin (SSH to other machines), no hooks; tooling, not a model correction." }, - { "id": "plugin-playgrounds", "path": ".claude/settings.json#enabledPlugins.playgrounds", "class": "convention", "action": "keep", "reason": "Already disabled." }, - { "id": "plugin-config-files", "path": ".claude/source-control.md, .claude/audit-pass.md, .claude/bugs.md, .claude/code-metrics.yaml, .claude/ai-slop.json, .claude/attribution.json, .work-item-tracker.json", "class": "convention", "action": "keep", "reason": "Plugin configuration read by skills at invocation; not context-loaded instruction." } - ], - "confounds": [ - { "kind": "user-scope-plugins", "count": 73, "with_hooks": ["actionlint", "autonomy", "bash-format", "biome-format", "claude-ops", "context-budget", "context-guard", "desktop-notification", "disk-hygiene", "eol-normalizer", "go-format", "guardrails", "instruction-placement", "markdown-format", "powershell-format", "rate-limit-guard", "ruff-format", "session-flow", "source-control", "typos-format"], "note": "Enabled at ~/.claude/settings.json by the cloud environment snapshot. Every skill listing and hook from these plugins stays loaded in the bare session; see decision D1." }, - { "kind": "user-scope-skills", "paths": ["~/.claude/skills/session-start-hook", "~/.claude/skills/synced"], "note": "User scope; out of scope by default." }, - { "kind": "unstripped-product-surface", "paths": ["plugins/autonomy/AGENTS.md", "plugins/machine-health/skills/audit/AGENTS.md", "plugins/attribution/skills/audit/AGENTS.md"], "note": "Path-scoped nested AGENTS.md files inside plugin directories; load only when those subtrees are read. See decision D4." }, - { "kind": "harness-injected", "note": "The remote-session system prompt (PR flow, GitHub rules, git branch requirements) is product-side and not part of this contract." } - ], - "strip_plan_summary": { - "strip_whole": ["rule-vendor-docs", "rule-catalog-taxonomy", "rule-hook-budget", "rule-worktree-base-ref", "rule-pr-body-contract", "agents-validate"], - "trim": ["rule-skill-bodies"], - "regenerate": ["agents-rules-index"], - "keep_convention": ["agents-draft-pr", "rule-ruff-pin", "nested-work-loop"], - "keep_policy": ["nested-model-adaptation", "hook-sessionstart-bootstrap", "settings-permissions-deny", "settings-env-telemetry", "settings-auto-memory-off"], - "unstripped_product_surface": ["nested-autonomy", "nested-machine-health", "nested-provenance"], - "register_holds": ["rule-pr-body-contract"], - "always_loaded_lines_before": 74, - "always_loaded_lines_after": 5 - }, - "commit_message": "experiment: strip instruction surfaces for unhobble baseline", - "mirror_note": "Durable mirror of the canonical manifest in the plugin data dir; absolute paths tokenized. The canonical file is authoritative when both exist." -} diff --git a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/stumbles.md b/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/stumbles.md deleted file mode 100644 index e84820fa83..0000000000 --- a/.claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/stumbles.md +++ /dev/null @@ -1,8 +0,0 @@ -# Stumble ledger: claude-code-plugins-fable-5-1-20260908-15744de6 - -Primary ledger for the unhobble experiment on this branch. One row per observed failure in the -bare state; mark deletions that prove themselves with severity `improvement`. A row earns an -instruction back only when a second row shares its cause (Phase 4 gate). - -| Date | Task | What happened | Expected | Suspected missing instruction | Severity | -|---|---|---|---|---|---| diff --git a/docs/adr/0029-admit-a-second-findings-producer-behind-a-targeted-run-clause.md b/docs/adr/0029-admit-a-second-findings-producer-behind-a-targeted-run-clause.md index c893236316..036f6ed42a 100644 --- a/docs/adr/0029-admit-a-second-findings-producer-behind-a-targeted-run-clause.md +++ b/docs/adr/0029-admit-a-second-findings-producer-behind-a-targeted-run-clause.md @@ -94,6 +94,7 @@ over a corpus, so it was measured once on this repository before shipping: | Real, after independent re-derivation | **1** | The one is `docs/hook-migration-audit.md`, which no file in the repository references under any form. +It was deleted in #5790. Both rejections are more instructive than the survivor. `docs/adr/0006-...` is cited twice, but under the `ADR 0006` form rather than the filename form, caught by this lane's own query-form-variation rule. `docs/ai-briefing-design.md` is cited by `docs/migration-playbook.md`, and was missed because diff --git a/docs/architecture/landscape.json b/docs/architecture/landscape.json index 8a6fda601e..385786c191 100644 --- a/docs/architecture/landscape.json +++ b/docs/architecture/landscape.json @@ -59,13 +59,13 @@ {"from":"claude-code-plugins","to":"lycheeverse/lychee","type":"cites","relation":"external","count":1,"files":["lychee.toml"]}, {"from":"claude-code-plugins","to":"mattpocock/skills","type":"cites","relation":"external","count":5,"files":["docs/upstream/aihero-course.md","docs/upstream/aihero-shipping-course.md","docs/upstream/mattpocock-skills.md","plugins/wizard/CHANGELOG.md"]}, {"from":"claude-code-plugins","to":"melodic-software/.github","type":"cites","relation":"internal","count":1,"files":["docs/adr/0037-seat-mandatory-reviews-on-the-operator-session-and-retire-the-oauth-lanes.md"]}, - {"from":"claude-code-plugins","to":"melodic-software/ci-workflows","type":"cites","relation":"internal","count":153,"files":[".claude/source-control.md",".claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D2-conventions.md",".claude/unhobble/claude-code-plugins-fable-5-1-20260908-15744de6/evidence/research-D3-state-durability.md",".github/actionlint.yaml",".github/dependabot.yml"]}, + {"from":"claude-code-plugins","to":"melodic-software/ci-workflows","type":"cites","relation":"internal","count":150,"files":[".claude/source-control.md",".github/actionlint.yaml",".github/dependabot.yml"]}, {"from":"claude-code-plugins","to":"melodic-software/claude-code-account-rotation","type":"cites","relation":"internal","count":1,"files":["docs/adr/0038-restore-the-claude-review-lanes-on-every-push.md"]}, {"from":"claude-code-plugins","to":"melodic-software/claude-code-plugins-ci","type":"cites","relation":"internal","count":1,"files":["package.json"]}, {"from":"claude-code-plugins","to":"melodic-software/dotfiles","type":"cites","relation":"internal","count":8,"files":["docs/adr/0004-rightsize-instruction-surfaces-by-incumbent-first-arbitration.md","docs/ci-runner-routing.md","plugins/dometrain/README.md","plugins/machine-health/CHANGELOG.md","plugins/miro/README.md"]}, {"from":"claude-code-plugins","to":"melodic-software/github-iac","type":"cites","relation":"internal","count":19,"files":[".github/actionlint.yaml",".github/dependabot.yml",".github/workflows/ci.yml",".github/workflows/silent-revert-canary.yml","docs/adr/0002-default-on-ai-review-advisory-with-earned-promotion.md"]}, {"from":"claude-code-plugins","to":"melodic-software/knowledge-corpus","type":"cites","relation":"internal","count":3,"files":["docs/adr/0022-consume-the-knowledge-corpus-from-a-separate-repository.md","docs/knowledge-integration-design.md","plugins/instruction-placement/skills/migrate/reference/sources.md"]}, - {"from":"claude-code-plugins","to":"melodic-software/medley","type":"cites","relation":"internal","count":15,"files":["docs/adr/0020-defer-three-medley-surfaces-with-explicit-recheck-triggers.md","docs/ai-briefing-design.md","docs/conventions/ecosystem-commands/CHANGELOG.md","docs/hook-migration-audit.md","docs/migration-playbook.md"]}, + {"from":"claude-code-plugins","to":"melodic-software/medley","type":"cites","relation":"internal","count":7,"files":["docs/adr/0020-defer-three-medley-surfaces-with-explicit-recheck-triggers.md","docs/ai-briefing-design.md","docs/conventions/ecosystem-commands/CHANGELOG.md","docs/migration-playbook.md"]}, {"from":"claude-code-plugins","to":"melodic-software/miro-mcp","type":"cites","relation":"internal","count":1,"files":["plugins/miro/server/package.json"]}, {"from":"claude-code-plugins","to":"melodic-software/provisioning","type":"cites","relation":"internal","count":5,"files":["plugins/harness-ops/skills/observability/context/operator-setup-collector-daemon.md","plugins/harness-ops/skills/observability/context/operator-setup.md","plugins/harness-ops/skills/observability/context/otel-pipeline.md"]}, {"from":"claude-code-plugins","to":"melodic-software/runner-policy-runtime","type":"cites","relation":"internal","count":1,"files":[".github/standards/runner-policy/package.json"]}, diff --git a/docs/catalog.md b/docs/catalog.md index 318d47011e..21f1180ed9 100644 --- a/docs/catalog.md +++ b/docs/catalog.md @@ -66,7 +66,6 @@ plugin manifests and kept in sync by CI. Never hand-edit it; the category vocabu - [`overengineering`](../plugins/overengineering): Evidence-earned-keep audit of an existing enforcement surface, covering agent hooks and standing instructions, repository and version-control hooks, CI lanes and gate scripts, branch protections, forge apps, and declared external integrations. It treats every incumbent mechanism as a retirement candidate until empirical evidence earns its keep, argues every verdict in cost of carry, caps retirement-direction verdicts on security-class artifacts at FLAG-FOR-HUMAN, and realigns to the simplest adequate solution behind an explicit per-item human gate. The audit is read-only and emits a diffable findings artifact; realignment is a separate, explicitly invoked skill; and a third read-only lane re-runs the audit on whatever cadence the consumer wires and reports only what moved since the last run, above a configurable noise budget. A justification lane applies the same method to whatever single artifact you point at, a decision record, a document, a component, a dependency, or a code construct, asking whether a reason existed for it and whether that reason still holds, and reporting how much evidence each verdict actually rests on. - [`improvement`](../plugins/improvement): Evidence-first, cross-dimension improvement finder. Point it at a repo, feature, concept, or process surface and it produces a ranked, evidence-cited list of improvement candidates led by value-to-effort, interviews on the pick, and hands off to the planning pipeline; runnable unattended as a tech-debt-sweep routine. - [`attribution`](../plugins/attribution): Finds prose in tracked markdown that restates content an external source owns (vendor docs, blogs, articles) without adequate attribution, confirms the source, and refactors the copy into a pointer, a citation, or a dated stamped record. Documentation attribution, not software supply chain. Nomination and judgment are LLM work; the scripts do only reasoning-free work (corpus scoping, breadcrumb extraction, stamp expiry, fingerprint compare of two concrete texts). Read-only audit by default; explicit fix and sweep actions apply dispositions behind a semantic-diff guard and live pointer verification. Findings conform to the detector-findings convention. -- [`provenance`](../plugins/provenance): Deprecated: this plugin was renamed to attribution. Install attribution@melodic-software instead; this entry only points its skills at /attribution:audit and /attribution:setup, does no work, and is removed in a later release. ## Maintenance diff --git a/docs/hook-migration-audit.md b/docs/hook-migration-audit.md deleted file mode 100644 index 732b5d9ab7..0000000000 --- a/docs/hook-migration-audit.md +++ /dev/null @@ -1,156 +0,0 @@ -# General-purpose hook migration audit - -Point-in-time audit of the **general-purpose** subset of `melodic-software/medley`'s in-repo hooks -for extraction into this marketplace's hook plugins (`guardrails`, `harness-ops`). This is an **audit -snapshot**, not durable policy. The [migration playbook](migration-playbook.md) is the policy; this -table records each candidate's gate compliance on the audit date and which follow-up issue owns each -accepted migration. Empirical claims decay: a row is only true as of the stamp below. - -Audited 2026-07-12 (`melodic-software/medley#1391`, under wave-2 map `melodic-software/medley#1369`). -Facts are Tier-0, read from each hook's `.sh`, its `.test.sh`, and medley's `.claude/settings.json` -registration this session. The shipped-standard column is measured against the published hook-plugin -conventions: the [four-seam extensibility contract](migration-playbook.md), the -[hook-telemetry envelope contract](conventions/hook-telemetry/README.md), and the -[shared-`hook-utils.sh` decision record](migration-playbook.md). - -## Scope - -Medley wires ~38 hooks via `${CLAUDE_PROJECT_DIR}/.claude/hooks/`. This audit grades **only the -general-purpose subset** the wave-2 map nominated: guardrail hooks (target: `guardrails`) and -telemetry/observability hooks (target: `harness-ops`). The remaining hooks are **out of scope by -nature**: the .NET-toolchain hooks (`block-dotnet-test-nologo`, `msbuild-introspect`, -`nuget-pack-prep`, `publicapi-diff`, `sarif-diagnostics`, `dependency-*`) and the worktree/branch -hooks (`branch-awareness`, `branch-protection`, `git-safety`, `worktree-*`, `onboard-drift`, -`single-test`) encode this repo's toolchain and workflow and stay repo-specific. - -## Gate dimensions - -Each candidate is graded against the shipped hook-plugin standards, the HARD gates a hook must -clear to ship repo-agnostic: - -- **Seam-clean**: zero surviving repo-path coupling under plugin cache isolation. Every candidate - `source`s a sibling `hook-utils.sh` (bundled at cutover per the shared-lib record, not a defect); - the gate is whether the hook's *behavior* de-couples, or whether it embeds a `${CLAUDE_PROJECT_DIR}` - path, a `tools/` shell-out, a `.work/`-artifact convention, or medley-specific injected content that - survives generalization. -- **Kill-switch**: a `HOOK__ENABLED` env gate (via `hook::check_enabled`), the ecosystem norm. -- **Telemetry seam**: a producer emits the generic envelope to `HOOK_TELEMETRY_SINK` (opt-in, - no-op when unset), **not** a direct write to an assumed repo observability store. -- **Shared lib**: sources `hook-utils.sh`, so it rides the `lib/hook-utils.sh` SSOT + - `scripts/sync-hook-utils.sh` sync at cutover. -- **Contract test**: ships a black-box `.test.sh` asserting the stdin→exit/stdout contract. -- **Target + verdict**: the destination plugin and the accept / defer decision. - -**Seam reconciliation (telemetry).** The wave-map nomination reads "telemetry hooks need a sink/dir -`userConfig` seam." The shipped contract resolves that intent differently and correctly: a producer -emits the envelope to the `HOOK_TELEMETRY_SINK` **env** target (literal in `settings.json`, per the -[envelope contract](conventions/hook-telemetry/README.md) "Sink path resolution"), **not** a -per-hook `userConfig`, which would diverge from every existing producer (`markdown-format`, -`secret-pattern-detection`, …). The nomination's real requirement, "the repo OTEL store must not be -assumed", is met by **stopping the direct store-write and emitting to the consumer's sink**. A -`userConfig` `directory` seam applies in exactly one place: `skill-usage-audit`'s bespoke second -store (`skill-usage.jsonl`), which does not flow through the envelope. - -## Verdict summary - -- **Accepted for migration: 9 of 13.** Two guardrail hooks (`block-hook-bypass`, - `workflow-resilience-check`) and the seven-hook `*-audit` telemetry-emitter family. Two retrofit - issues filed, one per accepted **group**, per the wave-map emitter protocol. -- **Deferred / repo-owned: 4.** `pr-prep-evidence-check`, `hook-telemetry-sink`, - `cc-telemetry-ensure`, `session-reinjection`. Each stays in medley with an explicit revisit - trigger (below). None is a clean generalization; each de-couples into a *different* parameterized - tool or is consumer-owned infrastructure by design. -- **Systemic gap surfaced: the generic sink.** Once the `*-audit` producers ship in `harness-ops` - emitting envelopes, they are inert without a consumer sink, and `harness-ops` ships none (medley's - `hook-telemetry-sink` is repo-owned by design). Recorded below. - -## Guardrails candidates (3) - -| Hook | Seam-clean | Kill-switch | Telemetry seam | Shared lib | Contract test | Target + verdict | -|---|---|---|---|---|---|---| -| `block-hook-bypass` | yes, behavior is generic (blocks Bash file-write workarounds such as `cat >`, `echo >`, and `python3 -c` that circumvent Write/Edit gates) | `HOOK_BLOCK_HOOK_BYPASS_ENABLED` | **rewire**: direct `hook::record_event` write to `.claude/observability/hook-events.jsonl`; migrate to the envelope | yes | yes (block/allow, false-positive regressions) | **guardrails → ACCEPT** | -| `workflow-resilience-check` | after one edit: advisory nudge when a `Workflow` script fans out un-throttled; genericize the hardcoded `.claude/rules/dynamic-workflows.md` cite in the emitted text | `HOOK_WORKFLOW_RESILIENCE_CHECK_ENABLED` | n/a (no telemetry; `additionalContext` only) | yes | yes (9 cases: when it speaks / stays silent) | **guardrails → ACCEPT** | -| `pr-prep-evidence-check` | **no**, the concept is bound to medley workflow: shells out to `tools/work-artifacts/derive-slug.sh`, globs `.work//review/*-pr-prep.md`, reads a `prepared_at_sha` frontmatter contract, cites medley skill paths | `HOOK_PR_PREP_EVIDENCE_CHECK_ENABLED` | n/a | yes | yes (26 cases, real git fixtures) | **DEFER, repo-owned** (see below) | - -Both accepts cover **distinct** surfaces from the shipped `block-no-verify` (git-hook *disabling* via -`--no-verify`/`core.hooksPath`/`LEFTHOOK=`): `block-hook-bypass` guards Bash *file-write* bypass of -Write/Edit gates; `workflow-resilience-check` is a `Workflow`-tool burst-resilience advisory. Zero -coverage overlap. - -## harness-ops candidates (10) - -| Hook | Seam-clean | Kill-switch | Telemetry seam | Shared lib | Contract test | Target + verdict | -|---|---|---|---|---|---|---| -| `api-error-audit` | yes | `HOOK_API_ERROR_AUDIT_ENABLED` | **rewire** to envelope (today: direct store-write) | yes | yes | **harness-ops → ACCEPT** | -| `config-change-audit` | yes | `HOOK_CONFIG_CHANGE_AUDIT_ENABLED` | **rewire** to envelope | yes | yes | **harness-ops → ACCEPT** | -| `instructions-loaded-audit` | yes (carries an extra `…_LOG_SESSION_START` knob) | `HOOK_INSTRUCTIONS_LOADED_AUDIT_ENABLED` | **rewire** to envelope | yes | yes | **harness-ops → ACCEPT** | -| `permission-denied-audit` | yes (privacy-safe `Bash:` subject) | `HOOK_PERMISSION_DENIED_AUDIT_ENABLED` | **rewire** to envelope (`status=blocked`) | yes | yes | **harness-ops → ACCEPT** | -| `pre-compact-audit` | yes | `HOOK_PRE_COMPACT_AUDIT_ENABLED` | **rewire** to envelope | yes | yes | **harness-ops → ACCEPT** | -| `tool-failure-audit` | yes (twin of permission-denied; privacy-safe subject) | `HOOK_TOOL_FAILURE_AUDIT_ENABLED` | **rewire** to envelope (`status=error`) | yes | yes | **harness-ops → ACCEPT** | -| `skill-usage-audit` | **outlier**: writes a bespoke second store `${repo}/.claude/observability/skill-usage.jsonl` via inline `flock`, in addition to the shared JSONL | `HOOK_SKILL_USAGE_AUDIT_ENABLED` | **rewire** to envelope **+ a `directory` `userConfig` seam** for the second store | yes | yes | **harness-ops → ACCEPT** (extra seam) | -| `hook-telemetry-sink` | n/a: this **is** the consumer sink (`HOOK_TELEMETRY_SINK` target), the envelope→JSONL adapter | ABSENT (governed by master `HOOK_OBSERVABILITY_LOG_ENABLED`) | n/a (terminus, not producer) | yes | yes | **DEFER, consumer-owned by design** | -| `cc-telemetry-ensure` | **no**: hardcodes `tools/observability/start-collector.sh`/`start-dashboard.sh`, DuckDB view names, Aspire ports/URL, medley slash-commands | `HOOK_CC_TELEMETRY_ENSURE_ENABLED` | n/a | yes | yes | **DEFER, repo-owned** | -| `session-reinjection` | **no**: payload is 100% medley content (rule paths, `PLAT001-PLAT015`, `Result`, `BannedSymbols.txt`); not telemetry (only an incidental completion event) | `HOOK_SESSION_REINJECTION_ENABLED` | n/a | yes | yes | **DEFER, repo-owned** | - -The seven `*-audit` hooks are one cohesive bulk-pattern unit: thin async advisory emitters over the -same `hook::emit_timed_event` path, all seam-clean at the behavior level, all `HOOK__ENABLED` -gated, all black-box tested. They share **one** migration seam, stopping the direct -`.claude/observability/hook-events.jsonl` write and emitting the envelope, plus per-hook `data` schemas -under `conventions/hook-telemetry/data/`. `skill-usage-audit` carries the lone extra seam. - -## Deferred / repo-owned surfaces: decision record (2026-07-12) - -Each deferred surface stays in `melodic-software/medley` with an explicit revisit trigger, so the -deferral is a decision, not a silent omission. The discriminator is **concept-specificity, not -path-count**: every candidate has repo paths today (all `source` a sibling `hook-utils.sh`); the -accepts have generic concepts that de-couple to seam-clean, while these de-couple into a *different* -parameterized tool or are consumer-owned by design. - -- **`pr-prep-evidence-check`** (guardrails-nominated): its concept is medley PR-workflow enforcement - bound to the `.work//` artifact convention, `derive-slug.sh`, and the `prepared_at_sha` - frontmatter contract. De-coupling produces a workflow-specific tool, not a universal guard. - **Revisit trigger:** a second repo adopts the `.work/`-prep-evidence-before-PR convention → extract - a generic prep-gate whose slug derivation, artifact glob, and freshness field are declared config. -- **`hook-telemetry-sink`**: the consumer sink the [envelope contract](conventions/hook-telemetry/README.md) - "Mediator boundary" and the playbook's [Reintegration](migration-playbook.md) step, which keeps the - sink script as the bridge, both say stays consumer-owned. It maps the envelope into medley's own store; it is - not a producer to migrate. **Revisit trigger:** see the generic-sink gap below. -- **`cc-telemetry-ensure`**: medley OTEL-pipeline enablement, bound to `tools/observability/*` - collector/dashboard scripts, DuckDB view names, and Aspire ports. `harness-ops` already owns - collector-lifecycle scripts on the *read* side. **Revisit trigger:** `harness-ops` grows a - SessionStart ensure-hook that drives **its own** bundled collector scripts through a store/collector - `userConfig` seam. -- **`session-reinjection`**: post-compaction context reinjection whose entire payload is - medley-specific prose. A generic version is a `userConfig`-templated content feature, not a - migration. **Revisit trigger:** a second repo wants post-compact reinjection → build a - content-templated hook reading a tracked file list, not this hook's baked content. - -## Systemic gap: the generic sink - -Migrating the `*-audit` producers to `harness-ops` completes only the **producer** half of the -telemetry contract. A fresh `harness-ops` consumer that enables the audit hooks emits envelopes into -the void: `harness-ops` ships no sink, and medley's `hook-telemetry-sink` is repo-owned by design. -This cuts against the playbook's "drop into any repo and work" intent. Keeping medley's sink -repo-owned is correct (per the mediator boundary); the gap is the **absence of a generic reference -sink**. Two candidate resolutions, to settle when the `*-audit` retrofit is scheduled: - -1. `harness-ops` ships a reference sink (envelope→`${project}/.claude/observability/hook-events.jsonl`, - the shape its observability skill already reads) that a consumer wires via `HOOK_TELEMETRY_SINK`. -2. The `*-audit` retrofit issue documents the sink-wiring requirement as a consumer setup step, and - the sink stays consumer-authored. - -Recommendation (1): a reference sink closes the loop and is the observability skill's natural -counterpart. The `*-audit` retrofit issue carries this decision; it also coordinates with the -`harness-ops` setup-action retrofit (`melodic-software/medley#1432`). - -## Net-new retrofit issues emitted - -One `retrofit()` issue per accepted **group**, sub-issue-linked under wave-2 map -`melodic-software/medley#1369`, `agent-ready`. These are **retrofit** issues (adding hooks to an -existing plugin), not cutover issues. No medley in-repo hook is removed here; the blue-green cutover -of each in-repo original follows on its own once the plugin hook ships and is verified. - -| Group | Scope | Issue | -|---|---|---| -| guardrails hooks | Add `block-hook-bypass` + `workflow-resilience-check` (two independent, atomic PRs): bundle `hook-utils.sh`, de-couple per the table, rewire `block-hook-bypass` telemetry to the envelope, ship `.test.sh` | `melodic-software/medley#1445` | -| harness-ops `*-audit` family | Migrate the seven-hook emitter family as one bulk unit: rewire the direct store-write to the `HOOK_TELEMETRY_SINK` envelope, add per-hook `data` schemas, add `skill-usage-audit`'s second-store `directory` seam, settle the generic-sink gap | `melodic-software/medley#1446` | diff --git a/docs/migration-playbook.md b/docs/migration-playbook.md index 43d76e68a3..5cd2f34b39 100644 --- a/docs/migration-playbook.md +++ b/docs/migration-playbook.md @@ -507,16 +507,10 @@ by a version bump and a changelog note. An install that still names an old id ge `Plugin "" not found in marketplace`, and the consumer re-enables the plugin under its new name. Plugin splits and file moves are not renames. -A rename whose tracker item scopes it may also keep the old id for one release as a deprecation -shim. The shim is a real catalog entry whose skills are `disable-model-invocation: true` stubs that -point at the successor. It keeps an existing install from reporting -`Plugin "" not found in marketplace`, since upstream has no -deprecation state of its own -([host-marketplace, "Rename or remove a plugin"](https://code.claude.com/docs/en/plugins/host-marketplace#rename-or-remove-a-plugin), -checked 2026-09-27; recheck when that page gains a deprecation field). The next release removes -the shim like any retirement. `provenance` → `attribution` (#4589) is the first. Consumers outside -this repository (the fleet list, dotfiles, user-scope `enabledPlugins`) migrate from their own -repositories. +A rename or retirement migrates every consumer in this repository in the same change, with no +deprecation shim, alias, or pointer to the old name: no stub catalog entry, no redirecting skill, +no second spelling a consumer can keep using. Consumers outside this repository (the fleet list, +dotfiles, user-scope `enabledPlugins`) migrate from their own repositories. ### Same-version commit drift (directory-source marketplaces) diff --git a/docs/setup-contract-campaign-follow-ups.md b/docs/setup-contract-campaign-follow-ups.md deleted file mode 100644 index 0c0303c712..0000000000 --- a/docs/setup-contract-campaign-follow-ups.md +++ /dev/null @@ -1,41 +0,0 @@ -# Setup-contract campaign follow-ups (#3138) - -The four follow-ups from the #3111 / #3112 / #3113 / #3127 campaign, bundled in -[#3138](https://github.com/melodic-software/claude-code-plugins/issues/3138), and how each is -settled. - -**Claim:** `${CLAUDE_PLUGIN_ROOT}` expands in a skill body but stays literal in a `context/` file -read via `Read`, and the Bash tool's environment does not carry it. The `worktree` skill therefore -resolves the scripts directory in `SKILL.md` and its `context/` files use ``. The -Claude Code pin follows the Dependabot policy in `.github/dependabot.yml`. The two worktree suites -skip on Windows Git Bash hosts. `scripts/check-drive-root-litter.sh` stays a host-wide advisory -scan. - -**Basis:** the `worktree` `SKILL.md` "Scripts directory (resolved)" line holds the measured probe -(two headless `claude -p` runs on Claude Code 2.1.284); `.github/dependabot.yml` lines 49-51 (the -daily schedule for the executable compatibility dependency) and 53-59 (the cooldown, with -`@anthropic-ai/claude-code` excluded); `native_mktemp_dir` in -`plugins/source-control/scripts/test-helpers.sh`, which the fixtures of -`worktree-root-doctor.test.sh` and `worktree-add-containment-gate.test.sh` use with no host gate; the -"ADVISORY BY DEFAULT" header of `scripts/check-drive-root-litter.sh`. - -**As of:** 2026-09-30, Claude Code 2.1.284. - -**Recheck:** a Claude Code release note that changes plugin-variable substitution or the Bash tool -environment; a change to the Dependabot Claude Code entry; a Windows Git Bash run of the two -worktree suites that fails; a maintainer wiring `check-drive-root-litter.sh` into a required live lane. - -## Positions - -| Follow-up | Position | -| --- | --- | -| 1. `CLAUDE_PLUGIN_ROOT` liveness | Fixed at the call sites. A `context/` file is read as raw bytes, so the token reaches Bash literal and the command exits 127. `worktree/SKILL.md` carries the resolved scripts directory, and `context/status.md`, `audit.md`, `cleanup.md` and `create.md` call helpers through ``, to be substituted before a command reaches Bash. | -| 2. Toolchain pin vs measured CLI | `.github/dependabot.yml` lines 49-51 keep the pin on a daily schedule, and lines 53-59 exclude `@anthropic-ai/claude-code` from the cooldown so bumps arrive immediately. No separate lag policy. | -| 3. Windows worktree test failures | Both suites run on Windows Git Bash. Fixtures create their temp dir with `native_mktemp_dir`, which returns the `cygpath -m` form native git stores and can follow in an `includeIf` path; the scripts need no change. | -| 4. `check-drive-root-litter.sh` machine-state sensitivity | A host-wide advisory scan of drive roots, not a repo-scoped gate: the script header says "ADVISORY BY DEFAULT" and [windows-path-emit](conventions/windows-path-emit/README.md) "The detection net" describes the host fingerprint it detects. Non-Windows is a reported no-op. An operator with a deliberate `C:\tmp` sets `DRIVE_ROOT_LITTER_IGNORE_SINKS=tmp`. | - -## What this close is not - -- Not a fleet sweep of interpolating call sites beyond the `worktree` `context/` files. -- Not a `package.json` bump or a change to the Dependabot policy. -- Not promoting `check-drive-root-litter.sh` into a required live lane. diff --git a/docs/upstream/claude-code-mods/research-2026-09-19/research-repo-fit.md b/docs/upstream/claude-code-mods/research-2026-09-19/research-repo-fit.md index f35fc729aa..9d12a064d5 100644 --- a/docs/upstream/claude-code-mods/research-2026-09-19/research-repo-fit.md +++ b/docs/upstream/claude-code-mods/research-2026-09-19/research-repo-fit.md @@ -65,8 +65,8 @@ per cohesive unit, with `marketplace.json` and `docs/catalog.md` conflicts resolved by **serializing the final merges, not authorship**. `OBSERVED` · HIGH. The one fleet-scale hook migration this repo has run -(`docs/hook-migration-audit.md`, an audit **snapshot** dated 2026-07-12, not -durable policy) sets three precedents a mods sweep inherits: the accept/defer +([`docs/hook-migration-audit.md`](https://github.com/melodic-software/claude-code-plugins/blob/6736cf547e168575a14b865ecb5394e87fc28f4f/docs/hook-migration-audit.md), +an audit **snapshot** dated 2026-07-12, not durable policy, since deleted) sets three precedents a mods sweep inherits: the accept/defer discriminator is **concept-specificity, not path-count**; every deferral carries an explicit revisit trigger; and cutover is **blue-green and staged**; no in-repo original was removed in the PR that added its replacement. diff --git a/plugins/attribution/.claude-plugin/plugin.json b/plugins/attribution/.claude-plugin/plugin.json index 12ceb29481..7f248566ba 100644 --- a/plugins/attribution/.claude-plugin/plugin.json +++ b/plugins/attribution/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "attribution", - "version": "0.9.0", + "version": "0.9.1", "description": "Finds prose in tracked markdown that restates content an external source owns (vendor docs, blogs, articles) without adequate attribution, confirms the source, and refactors the copy into a pointer, a citation, or a dated stamped record. Documentation attribution, not software supply chain. Nomination and judgment are LLM work; the scripts do only reasoning-free work (corpus scoping, breadcrumb extraction, stamp expiry, fingerprint compare of two concrete texts). Read-only audit by default; explicit fix and sweep actions apply dispositions behind a semantic-diff guard and live pointer verification. Findings conform to the detector-findings convention.", "author": { "name": "Melodic Software", diff --git a/plugins/attribution/CHANGELOG.md b/plugins/attribution/CHANGELOG.md index cea018f47e..2a5f679607 100644 --- a/plugins/attribution/CHANGELOG.md +++ b/plugins/attribution/CHANGELOG.md @@ -1,5 +1,16 @@ # Changelog +## [0.9.1] - 2026-10-02 + +### Removed + +- **BREAKING:** the `provenance` plugin is gone from the marketplace. An install that still enables + `provenance@melodic-software` gets `Plugin "provenance" not found in marketplace`; enable + `attribution@melodic-software` instead. +- The detector scripts no longer warn about a leftover `provenance.json` or + `provenance.local.json` in a config layer; such a file is silently ignored. `/attribution:setup` + still reports and migrates the two repo-level files through its retirement records. + ## [0.9.0] - 2026-10-02 ### Changed diff --git a/plugins/attribution/README.md b/plugins/attribution/README.md index a2571d36cb..b1942be593 100644 --- a/plugins/attribution/README.md +++ b/plugins/attribution/README.md @@ -122,12 +122,6 @@ stamp forms; enable it where that holds. Every detector script accepts `--show-config` and names the layer each value came from. -A legacy `provenance.json` or `provenance.local.json` left in any layer is never read. When a -layer has the legacy file and no `attribution` file, each script that reads the config cascade -prints one warning naming it and the `attribution` file name to rename it to. -`/attribution:setup check` reports the two repo-level legacy files as retired conventions, and -`apply` migrates their keys and removes them once you confirm. - ## Prerequisites - **bash** for `list-corpus.sh`, `extract-breadcrumbs.sh`, `check-stamps.sh`, diff --git a/plugins/attribution/skills/audit/scripts/lib.sh b/plugins/attribution/skills/audit/scripts/lib.sh index 42bea2f905..e11908752f 100644 --- a/plugins/attribution/skills/audit/scripts/lib.sh +++ b/plugins/attribution/skills/audit/scripts/lib.sh @@ -45,19 +45,14 @@ cfg_layers_init() { } # cfg_layer_add : append / when present and not already a -# layer (two paths naming one file are read once). A layer holding only the -# legacy provenance file name is never read; it draws one warning. +# layer (two paths naming one file are read once). cfg_layer_add() { - local dir="$1" name="$2" legacy="$1/provenance${2#attribution}" layer - if [[ -f "$dir/$name" ]]; then - for layer in ${CFG_LAYERS[@]+"${CFG_LAYERS[@]}"}; do - config_root_paths_same "$layer" "$dir/$name" && return 0 - done - CFG_LAYERS+=("$dir/$name") - elif [[ -f "$legacy" ]]; then - echo "warning: legacy config $legacy is not read; rename it to $dir/$name" >&2 - fi - return 0 + local dir="$1" name="$2" layer + [[ -f "$dir/$name" ]] || return 0 + for layer in ${CFG_LAYERS[@]+"${CFG_LAYERS[@]}"}; do + config_root_paths_same "$layer" "$dir/$name" && return 0 + done + CFG_LAYERS+=("$dir/$name") } # cfg_layers_print: the layer listing every --show-config output opens with. diff --git a/plugins/attribution/skills/audit/scripts/list-corpus.test.sh b/plugins/attribution/skills/audit/scripts/list-corpus.test.sh index 5643ce90fa..a4f889ee52 100755 --- a/plugins/attribution/skills/audit/scripts/list-corpus.test.sh +++ b/plugins/attribution/skills/audit/scripts/list-corpus.test.sh @@ -228,40 +228,6 @@ assert_not_contains "a non-repo root: its .claude file is not a team layer" \ "$NONREPO_SC" "$NONREPO/.claude/attribution.json" rm -f "$HOME/.claude/attribution.json" -# --- Legacy config name ---------------------------------------------------------- -# A layer holding only the former plugin's file name is never read, and draws one -# warning naming that file and the name to rename it to. - -legacy_run() { - LEG_ERR="$(cd "$REPO" && CLAUDE_PROJECT_DIR="$1" bash "$LIST_CORPUS" 2>&1 >/dev/null)" - LEG_FILES="$(cd "$REPO" && CLAUDE_PROJECT_DIR="$1" bash "$LIST_CORPUS" 2>/dev/null | jq -r '.files[]')" -} - -printf '%s\n' '{"excluded_paths":["legacy/**"]}' >"$CFG_DIR/provenance.json" -legacy_run "$REPO" -assert_contains "a legacy team config draws a warning naming it" "$LEG_ERR" "$CFG_DIR/provenance.json" -assert_contains "the team warning names the new file name" "$LEG_ERR" "attribution.json" -assert_eq "the team warning is one line" "$(printf '%s\n' "$LEG_ERR" | grep -c 'provenance')" "1" -assert_contains "the legacy team config's keys are not applied" "$LEG_FILES" "legacy/old.md" - -printf '%s\n' '{"excluded_paths":[]}' >"$CFG_DIR/attribution.json" -legacy_run "$REPO" -assert_not_contains "no warning when the new name sits beside the legacy one" "$LEG_ERR" "provenance.json" -rm -f "$CFG_DIR/provenance.json" "$CFG_DIR/attribution.json" - -printf '%s\n' '{"excluded_paths":["docs/**"]}' >"$CFG_DIR/provenance.local.json" -legacy_run "$REPO" -assert_contains "a legacy local overlay draws a warning naming it" "$LEG_ERR" "$CFG_DIR/provenance.local.json" -assert_contains "the overlay warning names the new file name" "$LEG_ERR" "attribution.local.json" -assert_contains "the legacy overlay's keys are not applied" "$LEG_FILES" "docs/guide.md" -rm -f "$CFG_DIR/provenance.local.json" - -printf '%s\n' '{"excluded_paths":["README.md"]}' >"$HOME/.claude/provenance.json" -legacy_run "$TEST_TMPDIR/noconfig" -assert_contains "a legacy user-global config draws a warning naming it" "$LEG_ERR" "$HOME/.claude/provenance.json" -assert_contains "the legacy user-global config's keys are not applied" "$LEG_FILES" "README.md" -rm -f "$HOME/.claude/provenance.json" - # --- --show-config --------------------------------------------------------------- OUT_SC="$(run_default --show-config 2>&1)" diff --git a/plugins/attribution/skills/setup/SKILL.md b/plugins/attribution/skills/setup/SKILL.md index e2129a67ae..bf5e77d74c 100644 --- a/plugins/attribution/skills/setup/SKILL.md +++ b/plugins/attribution/skills/setup/SKILL.md @@ -1,5 +1,5 @@ --- -description: "Set up and maintain this repository's attribution audit configuration: `.claude/attribution.json` across the config cascade's three layers. Manages the categorical exclusions (including the eval-fixture tree, which is a config entry by design and never a rule in a script), the per-candidate and corpus fetch budgets, the separation-rule constants, the stamp expiry window, the accuracy dials for nomination passes and judge sampling, and the fix-eligibility gates. Enables the off-by-default trigger-less-stamp check for a repository whose stamp forms are uniform enough to greppably support it. Use when: 'set up attribution', 'configure attribution', 'exclude a path from the attribution audit', 'change the stamp expiry window', 'the attribution audit flags too much', 'turn on the trigger-less stamp check', after installing the plugin, or to migrate a leftover pre-rename `.claude/provenance.json`. Writes only the consuming repository's own config, never source." +description: "Set up and maintain this repository's attribution audit configuration: `.claude/attribution.json` across the config cascade's three layers. Manages the categorical exclusions (including the eval-fixture tree, which is a config entry by design and never a rule in a script), the per-candidate and corpus fetch budgets, the separation-rule constants, the stamp expiry window, the accuracy dials for nomination passes and judge sampling, and the fix-eligibility gates. Enables the off-by-default trigger-less-stamp check for a repository whose stamp forms are uniform enough to greppably support it. Use when: 'set up attribution', 'configure attribution', 'exclude a path from the attribution audit', 'change the stamp expiry window', 'the attribution audit flags too much', 'turn on the trigger-less stamp check', or after installing the plugin. Writes only the consuming repository's own config, never source." argument-hint: "[check|apply]" user-invocable: true disable-model-invocation: true @@ -52,12 +52,9 @@ Exit 0 → PASS. Exit 1 → one finding per TSV row: `migrate` is FAIL, `delete` `report-only` INFO; remediation is `apply`. Exit 2 → FAIL, never silent. Bash unavailable → report the step UNKNOWN with remediation, never green. -In this plugin's manifest that yields `attribution-r001` FAIL while `.claude/provenance.json` -persists and `attribution-r002` FAIL while `.claude/provenance.local.json` persists. Both files are -from before the rename, and the detectors never read them, so every value in them is silently not -applied. The user-global `~/.claude/provenance.json` sits outside the repository and has no -record; the detectors warn about it on every run, and `check` reports it as WARN with the same -remediation: move it to `~/.claude/attribution.json`. +In this plugin's manifest each `migrate` record (`attribution-r001`, `attribution-r002`) is a +retired config file the detectors never read, so every value in it is silently not applied until +it is migrated. `apply` cleans up after writing any agreed keys. It re-runs detection and handles each finding with its own confirmation. It carries the file's keys into the successor the record names; the old @@ -189,7 +186,7 @@ wording the check does not recognize, the same conclusion follows. Leave it off ## What this skill does NOT do - **Does not edit source.** The only files it writes are the consuming repository's own config, - and the only files it removes are the retired pre-rename config files, after migration. + and the only files it removes are retired config files its manifest names, after migration. - **Does not add per-instance suppressions.** There is no per-finding keep in this schema by design; a passage-level exception is the operator's, through the finding-suppression convention. diff --git a/plugins/provenance/.claude-plugin/plugin.json b/plugins/provenance/.claude-plugin/plugin.json deleted file mode 100644 index 0ac84bb06c..0000000000 --- a/plugins/provenance/.claude-plugin/plugin.json +++ /dev/null @@ -1,16 +0,0 @@ -{ - "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", - "name": "provenance", - "version": "0.6.2", - "description": "Deprecated: this plugin was renamed to attribution. Install attribution@melodic-software instead; this entry only points its skills at /attribution:audit and /attribution:setup, does no work, and is removed in a later release.", - "author": { - "name": "Melodic Software", - "email": "info@melodicsoftware.com" - }, - "license": "MIT", - "keywords": [ - "deprecated", - "provenance", - "attribution" - ] -} diff --git a/plugins/provenance/CHANGELOG.md b/plugins/provenance/CHANGELOG.md deleted file mode 100644 index 80f8738b35..0000000000 --- a/plugins/provenance/CHANGELOG.md +++ /dev/null @@ -1,1287 +0,0 @@ -# Changelog - -## [0.6.2] - 2026-10-02 - -### Fixed - -- The `setup` `argument-hint` uses Claude Code's official bracket notation: it leads with its check - action. - -## [0.6.1] - 2026-09-28 - -### Changed - -- Version serializes with `attribution` 0.6.1; the shim still points at `/attribution:audit`. - -## [0.6.0] - 2026-09-27 - -### Deprecated - -- **This plugin is now `attribution`.** Every skill, script, and reference moved to - `attribution@melodic-software`: `/provenance:audit` is `/attribution:audit` and - `/provenance:setup` is `/attribution:setup`. This version is a deprecation shim: its two skills - only tell you where the real skill went and do no work. Install `attribution@melodic-software`, - remove `provenance@melodic-software` from `enabledPlugins`, and rename `.claude/provenance.json` - to `.claude/attribution.json`. The shim is removed in a later release. - -## [0.5.16] - 2026-09-27 - -### Fixed - -- `audit`'s "Does not scan" line routes doc-vs-code drift to the `review` plugin's `doc-drift-detector` agent (no skill by that name exists) and AI-writing style to `ai-slop:audit` instead of the bare `ai-slop` (#4119). - -## [0.5.15] - 2026-09-25 - -### Changed - -- Comment-only pass with /code-tidying:dissolve-comments: restating comments, history narration and ticket back-references removed from scripts and tests, over-budget rationale shortened. Every edit is certified comment-only by a token-level proof, so behavior is unchanged; the removed text is recorded in the commit bodies. - -## [0.5.14] - 2026-09-21 - -### Changed - -- American spellings throughout this plugin's prose, ahead of the `en-us` locale the - shared typos config adopts. Wording only: no behavior, option, default, or identifier - changes. Released sections were corrected in place on the same terms. - -## [0.5.13] - -### Changed - -- The five provenance audit scripts source one lib.sh for require_opt_value, json_str and the config-layer cascade instead of carrying private copies. Diagnostics, JSON output and exit codes are unchanged. - -## [0.5.12] - -### Changed - -- The provenance audit scripts split test case specs with a single read instead of parameter-expansion chains, inline a single-use jq helper in emit-findings, and drop dead default guards in list-corpus and score-golden, with byte-identical findings and scores. - -## [0.5.11] - -### Changed - -- Cite the marketplace `docs/` doctrine files by their lower-kebab names (`docs/plugin-philosophy.md`, `docs/migration-playbook.md`, and siblings); the files were renamed and the old uppercase paths no longer resolve. - -## [0.5.10] - -### Changed - -- **audit:** the two citations of the `upstream-drift` convention's fetch-route section now quote - its current heading, "Reading the basis: the fetch route". That heading lost its em dash in the - marketplace repository, so the quoted wording in `reference/source-fetch.md` no longer matched - the section it names. Wording of the citation only; the claim, its basis, the `As of:` date, and - the recheck trigger are unchanged. - -## [0.5.9] - -### Fixed - -- `skills/audit/CLAUDE.md` shim added beside the skill's contributor `AGENTS.md`, so the conventions load when Claude reads the skill (Claude Code reads `CLAUDE.md`, not `AGENTS.md`). - -## [0.5.8] - -### Changed - -- **audit:** the neutral tier is published as `not-found` everywhere; `source-not-identified` stays readable as its former name (prompt-audit follow-up F20). -- **setup:** the inert `shell: bash` frontmatter key is dropped, since no injection remains in the file (prompt-audit follow-up F12). -- **audit:** `list-corpus.sh` reads a target's repo-relative position from `git rev-parse --show-prefix` instead of subtracting the corpus root as a string, so a directory, file, `.` or `--paths-file` target no longer reports an empty corpus on a host where one directory has several absolute spellings (prompt-audit follow-up F10). -- **audit:** `list-corpus.sh` keeps a corpus path git cannot place as it was written, so a `--paths-file` entry under a missing directory declines instead of collapsing onto a same-named file at the repository root, and a nested checkout no longer reads as the whole corpus (prompt-audit follow-up F10). - -## [0.5.7] - -### Fixed - -- **`audit`:** the skill published one spelling of the neutral tier and described another. - `SKILL.md`'s frontmatter listed the tier as `source-not-identified`; the rubric, the - dispositions and source-fetch references, `context/persist-findings.md`, the plugin README and - the script's own prose all say `not-found`. `SKILL.md` now says `not-found` too, so one name is - published. `emit-findings.sh` still recognizes both spellings, deliberately and permanently: a - sidecar is model-authored against whatever description was in context, and recognizing one name - too many can only withhold a record, while one too few walks a judgment verdict onto a relay - row. The comments beside the two predicates and the tests that exercise both spellings now say - the tolerance covers a retired name rather than a currently published one. -- **`audit`:** the comment introducing the stamp-rule relay exception in `emit-findings.sh` - claimed "ONE exception" where the code has two. `withheld_verdict` is evaluated before any rule - id is read, so `{"rule": "rule-stamp-expired", "verdict": "not-found"}` is withheld on the - judgment-verdict check and never reaches the unreadable-`tier` predicate the comment described. - The comment now names both conditions and states which runs first; - `context/persist-findings.md` gets the same correction, since it framed the same claim as a - single exception. No behavior changed: the classification chain is untouched, and the case the - comment miscounted was already correct, withholding a record that declared a judgment verdict. - `emit-findings.test.sh` gains a case pinning it, asserting zero relay rows, the withheld count - rather than the not-relay-eligible count, and no payload leak. - -### Changed - -- **`audit`:** `context/persist-findings.md` states the normalization mechanism before the motive - it approximates, rather than framing rendering-equivalence as a goal and then carving out the - homoglyph exception. The limit is unchanged and still stated. - -## [0.5.6] - -### Changed - -- **`audit`:** `reference/rubric.md` drops its version-history preamble and the two in-line - version notes, keeping the invalidation rule; the editor rule about measurement results moves to - a new `skills/audit/AGENTS.md`, which also takes the `affected-tests.sh` exit-3 note from - `context/gotchas.md`. `context/persist-findings.md` states its eight relay-boundary rules in the - present tense instead of as past bug reports. `reference/nomination.md` drops its rubric-version - pins, states the judge and review case block once, and trims the reviewer rationale to the one - fact the Judgment section does not carry. `reference/source-fetch.md` and - `reference/dispositions.md` drop the research-phase incident, the marketplace-specific generator - path, and the golden-set starting count. `context/gotchas.md` drops its preamble and the - 8 percent figure. -- **`setup`:** removed the editor-addressed sentence after the fixture-exclusion rule. -- Applied from the 2026-09 prompt-audit against Claude Fable 5.1 - (docs/specs/prompt-audit-skills-2026-09.md). - -## [0.5.5] - -### Fixed - -- **`audit`, `setup`:** the git pre-compute lines moved out of `## Pre-computed context` into a - "Repository context. Gather first" body section of individual Bash calls, one command per call, - each `head` bound kept inside its command and a failure read as an unknown value. The harness - composes a skill's whole pre-compute block into one shell invocation, and a worktree-isolated - session refuses a git-bearing compound command, which blocked these skills from loading inside a - worktree. Same shape as the worktree skill's fix in #1619. Non-git pre-compute lines stay where - they were. setup tests for the team config under the literal root the previous call returned. - -## [0.5.4] - -### Changed - -- **`fingerprint.mjs`: `shingleAt`'s doc block now states why one spelling has to serve both - readers.** The set `shingles()` builds is exactly what `matchedSpans()` looks a local shingle - up in, so any divergence between the two constructions would move containment and the reported - spans together and silently. The parameter is renamed `i` to match both call sites. Comment and - parameter name only: fingerprint output is byte-identical, verified across both call sites and - over a 400-file corpus at eight window sizes. -- **`list-corpus.sh`: `cfg_array` drops a single-use binding**, assigning the layer's joined - value straight to `out`. Last-writer-wins across config layers is unchanged, and so is the - `!= null` test that separates "key defined as empty" from "key absent". -- Note for maintainers: no test pins the fingerprint algorithm. Changing the shingle join - separator or the window bound leaves the suite 40 of 40 green while altering every fingerprint - this plugin has ever emitted. A frozen input-to-fingerprint assertion would be worth adding. - -## [0.5.3] - -### Fixed - -- **`emit-findings.test.sh` counted a host skip as a pass.** One case routed a - skip through `pass()`, contradicting the rule stated thirty lines above it in - the same file: a skip never routes through `pass()`, so a proof the host could - not run can never be read off the summary as one that did. The suite's honest - count on a host where `chmod a-w` does not bite is 384 passes and 1 skip, not - 385 passes. The real assertion still runs wherever the subject can be built. -- **`score-golden.sh` no longer scores an uncovered case as covered.** A - simplification during this sweep replaced an iterating membership test with - jq's `index`, which does SUBSTRING search when handed a string. Since - `cases_run` comes from a model-authored sidecar validated only as parseable - JSON, a string there is reachable: with `"cases_run": "c1-long"`, the script - scored case `c1` as covered and exited 0, where it had previously aborted with - a type error. Now uses `any(. == $id)`, which reads as well and still refuses - to guess. Caught by the group's refutation verifier before landing. - -### Changed - -- **Audit-script tidyings.** `fingerprint.mjs` extracts two expressions each - duplicated across two functions and drops a guard the surviving union check - already covers, returning literal 0 for two empty sets rather than NaN; - `extract-breadcrumbs.sh` drops a write-only awk global, safe because both forms - read the match position at the identical program point; a helper defined but - never called is removed; and sixteen history-narration comments become - present-tense hazard statements, keeping every measurement and the point that - agreement between two copies of one rule is blind to a defect they share. - -## [0.5.2] - -### Changed - -- **`audit`: the two absolute-path cases report a host skip instead of failing on a third root - spelling.** `emit-findings.sh` relativizes Location against two spellings of the repository root, - git's toplevel and `cd`-then-`pwd` of it. On a Git Bash host `mktemp -d` answers a third, a mount - alias git never reports, so the fixture's finding path matches neither anchor and the row is - declined as outside the root. That is the producer working as designed on a path it was never - handed, not a defect the cases can prove anything about. - - The suite now probes for that third spelling directly: where either anchor names the fixture - root, the case runs; where neither does, it prints a visible `SKIP (host: ...)` line counted - apart from the pass total. The producer is unchanged. - -## [0.5.1] - -### Fixed - -- **Three statements in 0.4.0 are no longer true, and this entry supersedes them rather than - editing them.** That entry deferred a fixture leak as "deliberately NOT fixed here", recorded - four case bodies stating their own answer as a limit that could not be worked around without - editing a fixture, and said the deterministic layer "reproduced every containment, jaccard and - matched-span figure the fixtures record". All three are now overtaken: the leaks are removed and - two jaccard figures moved as a direct result. Dated records are not rewritten here, so 0.4.0 - stands as what was true when it was written and this entry carries what changed. - -- **The shared source page for `c08`, `c09` and `c10` named the cases and gave away the answer.** - It opened by stating it was the shared basis for the three synonym-rotation cases and that - holding it fixed made rotation density the single variable. Judges read `source.md`, so that - paragraph settled C1 and C2 before either was graded, and nine of the thirty judges in the - version-3 re-score read it. - - **The removal and the re-measurement are one change, because the first moves the second.** The - fingerprint compares the case body against `source.md`, so deleting from the source changes the - arithmetic those three cases record, and shipping the deletion alone would leave a measurement - that no longer reproduced from its own fixtures. Containment, longest matched span and the - matched-span set are unchanged, because the removed paragraph shares no five-word shingle with - any case body; the union shrank while the intersection did not, so jaccard rises: `c08` 0.232 to - 0.280, `c09` 0.146 to 0.178, `c10` unchanged at zero. Each case still classifies as its - `expected.json` says, `c09` still on the containment limb alone and `c10` on neither, so the - density ladder still measures density. - -- **Four case bodies stated their own answer, in the graded passage itself.** `c06` and `c07` - opened "A hard negative", `c08` called the passage below it "the copied passage", and `c10` - supplied its own C1 and C2 findings outright along with a tier hint. 0.4.0 recorded these as - accepted because withholding them "would mean editing a fixture", an objection the fixture edit - above overtakes. They are removed under the same fix-and-re-measure discipline, and the affected - `notes.measured` figures are corrected in the same commit: `c06` containment 0.031 to 0.039, - `c08` 0.473 to 0.570 with jaccard 0.312. - - Two `span` line ranges moved with them. That is a re-anchor, not a verdict change: deleting - leading lines shifts every line number below, the shift matches the deleted count exactly, and - the spanned text is byte-identical before and after. No `class` or `tier` field was touched. - -- **A weaker date signal could delete a stamp instead of reinforcing it.** `may_form()` reports two - signals for the month May, a digit beside the word and a capital M on the original line, - through a single `RSTART` that all three call sites read to decide whether the match began inside - the keyword window. It returned on whichever branch matched first, so a digit-adjacent "may" out - in the window's slack handed back an out-of-window offset, the caller rejected it, and a capital - "May" sitting inside the window was never consulted: appending a stray "7 may" to "Verified this - May" removed the line from the candidates entirely. Both signals are now evaluated and the - leftmost returned, in `check-stamps.sh` and `extract-breadcrumbs.sh` alike, so a signal the - window would reject can no longer hide one it would accept. Corpus output is unchanged over - 1,352 files: 530 candidates, 499 parsed, 31 declined, 0 findings. - - Swapping the branch order was rejected as a fix because it mirrors the bug rather than removing - it. The cross-implementation agreement assertion is what let this survive: it compared the two - scripts' candidate sets and **passed while both were wrong**, agreeing on one where the answer - was two. Agreement tests are blind to a shared error by construction, and the assertion now pins - the count rather than only the agreement. - -- **The relay boundary leaked through the `## Unparsed` appendix.** A judgment verdict - (`source-fetched-similar`, `llm-suspected`, `not-found`) carrying no rule id matched no branch in - the projection and was dumped verbatim into the findings file, tier name and payload included, - into the one file those verdicts are withheld from, and the apply relay's input. Withholding is now - decided on the **declared** tier ahead of any rule lookup, read from a fixed key allowlist and - matched exactly against the three verdict names. The record is still counted in `## Surfaces`, so - nothing is dropped. - - **Reaching that took ten rounds, and for the first nine each fix opened the next hole.** Recorded - in full because the sequence, not any one defect, is the finding. - - 1. The first version matched the tier exactly at the top level. An adversarial probe defeated it - four ways: a padded `" not-found "`, an array-valued `["not-found"]`, an object-valued - `{"name":"llm-suspected"}`, and a capitalized `"Tier"` key. - 2. Widening it to any key named `tier` at any depth closed those and **silently dropped - relay-eligible findings**: a `fingerprint-confirmed` copy carrying an unrelated nested tier, - `"review":{"tier":"one agent argued llm-suspected and was vetoed"}`, a note `SKILL.md` invites, - was withheld, reaching neither the relay table nor `## Unparsed`, while `## Surfaces` called - it a judgment finding that stays on a human report it was never on. Seven vectors. - 3. Narrowing to a key allowlist fixed the drop and **relayed a judgment verdict**: the allowlist - read `verdict.tier` but not `verdict` itself, so `{"verdict":"not-found"}` on a stamp rule - reached the relay table. - 4. Reading a whole `verdict` closed that and re-introduced the drop from a different direction. A - `verdict` holds the judges' output while the tier is mapped by fixed rule from the evidence, - per `SKILL.md` step 9, "never from a judge's confidence", so a confirmed copy beside - `"verdict":{"prior":"llm-suspected"}` was withheld again, and one shape refused the whole - sidecar. The same round trimmed invisible characters by enumerating two code points, leaving - six other `Cf` characters to walk a verdict onto a relay row; an unhandled one at the end even - neutralized a handled one at the start. - 5. The declared `tier` now wins whenever the record has one, falling back to the `verdict` only - when it does not, and the fallback turns on the slot **naming** a known tier rather than the - key merely being present, which is what `{"tier":null}`, `{"tier":[]}` and `{"tier":"pending"}` - beside a verdict had been slipping through. - 6. The same defect one container down: the `verdict` → `verdict.tier` step still keyed off the - child being present, so `{"verdict":{"tier":"pending","result":"not-found"}}` declared nothing - and printed its outcome verbatim. Both steps now share one definition, the first fix in the - sequence to address the class rather than an instance. - 7. Format characters were stripped only at the ends of a value, so one sitting *inside* the name - failed the exact match and relayed: a word joiner placed mid-word in `not-found`. Stripped - everywhere now. - 8. Stripping was by an enumerated class, which missed a variation selector and a combining - grapheme joiner. It now strips by the Unicode property that defines rendering as nothing. - 9. Three routes at once. A verdict spelled as a **key** (`{"tier":{"not-found":true}}`) was read - by neither reader, because both walked string values only. A hyphen homoglyph (U+2010) - rendered identically to the name it spelled and relayed, so hyphen-likes now fold to ASCII by - the dash class. And `Location` was the one cell carrying input that was never pipe-escaped, so - a path like `a|b.md` split the row and a consumer read the Finding cell as a Surface. - 10. The `searched` key was read literally while `tier` and `verdict` were case-folded, so a - sidecar that **did** name its surfaces under `Searched` was refused whole, taking every - relay-eligible finding beside it, the one direction that gate has no excuse for failing in. - And the stamp rules relayed on any tier at all, which falsified round 9's own safety argument - for the homoglyph limit: a Cyrillic-`о` spelling took a relay row instead of the ordinary - path. A stamp rule now still relays whatever a record does or does not declare, except when - its own `tier` field names no tier this reader knows. - - **Every one of these passed review before it was probed.** Across the rounds a bot reviewer, five - security passes and three code-review passes read this file; all of them characterized the - failure direction as over-withholding and therefore safe. Over-withholding was destroying - relay-eligible findings, and the boundary was leaking in two separate rounds. Reading a diff and - attacking an invariant are different activities, and only the second found any of this. The suite - went from 110 assertions to 307, the gap being almost entirely the direction nobody was testing: - that a legitimate finding still **survives**. - - Two limits are stated rather than papered over. A record that is not an object has no declared - tier to read, so it is withheld when a verdict name appears anywhere inside it, the blast radius - the malformed-record route exists to avoid. And `source-not-identified`, the neutral tier name - `SKILL.md` publishes, is not one of the three the reader knows. - - `context/persist-findings.md` required both that withheld tier names never appear and that an - unmappable finding lands in `## Unparsed` verbatim, never a silent drop, without saying how the - two coexist, a conflict landing precisely on the leaking record. It now states the ordering and - names the `## Surfaces` count as where the no-silent-drop guarantee is discharged, so the next - reader does not restore the leak as a bug fix. - -- **The relabeling that kept fixture paths away from judges was a habit, not a rule.** 0.4.0 records - that the version-3 re-score relabelled cases "so no directory name or path reached a judge". - Nothing in the plugin required it: a grep across `SKILL.md`, every `reference/*.md`, - `evals/evals.json` and every script found exactly one mention of relabeling in the whole plugin, - in that changelog entry. The golden directories are named for their own answers, as - `c06-negative-quoted-and-cited` and `c08-adversarial-rotation-sparse` show, so a judge handed a - path reads the class, the carve-out and the rotation density before opening the file. - - `reference/nomination.md`, which constructs all three subagent prompts, now carries "Neutral - labels (required)": a case reaches any subagent the run dispatches over it, whether nominating, - judging, reviewing or guarding a fix, under an opaque label, and the run holds the label-to-path - mapping. - `SKILL.md` and `reference/dispositions.md` reference the rule rather than restating it. - - Two channels beyond the directory name are closed with it. `SOURCE TEXT` said "fetched bytes, - with its URL and the rung it came from"; under the vendored-snapshot route and in the golden set - the source is served from a local file, so that field could hand over an in-repo path. It now - carries the source's declared URL and route, never the local path. And every golden `source.md` - opens by naming the golden set and calling the page invented for these fixtures, the answer - arriving in the body text once the path was shut, so that paragraph is dropped from the copy a - subagent is handed. `fingerprint.mjs` is not a subagent and reads the file as committed, so no - containment, jaccard or span figure moves. - - The directories are not renamed: the names carry meaning for the humans maintaining the set, and - renaming would churn the 30 paths `evals.json` enumerates for no gain over fixing the dispatch. - Four verifier rounds went into this, three of which failed. The third caught that a fix had - quietly narrowed the judge prompt to four of the rubric's six carve-outs, which is a grading - change this work was not allowed to make. It was reverted. The requirement remains unmeasured: - no `evals.json` expectation asserts that a run relabelled before dispatch. - -### Added - -- **The `not-found` searched-surfaces listing is schema-checked at the sidecar.** - `emit-findings.sh` refuses a report sidecar whose `not-found` finding names no surface at all, on - exit 3, the input-refusal code it already uses for a sidecar that is not audit output. The check - validates the sidecar rather than an emitted row, and it has to: the relay boundary withholds - every `not-found` finding from the findings file, so there is no row to check. - - Presence is all it can assert. Nothing knows which surfaces a run actually visited, so a listing - omitting one it checked is still indistinguishable from a complete one, and a report's listing - remains the run's own claim rather than validation evidence that no source exists. - -### Changed - -- **Sweep resume semantics are stated where a run will read them.** The closure ledger is now named - by its path, `.work//sweep-ledger.md`, rather than left to a repo-local spec no - plugin file pointed at, and the three facts the sweep Brief requires and the shipped docs - contradicted are recorded: `corpus_fetch_ceiling` is spent across the whole sweep and a resume - restores its spend rather than restarting at zero; the response cache is per-sweep and a resume - re-validates an entry rather than reusing a body nobody in this sweep read; the ledger is - checkout-local, so a sweep resumed elsewhere is a new sweep and reports itself as one. - `reference/dispositions.md` carries the entry's required fields. - - **No ledger machinery exists, and the prose says so at every mention.** Nothing in this plugin - creates, reads, or validates that file. These are rules for how a run conducts itself, and they - hold only as far as the run keeps the ledger honestly. - -### Method, and what it does not support - -- **The version-3 re-score reported here was not a blind panel, and is not offered as one.** The - worker that ran it had no subagent tool, so it graded sequentially inline, one pass per case - rather than three. `c01` through `c07` were graded without their `expected.json` ever being - opened. `c08`, `c09` and `c10` were graded *after* it, because checking the classification - against the answer key required reading it; those three verdicts are worth less than the other - seven and are marked contaminated rather than averaged in silently. - - The result reproduces the recorded table, **8 tp / 0 fp / 0 fn / 2 tn, precision 1.00, recall - 1.00, no verdict moved**, and no class becomes fix-eligible, every one still below - `min_n_per_class` 10 at n = 2, 5, 1, 2. The arithmetic beside it was re-derived independently and - holds; it is the *method* claim that is narrower than 0.4.0's. - - `c09` is the control worth noting: it carries no self-describing line, and it graded identically - to the two that did. - -- **This entry now lands above 0.5.0, and every panel figure it reports is pinned to rubric - version 3.** 0.5.0 moved the rubric to version 4, narrowed carve-out 5, and invalidated the - version-3 golden-set measurement. The `8 tp / 0 fp / 0 fn / 2 tn, precision 1.00, recall 1.00` - table restated above, and the "no class becomes fix-eligible" conclusion drawn beside it, are - therefore superseded by 0.5.0 and are not offered as version-4 claims. This work was authored - against version 3 and is recorded as it was measured rather than rewritten. - - **What the rubric change does not touch is the deterministic arithmetic.** Containment, jaccard, - longest matched span, the matched-span sets and the span re-anchors are computed by - `fingerprint.mjs` over the committed bytes; they read no rubric and no carve-out, so every - recomputed figure in the entries above stands exactly as recorded. So do the fixture edits - themselves: an answer key removed from a case body or a `source.md` is a leak closed under any - rubric version. The golden set still awaits its version-4 re-score, per 0.5.0. - -## [0.5.0] - -### Changed - -- **Rubric version 4: carve-out 5 (distilled-product architectures) gains a span-level qualifier, - and the version-3 golden-set measurement is invalidated.** Version 3 asked only whether the - surface's product is a distillation. That question is satisfiable by almost any reference file - that credits a source, and because carve-outs are graded before the criteria and stop grading, a - broad reading silently absorbs the C2 and C4 failures the criteria exist to catch. Version 4 - keeps the purpose test and adds the boundary a distilling file draws for itself: **a verbatim or - near-verbatim span the file's own attribution does not enumerate is not covered**, and - reformatting (un-fencing a prompt block into prose, tabulating a source's prose) is not - distillation. - - **The evidence, recorded here rather than in the rubric** (the rubric is inlined into every judge - prompt, so a measurement written there is read by every judge before it grades). In a repo-wide - run over 1292 tracked files, carve-out 5 drew **110 of roughly 230 carve-out citations across 138 - panels**, more than double the next carve-out. An adversarial review pass over the unanimous - clears returned three challenges, two of which attacked carve-out 5 specifically and both on the - rubric rather than on the file: one on `plugins/playbooks/reference/model-adaptation/opus-5.md`, - whose own Sources section enumerates which spans are verbatim while the matched block appears on - none of those lists; one on `plugins/mcp-tools/skills/audit/reference/checklist.md`, whose - opening line declares a pointer discipline ("do not recap them here, read them at the source"), - the opposite of a distillation product. The enumerating-file test in version 4 is the first of - those arguments generalized, because it is the one a judge can apply to a span. - - **Consequences.** The golden set must be re-scored against version 4 before any precision figure - is cited against it, and no class becomes fix-eligible on a measurement pinned to version 3. The - version-3 re-score recorded below is superseded. Expect the change to move findings in one - direction only: spans inside distilling files that were previously declined before grading now - reach the criteria, where C2 and C4 decide them on their merits. - - This also matters beyond the copy lane. A `restated-upstream-fact` detector of the kind proposed - in #3525 would inherit this carve-out, and restated facts live disproportionately in exactly the - distilling files version 3 declined wholesale, so the broad reading would have suppressed the new - lane before it shipped. - -## [0.4.2] - -### Changed - -- **`audit`: `check-stamps.sh` collapses a duplicate branch in `from_label()`.** The `--*` arm and - the fallback arm of the three-way conditional printed the character-identical `(from %s)` string, - so the split carried no behavior. The two arms are now one. Verified over an executed 25-case - input matrix (flag forms, config-layer paths, format-string hazards, whitespace, multi-arg and - no-arg calls): old and new outputs are byte-identical everywhere, and the 71-case suite passes - unchanged. - -## [0.4.1] - -### Fixed - -- **`audit`: the two config probes broke on any install path containing a space.** Both expansions - were unquoted, so an install under a path with a space in it (a Windows profile directory named - ``, a macOS `Application Support` tree) word-split into two arguments, the script was - never found, and the guard short-circuited to `detector unavailable` on scripts that are present - and working. Reproduced by copying a detector into a directory whose name contains a space: - unquoted renders `detector unavailable`, quoted renders the config, and on a space-free path both - render it. Both the guard run and the data run are quoted now, matching the form - `firecrawl:update` already ships. - - Quoting changes the literal command string, and Bash permission rules are globs over that literal - string, so the existing `Bash(${CLAUDE_SKILL_DIR}/scripts/list-corpus.sh:*)` and - `Bash(${CLAUDE_SKILL_DIR}/scripts/check-stamps.sh:*)` rules no longer match the quoted - invocations. A companion quoted rule is added for each. The unquoted rules are kept, because the - audit flow's own steps 1 and 3 still invoke both scripts unquoted. Each pair names one script - under the same `${CLAUDE_SKILL_DIR}` anchor with the same `:*` argument scope, so this authorizes - nothing the plugin could not already run. - -### Changed - -- **Both config probes adopt the pipefail-proof filtered-probe shape, on shape rather than on an - observed failure.** `list-corpus.sh --show-config` emits 7 lines against a `head -10` cap, so - `head` never closes the pipe early, and `check-stamps.sh --show-config` is piped into `tail -3`, - which drains its input and cannot raise SIGPIPE at all. Both were verified by execution in three - states under both `pipefail` settings and neither showed any difference: **these were latent by - shape, not live defects, and nothing observable is fixed here.** The change is that a later - `--show-config` growing past ten lines would silently start asserting `detector unavailable` - under a correct detector, which is exactly what happened to `ai-slop:audit`. Confirmed by - substituting a 500-line detector: the old shape renders 10 lines plus the token under `pipefail`, - the new one renders 10 lines. The filter pipelines now sit in a brace group closed by `:`, the - shape `docs-hygiene` 0.21.23 and `code-tidying` 0.14.13 established. - -## [0.4.0] - -### Changed - -- **The version-3 re-score is done, and the measurement carries forward unchanged.** Version 3's - own rule blocked every precision figure, and with it every class's fix eligibility, on a table - pinned to version 2. All ten golden cases were re-judged by a three-judge panel each, thirty - judges in independent processes, cases relabelled so no directory name or path reached a judge. - Each saw the candidate passage, the fetched source, the whole containing file and the rubric, - per the version-3 dispatch; none saw the case's `expected.json`, the fingerprint figures, or - another judge's verdict. The deterministic layer was re-run alongside and reproduced every - containment, jaccard and matched-span figure the fixtures record. - - Result: **8 tp / 0 fp / 0 fn / 2 tn, precision 1.00, recall 1.00**, the table version 2 - recorded, now pinned to version 3. Every panel unanimous, **no verdict moved.** `c04` is the - only case whose attribution reaches grading, so it is the only one C3's stated scope could have - moved, and all three judges took the new test where version 2 left it: the derivation is one - lift inside otherwise-original material, so a `See also` bullet two sections below understates - its scope and C3 passes. - - **No class becomes fix-eligible, for the reason that was already there.** Every class measures - 1.00 against the 0.95 bar and every class sits below `min_n_per_class` 10 (n = 2, 5, 1, 2). The - re-score lifts the rubric-version block and leaves the class-size one standing, which is what - keeps the sweep in #3465 report-only. - -### Fixed - -- **The rubric carried its own answer key, and every judge read it.** Version 3's version-history - paragraph recorded the expected tally, the panel size, and an enumeration of which golden case - turns on which criterion, including the one case the scope change exists to restate. The - pipeline inlines the whole rubric into every judge prompt at the judgment step, so all thirty - judges in the re-score above read the prediction before grading, and that run had to withdraw - its claim of a blind panel. Found by that run's own fresh-context verifier, which returned FAIL - on the method while confirming the arithmetic. - - The paragraph was written to keep the figures from sitting under a cloud they did not deserve. - Putting it in the file judges read at judgment time is what made a blind measurement against - that rubric impossible. The prediction and the enumeration are changelog material and now live - here; the rubric keeps the criteria, the carve-outs, the scope rule, the worked examples and the - tier table, and says explicitly that a judge should be able to read all of it and still not know - the answer. - - **The verdict was tested against the leak rather than assumed safe.** `c04` was re-judged by a - second three-judge panel against the same rubric with the version-history preamble removed and - every criterion, carve-out, scope sentence and worked example intact. All three returned STANDS - with C3 PASS on the same scope-mismatch reasoning. The leak did not drive the verdict; the claim - that a fully blind panel produced it is still withdrawn. - - **A second leak of the same kind sits in a fixture and is deliberately NOT fixed here.** The - `source.md` shared by `c08`, `c09` and `c10` announces itself as "the shared basis for the three - adversarial synonym-rotation cases", naming them and asserting the local text is a rotation, - which pre-answers C1 and C2 for nine of the thirty judges. That line is inside the text the - fingerprint module compares, so removing it moves every containment and matched-span figure - those three cases record. **The fix and a re-score are one atomic change**, and splitting them - would leave a recorded measurement that no longer reproduces from its own fixtures. It is filed - for the round that next re-scores rather than taken now. - - Two smaller limits, recorded rather than worked around. Four case bodies state their own intended - answer (`c06` and `c07` open "A hard negative", `c08` and `c10` open "Adversarial case"), and - version 3 requires the judge to read the whole containing file, so those judges saw it; - withholding it would mean editing a fixture. And `c07`'s `expected.json` explains the case as a - C1 failure while also recording that the owned-content carve-out applies, which the rubric's own - order of evaluation makes exclusive. All three judges declined it at the carve-out, which is what - that order requires. The route differs, the recorded answer does not, and neither the fixture nor - the rubric was changed to match the run. - -### Added - -- **The evidence-tier contract now covers a vendored-snapshot basis.** Every tier row gated on - either a fetched source or no source at all, and the sweep hit a third case the table could not - express: a finding compared against an in-repo copy of upstream, carrying a declared upstream ref - and a sync date, reached because every live fetch rung failed. Strong provenance, weak currency. - It happened at `plugins/playwright/skills/playwright/reference/test-generation.md:80`, where both - candidate upstream URLs returned 404 and only the committed baseline remained. - - Such a finding now caps at `source-fetched-similar`, records `source.route: vendored-snapshot` - together with each live fetch that failed and how, and is **never fix-eligible**. The reason is - the plugin's whole subject: fix eligibility rests on current upstream state, and a snapshot - cannot establish it. Stale evidence licenses no edit. The tier borrow is declared deliberate - rather than left to read as accurate, since `source-fetched-similar` is worded for a source that - was fetched and this one was not; the recorded route is what keeps the report honest about the - difference. The follow-up is human: re-run the candidate when upstream is reachable or the - snapshot re-syncs, rather than holding the finding open. - -- **The modal "may" is no longer read as a month name.** `may` is a month and an ordinary English - modal verb, and both stamp detectors matched it bare, so prose like "the first read may raise a - permission prompt" became a stamp candidate whose date could not be parsed and landed in the - declined bucket, indistinguishable to a reader adjudicating that bucket from a real stamp the - parser failed on. **19 of the 24 month-name declines carried the word**, measured over 1,352 - files at `--as-of 2026-08-28` on this branch's head. - - That figure is tree-dependent and three different numbers for it appeared during this change, - which is worth recording rather than tidying away. An earlier draft said 17 of 22, measured - correctly against a branch base that was two commits behind `main` and therefore missing merged - changelog entries whose own prose contains the modal; repairing the base moved it to 19 of 24. - The script comments said 13, which matched no tree. Both are corrected. The lesson is the same - one 0.3.2 records about this file: a count taken over a corpus that includes this repository is - a reading at a commit, not a constant, and it needs the commit attached or it will not - reproduce. - - `may` counts as a date when a digit sits beside it, since every date form has one and the modal - does not, **or** when the original line capitalizes it. The other eleven months still match bare, - because over-reporting into a bucket a human reads is the safe direction and this fix must not - trade it for under-reporting. - - **The first version of this fix did trade it, and three independent reviewers caught that.** - Requiring a digit made a digitless stamp vanish: `Verified this May` and `Checked last May - against the vendor page` stopped matching anything, and the loss was *upstream* of the declined - bucket rather than inside it: `keyword_window()` returned empty, the caller dropped the line - before classification, and `is_stamp()` did the same to the inventory. Not declined, not - inventoried, gone. `Verified in June` was still declined and still visible, so the same shape got - two different treatments purely because of the modal collision. - - The capital `M` is what fixes it. Both scripts lowercase before matching and so discard the one - signal that separates these in edited prose. The rule now lives in a named `may_form()` carried - at all three sites rather than three copied regexes. - - **The case signal was measured on this corpus, not assumed.** At `3c538bcc`, over 1,352 files: - 1,458 lines carry a lowercase `may`, overwhelmingly the modal; 24 carry a capital `May`, of which - **14 are month dates and 10 are capitalized modals** in table cells, bullets and sentence - openings; and 34 carry an ALL-CAPS `MAY`, of which **none is a date**. They are permission - modals. So two costs are accepted knowingly. - A capitalized modal opening a sentence or a cell now reads as a month when a stamp keyword sits - in its window, which over-reports into a bucket a human adjudicates. And ALL-CAPS defeats case, so - a digitless `MAY` date stays invisible; buying it back would mean reading `MAY` as a month, which - on this corpus means 34 false candidates for a form nobody writes. Both are pinned by tests, - including one named for the over-report so it is not rediscovered as a bug. - - **A third cost is recorded and left rather than chased.** `may_form()` returns on the digit - branch, so `RSTART` belongs to that match; when a digit-adjacent `may` sits beyond the window and - a capital `May` sits inside it, the caller rejects the out-of-window digit match and never - consults the in-window capital. Appending a stray `7 may` to an otherwise valid line therefore - removes its candidacy, which is under-reporting, the direction this fix exists to prevent. No - corpus line has that shape. Returning the leftmost of the two matches would fix it; trying the - capital branch first only mirrors the bug, so the obvious one-line swap is not a fix. Both - scripts inherit it - identically, so their cross-script agreement assertion is blind to it, exactly as it was to the - original regression. Recorded at the rule so the next reader is warned rather than surprised. - - **The suites could not have caught this, which is the part worth keeping.** They assert that both - scripts return the same count over a shared fixture, an assertion that passes when both are - equally wrong, which is exactly what happened. A cross-implementation agreement test detects - divergence and is blind to a common error, and a shared definition is what makes a common error - likely. The new cases pin a **non-zero** expected count in both suites, so agreement is now - backed by a known answer rather than by two implementations nodding at each other. - - Corpus effect of the correction, **at `3c538bcc`**: none. Candidates, parsed, declined and - findings all unchanged at 527 / 499 / 28 / 0, the two JSON products byte-identical, because the - corpus carried no digitless-May stamp at that commit. The regression was latent there and real in - principle. - - **It stopped being latent one commit later, and this entry is why.** The paragraph above quotes - `Verified this May` as an example of the shape, inside a `verified` keyword window, in a file the - corpus scans. So from the commit that documents the fix onward the corpus does carry a - digitless-May stamp, the one written to explain that it carried none. Measured at `a827aa58`: - 529 / 499 / 30 / 0 post-fix against 528 / 499 / 29 / 0 pre-fix, an effect of +1 candidate rather than - none. - - That is the fourth time on this branch that prose about a detector has moved what the detector - reports, after the stale Phase 6 baseline, the figures that went stale as the entry describing - them was written, and a count that changed when the branch base was repaired. The rule this file - keeps relearning is the one it already states: **a count over a corpus that includes this - repository is a reading at a commit, and it needs the commit attached or it will not reproduce.** - Every figure in this entry now carries one. - - Three sites changed, not two: both detectors and **the classifier in `check-stamps.sh`**, which - a single-site fix would have missed and which decides the decline reason a human then reads. - `check-stamps.sh` and `extract-breadcrumbs.sh` change together and their suites now assert the - agreement over a shared fixture, because a previous fix in this area landed in one script and - needed a follow-up commit to reach its sibling. - - Over 1,352 files: declines 45 to 28, month-name declines 22 to 5, 17 lines removed and none - added. **`parsed` is unchanged at 499 and `findings` unchanged at 0.** Those two are the numbers - the conclusion rests on: they say no real stamp was reclassified in either direction and none - had been masked. A real `May 2026` stamp is still detected in both month-first and day-first - forms. - - Two adjacent false positives are deliberately left in place and recorded rather than fixed: - `SC2034` read as a bare year, and `read` matching inside `cache_read_input_tokens`. Both have a - different root cause (token boundaries, not the month list), and both plausible fixes push toward - under-reporting: a year-boundary fix shifts `RSTART` into the window machinery two prior commits - tuned, and excluding `_` from the keyword boundary would stop matching a real stamp that exists - today: `plugins/work-items/skills/track/actions/add.md:104` carries `"last_checked": - "2026-04-08"`, which parses now and would be lost. That exhibit is named because the first draft - of this entry cited an invented one; the verifier checked the repo, found no such token, and - supplied the real line. The conclusion held, the evidence for it did not. - - The bare-year bucket needs its own designed pass rather than a boundary tweak bolted onto a - word-sense fix. Of its 23 declines, a token-boundary change repairs exactly one: the rest carry - years that are already token-boundaried, among them an issue number `#1941` and a glob example - `photos [2024/**`. - -- **The `not-found` searched-surfaces listing is recorded as prose-only and unenforced.** Three - separate requirements say a run must list every surface it searched before concluding no source - exists. Nothing checks that: `emit-findings.sh` projects no `searched` array and no budget block, - and the relay boundary withholds `not-found` findings from the file entirely, so a listing that - omits a rung is indistinguishable downstream from a complete one. Rather than leave three - requirements reading as though something enforced them, the docs now say a run's listing is that - run's own claim and never validation evidence that no source exists, and name what an - implementing change would need. This repo's house style prefers a recorded limitation to an - asserted capability. - -- **The fix contract warns that a corpus file can be a generator's rendered output.** The sweep - found one whose authoritative home is `docs/native-surfaces/records.json`, outside the markdown - corpus, with its rendering marked never-hand-edit. A fix applied to the rendering would edit a - file its own header forbids editing, and the next regeneration would overwrite it. Dispositions - now say to check the file head for a generated-output marker and route to the human naming the - generator's input as the real fix site. No script enforces this check. - -## [0.3.2] - -### Fixed - -- **The window fix landed in one of the two scripts that share the definition.** - `extract-breadcrumbs.sh:is_stamp` still sliced the keyword window at exactly its length after - 0.3.1 fixed `check-stamps.sh:keyword_window`, so the two disagreed about what a stamp candidate - is while the audit flow passes the extractor's output to nomination. Measured across every file - the 0.3.1 fix newly parsed, **five** were short a stamp line the extractor should have - inventoried: `docs/CLOUD-SESSIONS.md` (3 against 4), - `docs/conventions/loop-lane/README.md` (4 against 5), - `docs/topics/fresh-eyes-checkpoint-audit/design/design-resolution.md` (1 against 2), - `plugins/context-guard/CHANGELOG.md` (4 against 5), and - `plugins/session-flow/CHANGELOG.md` (7 against 8). All five agree now. - - An earlier draft of this entry said three, because it sampled five of the seven affected files - and reported the differences it happened to catch as the total. The number here comes from - sweeping all seven against both versions of the extractor. - - `docs/CLOUD-SESSIONS.md:320` is the worked case, and it is worse than the 0.3.1 one rather than - a repeat of it. Its date begins at offset 60 of the 60-character window, so the cut left a bare - `2` and **no** form matched, not even the bare-year fallback that at least kept the 0.3.1 case - visible in the declined bucket. The line did not decline; it left the inventory entirely, which - is the quieter failure of the two. - - The same slack-and-start-boundary rule now applies in both: slice `wlen + 9`, require every - match to begin at or before `wlen`. A regression test pins the real corpus line, with a negative - control at offset 64 confirming the added slack does not admit a date that starts outside the - window. - - One property of `is_stamp` is worth recording, because it masked this and will mask the next - attempt to reproduce it: the function rescans from each keyword in turn, so a line carrying a - second keyword beside its date (an `as-of` immediately before it, say) matches there regardless - of what the first window truncated. A fixture written to exercise the boundary must carry - exactly one keyword, or it passes against the unfixed script and proves nothing. Two fixtures - written the other way did exactly that here, and a third placed the date by eye rather than by - measurement; only lifting a real corpus line verbatim produced a genuine red. - - That sentence is also why this paragraph names no literal date. An earlier draft of it quoted - one as an example of the shape, and the corpus run then reported this changelog as carrying an - expired stamp, one day over. Prose *about* stamp syntax is indistinguishable from a stamp to a - mechanical detector, and this file is inside the corpus it documents. - -- **The review dispatch could not execute rubric v3 either.** 0.3.0 gave the judge the containing - file and left the reviewer, which runs when `accuracy.review_agents > 0`, holding only the - passage, the source text, and the quoted grades. Its job includes checking the C3 grade and - whether a carve-out was missed; C3 is graded across the file and carve-outs 1, 4 and 5 are - file-level. A reviewer without the file either declines the check or waves through an - unsupported C3 PASS, and review is the last stage before fix eligibility, so waving one through - is what puts an unsupported finding in reach of an automatic edit. The review prompt now carries - `LOCAL FILE:` on the same terms as the judge prompt. - -## [0.3.1] - -### Fixed - -- **A conforming ISO stamp was declined as a bare year, and a declined stamp is never - expiry-checked.** `check-stamps.sh` sliced the keyword window at exactly its length (60 - characters, 30 after `read`), which cut through any date that started inside the window but ended - past it. At `docs/upstream/aihero-course.md:127` the window ended mid-date: `as-of 2026-08-17` - was read as `2026-08-`, the ISO test failed on the fragment, and the bare-year fallback then - matched the `2026` it left behind. Declining routes the line into the reported `declined` bucket - and skips it, so the one thing the script exists to do, compare the date against the currency - window, never ran on a date that parses perfectly well, and the output said only that the stamp - date went unparsed. - - The window is now a distance from the keyword rather than a cut through the text. The slice - carries nine more characters, one short of the longest form the tests match, and every form must - begin at or before the window length, so the added slack lets a date finish without admitting one - that starts outside the window. - - Corpus effect at `--as-of 2026-08-28` over 1,352 tracked files, stated as the delta because that - is the part that stays true: **+7 candidates, +7 parsed, declined unchanged** (month-name and - bare-year trading 2, as the two below flip), **expiry findings unchanged at 0**. Absolutes - measured at `fb11cf6a` are 538 to 545 candidates and 493 to 500 parsed, against 45 declined. - - **Those absolutes will not reproduce at another commit, and the reason is worth more than the - numbers.** This changelog is inside the corpus it measures, so each paragraph added here creates - new stamp candidates and moves the totals. The figures first published in this entry were taken - before the entry itself was written and were already stale by the time it shipped: a smaller, - quieter instance of exactly the staleness 0.2.1 was written to correct. The delta is the durable - claim; an absolute needs the commit it was taken at, and even then only holds there. - - Two of the seven newly parsed stamps had been declined as bare years - (`docs/upstream/aihero-course.md:127`, - `plugins/context-guard/reference/cloud-headless-capture.md:78`); the other five were not detected - as candidates at all, because truncation left nothing date-shaped in the window. None of the seven - is expired. The oldest is 40 days, and the oldest parsed stamp anywhere in the corpus is 142 days - against a 180-day window, so no lapsed stamp had been hidden by this. - - Two new declines appear, both instances of the separate `may` false positive, where the month-name - test reads the ordinary English word as a month name: `plugins/planning/skills/interview/SKILL.md` - at line 117 and `plugins/repo-hygiene/skills/clean/context/git-branch-cleanup.md` at line 42. In - each the word starts inside the window (at offset 60 and 59) and the old slice cut it after one - character, so the same truncation that hid the ISO dates had been hiding these. That defect is - untouched here, and reproduces identically on the previous script: it over-reports into the - declined bucket, which is the direction that stays visible to a reader, and is left for its own - fix. - - Patch rather than minor: no flag, no output shape and no configuration changes. The counts move - because the existing expiry check now reaches stamps it had been dropping. - -## [0.3.0] - -### Changed - -- **Rubric version 3: version 2 never said at which scope C3 is graded.** Applying the rubric to - a real corpus passage surfaced it. Two readers reached the same verdict on - `plugins/dometrain/skills/grounding/SKILL.md:50-64` at `d7e391da` (containment 0.589, a - 142-token matched span against `Dometrain/mcp@master` fetched 2026-08-28) and disagreed on which - scope produced it. Under version 2 both readings were available and they resolve in opposite - directions: grade C3 and C4 both at the file and a majority-adapted file *that carries adequate - file-level attribution* clears twice; grade both at the span and a well-attributed derived file - stands every time. - - Version 3 states it: **C3 is graded outward across the whole file, C4 on the passage.** What C3 - tests is whether the attribution's declared scope matches the derivation's. File-scope - attribution discharges C3 when the derivation is file-wide, and does not when one lift sits - inside otherwise-original material, where the header understates and the reader misallocates. - This is a substantive addition, and version 2's "a bare link at the bottom of a long file does - not attribute a specific paragraph in the middle of it" cuts against it. **C4's half is only - written down**: its worked examples and its closing replacement test were already - passage-scoped, so nothing about C4 changes. - - The rejected reading is worth recording because it is the one a judge reaches for: "the - attribution exists and is complete." That is not the test. It would let a single lift into an - otherwise-original file escape C3 on the strength of a header line about something else. - -- **The judge dispatch could not execute the new rule, and now can.** `reference/nomination.md` - handed each judge the local passage, the fetched source, and the rubric, never the containing - file. A C3 graded across the whole file is unanswerable from that, and both the rubric and the - judge prompt instruct UNKNOWN when the text to quote is absent, so a *conforming* judge under - version 3 would have graded C3 UNKNOWN on every candidate, stopping every verdict and routing - every run to the human. The motivating case proves it: the attribution that clears it sits about - 35 lines above the passage. The dispatch now supplies `LOCAL FILE:` and says which criteria are - graded against which input. Blindness in this panel means blind to the pipeline's own suspicion, - meaning the fingerprint numbers, the nomination's reasoning and the other judges, never blind to - the material a criterion is defined over. The lens-diversity stance that read for "whether the - attribution present already discharges the obligation" was pointing judges at the reading - version 3 rejects, and now reads for scope match. - - Carve-outs 1, 4 and 5 are file-level judgments too, and were under-supplied by the - passage-only dispatch before this change. That gap predates version 3; it is closed by the same - fix. - -- **The measurement version 2 stands on does not carry forward.** This file's own rule is that a - criterion change invalidates any measurement pinned to the prior version, and version 3 adds a - scope-match test to C3 that can decide a case either way. **The golden set must be re-scored - against version 3 before any precision figure is cited against it**, and no class becomes - fix-eligible on a measurement pinned to a superseded rubric. Version 2 took the one exception to - that rule on the argument that it changed no criterion's substance; version 3 cannot make that - argument and does not try. - - Said plainly so the figures are not left under a cloud they do not deserve: **no current golden - case appears to turn on the scope question.** Seven carry no attribution anywhere, one is - declined at a carve-out before grading, one fails C1, and the single case with attribution is a - lift inside an otherwise-original file, which resolves identically at either scope. The re-score - is expected to reproduce 8 tp / 0 fp / 0 fn / 2 tn. It is still required, because the rule keys - on a criterion changing rather than on a recorded case flipping, and inventing a second, weaker - exception ("substantive change, but the set does not happen to exercise it") to save a ten-case - re-score that costs nothing is the bad trade. - -## [0.2.1] - -### Fixed - -- **The Phase 6 corpus baseline is stale: it reports Phase 3 figures.** The 0.2.0 entry records - 1,347 tracked files after carve-outs, 525 stamp candidates, 482 parsed, 43 declined, 0 expired, - oldest parsed stamp 2026-04-08. All six reproduce exactly at `33dccc59` - ("corpus, breadcrumb, and stamp scripts, Phase 3 part 1", the commit that introduces - `list-corpus.sh`), clean tree, running the scripts as they existed there. They were then carried - into the Phase 6 paragraph several commits later without re-measuring, so a paragraph presenting - itself as the Phase 6 measurement reports a Phase 3 one. - - The current baseline, at `619199ee` with `--as-of 2026-08-28`: **1,352** tracked markdown files - after carve-outs (1,395 considered, 43 declined at path level), **535** stamp candidates, - **491** parsed, **44** declined at stamp level (20 month-name forms, 24 bare years), **0** - expired at the 180-day default, oldest parsed stamp 2026-04-08 at - `plugins/work-items/skills/track/actions/add.md:104`, and 9 findings at a 60-day window. - `list-corpus`'s path-level `declined: 43` and `check-stamps`'s stamp-level `declined: 44` count - different populations and are not an inconsistency. - - **Two of those figures are date-relative and expire**, which is why the as-of date is pinned - beside the commit rather than left implicit. `0 expired at the 180-day default` holds only - until 2026-10-05 on the current oldest stamp, and `9 findings at a 60-day window` moves daily. - `check-stamps.sh --as-of` reproduces both at the recorded date. A baseline recorded without one - is the same staleness this entry corrects, one turn later. - - **The delta is not what a first reading of it suggested.** It is not `main` moving across #3467 - to #3469: those three contribute **+1 in total**, one added file in #3468. #3467 adds 20 - markdown files and contributes **zero**, because every one lands under `evals/fixtures/golden/` - inside the excluded tree, which is why it raises `considered` by 20 and the fixture decline - from 3 to 23 while leaving the corpus untouched. The rest of the gap is the four months of - corpus growth between Phase 3 and now. Separately, `.claude/provenance.json` is first tracked in - `d7e391da`, so the `excluded_paths` layer postdates the figures in the 0.2.0 paragraph. - - **A note on how the wrong diagnosis was nearly recorded instead**, because the method matters - more than this particular number. A first pass replayed the carve-out filter across 60 commits - reachable from `main`, found 1,347 at none of them, and concluded the figure came from no commit - at all. The originating commits sit on the pre-squash build branch, which the squash-merge made - unreachable from `main`; the reflog held them throughout. A history replay bounded at a squash - boundary cannot answer "does this number come from a commit", and reporting that it can converts - a missing sample into a false negative. - -## [0.2.0] - -### Added - -- **Rubric version 2: an inverted polarity in C3 and C4, caught by blind adjudication.** The - verdict rule says a finding STANDS only if all four criteria PASS, and it says so three times. - But C3 and C4 were phrased as questions whose intuitive "yes" is exculpatory, namely "is the - attribution adequate" and "does the text transform", and their worked examples labeled that - exculpatory answer PASS. Read literally, the two halves of the file contradicted each other and - **no finding could ever stand**. - - Nothing measured was wrong: the version-1 run and the independent blind pass both graded on the - operative verdict rule rather than the labels, and both returned the same eight positives. The - defect was in what the file told the next judge to do. Both criteria are now phrased in the - negative so all four point the same way, the four worked-example labels are corrected, and the - polarity is stated once, explicitly, at the head of the criteria section. Version 2 carries the - version-1 measurement forward and says why, which is the single exception to this catalog's own - invalidation rule. - - Worth recording how it was found: three review passes and a self-check had read this file - without noticing. What surfaced it was asking an agent to actually apply the rubric with the - expectations withheld, the first reader with no way to infer the intended answer. - -- **A contested class the golden set records rather than settles.** Case `c10` is a copy rotated - until no five-word window survives. The pipeline classed it `near-verbatim` at tier - `source-fetched-similar`, on the grounds that a source was fetched and compared; the blind - adjudicator classed it `paraphrase` at `llm-suspected`, on the grounds that zero lexical - evidence is available to a reader who does not already know it was rotated. Both readings are - defensible under the current tier table, which is the finding: a rotated copy with a fetched - source fits neither tier cleanly. The practical stakes are nil today, since both tiers are - report-only and neither is fix-eligible, so the disagreement is recorded here and carried to - the growth round rather than resolved by picking the answer that flatters the score. - -- **The golden set, the first measurement, and the loop that grows it.** Ten synthetic cases under - `skills/audit/evals/fixtures/golden/`, one directory each carrying `case.md`, `expected.json`, - and the `source.md` the case is judged against, so every case runs offline: the source is served - to the fingerprint module directly and the fetch stage is short-circuited rather than mocked. - Coverage is two verbatim positives, five near-verbatim, one paraphrase, and two hard negatives: - a quoted-and-cited excerpt, and the paraphrase-styled-never-copied distractor, which is the false - positive this detector is most likely to produce. Every fixture describes the same fictional - build tool the earlier fixtures use. A golden set holding real copied prose would make this - repository carry the defect the plugin exists to find, in the one tree its own scan is - categorically forbidden to read. - - **The gate decision table**, from `score-golden.sh` over the run recorded at the same date. - Classes are the scorer's own grouping: a case's class is its first expected finding's class, and - a hard negative groups as `negative`. - - | Class | n | Precision | Recall | Gate outcome | - |---|---|---|---|---| - | `verbatim` | 2 | 1.00 | 1.00 | report-only (n=2 below `min_n_per_class` 10) | - | `near-verbatim` | 5 | 1.00 | 1.00 | report-only (n=5 below `min_n_per_class` 10) | - | `paraphrase` | 1 | 1.00 | 1.00 | report-only (n=1 below `min_n_per_class` 10) | - | `negative` | 2 | n/a | n/a | report-only (n=2 below `min_n_per_class` 10) | - | *overall* | 10 | 1.00 | 1.00 | 8 tp, 0 fp, 0 fn, 2 tn, 0 declined | - - **Every class ships report-only, and that is the gate's arithmetic rather than a shortfall.** Ten - cases across four groupings cannot put any class at n=10, so no class is fix-eligible at v1 and - the precision column decides nothing yet. The numbers were not tuned to change that and the gate - was not lowered to meet them: gates bind fix eligibility and release readiness only, never what - the report shows. `verbatim` and `near-verbatim` reaching n=10 at or above the 0.95 bar is the - named exit condition of the first growth round. At n near 10 that bar behaves as a ratchet rather - than as a statistic, since one error demotes a class, and that is accepted. - - **What a perfect score here does and does not establish.** It does not say the detector is - accurate on a corpus. Ten cases were authored at chosen points on the separation curve, and in - this first round the agent that wrote the expectations is the agent that ran the pipeline, so - recall is measured against expectations written by the same hand. What it does establish is a - floor: the deterministic half is genuinely measured, not asserted, and the run would have failed - the set on any contract violation: a paraphrase promoted to `fingerprint-confirmed`, a hard - negative that fired, a span the scorer could not overlap. The adjudication loop below is what - breaks the circularity, because a case converted from a rejected finding is a case nobody - authored to pass. One limit of the tally is worth stating so it is not read as broader than it - is: `score-golden.sh` matches on class equality and span overlap and never compares tiers, so - tier fidelity is pinned by the two new eval cases rather than by this table. - - An independent blind pass was attempted and did not land. A separate agent was given the ten - `case.md` and `source.md` pairs and the rubric, with the expectations and every lexical - measurement withheld, so that its verdicts could stand as the run instead of the author's. It - completed, but no channel was available to retrieve its per-case verdicts, so nothing it produced - contributed to the table above and the figures are single-agent. Recorded as an open item rather - than quietly dropped: the first growth round should re-run that pass and record whether a blind - judge clears both hard negatives, since the distractor in `c07` is the case most likely to - separate an independent judge from this one. - - **The adversarial synonym-rotation probe (design thread T15) returned a real answer.** Cases c08, - c09 and c10 share one source page and differ only in rotation density, which makes density the - single variable. At roughly one substitution every nineteen words the copy is untouched: - containment 0.473 with a 22-word span, both limbs of the separation rule firing. At one every - nine words the span limb dies and containment alone carries it: 0.413 with a longest span of 10, - below the 15-word floor. At one every four words nothing survives: containment 0.0, no matched - spans, against a source that was fetched and identity-checked, which lands the finding at - `source-fetched-similar`, a human report, not fix-eligible, and deliberately not - `llm-suspected`, because a source was in hand. - - Three consequences, recorded rather than acted on. Both limbs of the rule are needed: dropping - either one loses c09. Word-shingling is evadable by an author who intends to evade it, and no - value of `min_containment` above zero recovers a passage with zero matching shingles, so the - answer is not a different number on this axis, which is why the constants were left at the - bundled 0.3 and 15. And c09's containment only clears the threshold because the copy dominates a - short file; the same rotation inside a long host file would dilute containment toward noise while - the 10-word spans stayed under the floor, which is the dilution the span axis was added to - survive and which rotation now defeats. A related note for fix mode: c09's eight matched spans - are too fragmented to fence an edit against, so `fingerprint-confirmed` is not by itself evidence - that a fix is applicable. - - **One measured behavior in the quoted-and-cited negative, found blind and then fixed.** The - blockquote always stripped to nothing, as designed, but a 12-word residue survived at local lines - 19-20 because that fixture's inline quotation opens on one line and closes on the next, and the - inline-quote stripper worked one line at a time: the opening mark is unpaired on its own line and - was left in place rather than swallowing the rest of it. `c06` still cleared, the residue sitting - far below both thresholds with carve-out 3 covering it, and this entry first recorded the gap as - an accepted measurement. The blind adjudication pass rejected that reading, and it was right to. - Hard-wrapped prose is ordinary markdown, so the clearance was luck rather than design: the same - wrapped quotation carried a few words further would have cleared the 15-word span floor and - fired, on a passage that is quoted and attributed. Inline stripping now runs over the paragraph - rather than the line. The paragraph is also the bound, since a blank line, a fence delimiter or a - blockquote line resets the open-quote state, so an unpaired mark or a stray apostrophe still - cannot reach past the block it sits in, which is the conservatism the per-line behavior was - protecting. Stripped characters are blanked in place and newlines are kept, so the line count and - every reported line offset survive untouched and the fix step still has spans it can fence an - edit against. Measured: `c06` moves from containment 0.100, jaccard 0.045 and a 12-word longest - span to containment 0.031, jaccard 0.012 and a 7-word longest span, and the other nine golden - cases do not move at all. The 7 words that remain at line 20 are not quotation residue but the - citation URL matching the source page's own canonical-location line, and they stay in on purpose: - a URL naming the source is evidence of attribution rather than of copying. Carve-out 3 keeps its - reason to exist, because a stripper that follows quotation marks still cannot see a borrowing - that carries none. - - **Widening the pairing scope exposed two further defects, both caught by review rather than by - the suite, and both fixed here.** They are recorded because each one is a case of a fix making a - latent bug reachable, which is the failure mode a widened scope invites. - - First, the closing scan had no word-internal apostrophe guard, though the opening mark has had - one all along. A single-quoted excerpt containing a contraction closed at the apostrophe in - "doesn't", leaving the rest of the excerpt in the token stream. Measured on a five-line fixture: - ten words of quoted upstream text survived, close enough to the 15-word floor that a slightly - longer excerpt would have fired the separation rule, which is precisely the false positive the - stripper exists to prevent. The closing scan now skips apostrophes with word characters on both - sides. The predicate is both-sided rather than the opening guard's one-sided test, because a - legitimate closer nearly always follows a word (`...opts in explicitly'`) and the one-sided form - skipped every real closer, stripping nothing at all. - - Second, and the worse of the two, the opening guard tested only whether a word character preceded - the mark. A possessive following markup, `` `Location`'s ``, `(FILE.md)'s ``, forms this - repository's own prose is full of, therefore opened a phantom quotation. That was survivable - while the closing scan stopped at the next contraction; once pairing learned to skip those, the - phantom ran to the next stray mark instead. Measured across 1,393 tracked markdown files, it - blanked 16,031 characters in the worst case and whole paragraphs of original prose in 32 of them. - **Over-stripping hides real copies, so this was the false-negative direction and the more - dangerous one.** The guard now tests the position: an apostrophe opens a quotation only at the - start of a paragraph, after whitespace, or after an opening bracket. - - The corpus differential over the same 1,393 files now reports 258 differing, of which 256 strip - LESS, recovering prose the previous behavior wrongly blanked, and 2 strip more, both in a file - whose subject is regex quoting patterns and whose extra stripping is a genuine wrapped quotation - being caught correctly. Line-count drift is zero across every file, and all ten golden cases hold - their recorded values. - - **`.claude/provenance.json`, this repository's own config, carrying the fixture-tree exclusion.** - `excluded_paths` lists `**/provenance/skills/audit/evals/fixtures/**`, and that is the whole of - the file: the separation constants, budgets and gates stay at their bundled defaults because - nothing measured here justified moving one. The exclusion lives in config and never in - `list-corpus.sh`, which is the #3041 resolution: an unconditional exclusion would decline the - fixtures under the eval harness's own config isolation and leave the eval author reading prose - instead of results. Measured over `plugins/provenance` with the file in place: 33 considered, 10 - included, 23 declined against that one pattern with its reason named. - - **The adjudication-to-fixture loop, in `reference/dispositions.md`.** A finding the human rejected - and a copy the audit walked past are both measurements the set does not yet contain, and both are - lost unless they are converted. The section states the conversion in order: synthetic rewrite - preserving the shape and never the text, the adjudicated verdict rather than the run's, - registration in `evals.json` before the case counts as landed, and a re-score of the whole set. - It adds the two limits that matter as it grows: a rubric change invalidates every recorded figure - while leaving the fixtures intact, and cases harvested from a sweep are a biased estimator - because they are the cases this detector already got wrong. - - **This run is sidecar-only.** `emit-findings.sh` was not invoked and no relay findings file was - written, because the crosswalk rows for `rule-verbatim-copy` and the two stamp rules land - separately. No relay file ever carries a rule id with no row behind it. - -## [0.1.1] - -### Fixed - -- **`audit`: both `detector unavailable` fallbacks could not render.** The effective-config probe - ended `| head -10 || echo "detector unavailable"` and the stamp-config probe ended - `| tail -3 || echo "detector unavailable"`. `head` and `tail` each exit 0 regardless of what the - script before them did, so a missing or broken `list-corpus.sh` or `check-stamps.sh` rendered an - empty line under a label that reads as a detector reporting nothing to report. Verified by - execution: with each script absent the old shapes rendered `[]` and the new ones render - `[detector unavailable]`; with the real scripts both still render their config lines. Each probe - now runs `--show-config` once to `/dev/null` and pipes the second run into the cap, so the `||` - binds to the script. The double runs cost 10 ms and 17 ms measured on this repository. Every - subcommand still begins with its `${CLAUDE_SKILL_DIR}/scripts/.sh` path, so the existing - grants cover each independently. Nothing was widened. Unchanged and pre-existing: the stamp probe - pipes into `tail`, which `allowed-tools` does not name, though `tail` is one of the Bash tool's - built-in read-only commands and never prompts. - -## [0.1.0] - -### Added - -- **Five review findings fixed, each verified by execution first.** All were real: - - - **`list-corpus.sh .` reported an empty corpus.** The repository root has several spellings and - every one means "the whole corpus", but `.` reached the directory-prefix filter as a literal - prefix, matched no tracked path, and returned zero files with no error. On this repository - that was 0 instead of 1,353, and it read as a clean repository rather than a broken - invocation. Every root spelling now normalizes to the empty prefix. - - **An explicit `"excluded_paths": []` could not clear an inherited exclusion.** Treating "no - elements" as "key absent" left the earlier layer's value in force, so an overlay could add - exclusions but never remove one. Presence, not emptiness, now decides whether a layer - overrides, which is what per-key override actually requires. - - **`emit-findings.sh` reported success having written nothing.** With `set -e` deliberately - off, an uncreatable directory or an unwritable path fell through to the "wrote" message and - exit 0. That is the worst failure a persistence step can have, because nothing downstream - contradicts it: the audit says the findings are relayed and the consumer never scans a file - that does not exist. Both writing steps are now checked, with a new exit 5. - - **Configured separation thresholds never reached the fingerprint module.** The module reads - no config by design, so a repository that tuned `min_containment` or `min_span_words` silently - got the bundled 0.3 and 15, the constants that decide which findings become fix-eligible. The - audit flow now resolves them through the cascade and passes them explicitly, and reports the - values it used. - - **`--show-config` did not say which layer supplied a value.** The setup skill promises - per-value provenance and tells the operator to read it from there rather than parsing the - layers by hand; listing the layers and the effective values separately did not deliver that. - Each value is now attributed to its layer, to the overriding flag, or to the bundled defaults. - -- **Design artifacts graduated to `docs/specs/`, contract slice pruned.** The - `copied-external-content` contract slice was pruned before merge per the topic-docs - contract-slice lifecycle. Its durable half graduated with history preserved: - `provenance-type-inventory.md` (script contracts, finding record, tier enum, config schema, - golden-set case shape, draft crosswalk rows), `provenance-capability-matrix.md`, - `provenance-design-threads.md`, `provenance-plugin-topology.md`, and - `provenance-convention-engagement.md`. Remaining phases 6 to 8 graduated to the work-item - tracker. Every in-plugin pointer to the old `docs/topics/` paths was rewritten, so a script - header names a contract that still resolves. - -- **The two skills, their evals, and the leaf-name registration.** `/provenance:audit` (default - read-only, plus explicit `fix` and `sweep`) and `/provenance:setup` (`check` by default, - `apply` on request), with 8 and 6 eval cases and a `context/gotchas.md` recording the build's - real failure history. - - The audit's action router keeps mutation behind an explicit argument, so a bare invocation - scopes, judges and reports and touches nothing. The untrusted-content spine is carried in the - fetch step, and the fix flow's pointer-liveness check cites that statement rather than - restating it, which keeps one contract in the file instead of two wordings of it. - - The setup skill is human-invoked under the setup contract, and the reason is specific rather - than ceremonial: config decides what the audit is allowed to ignore, so a model proposing its - own exclusions could quiet its own findings. Its eval set pins the refusals that matter, among - them declining to hardcode the fixture exclusion into `list-corpus.sh` and correcting the - premise that raising a gate shortens a report. - - Eval fixtures describe a fictional build tool. A fixture that planted real copied text would - make the plugin's own repository carry the defect it exists to find. - -- **The reference artifacts, the audit's judgment half.** `rubric.md` (shipped at version 1 here, - corrected to version 2 above before this release closed), - `dispositions.md`, `source-fetch.md`, `nomination.md`, and `context/persist-findings.md`. Each - is read at the step that needs it rather than preloaded, so a read-only audit never pays for - the fix discipline and a run with no fetch never reads the fetch route. - - The rubric states its own boundary first, because the four criterion names resemble fair-use - factors and the resemblance is misleading: the verdicts are editorial, the remedies are - maintenance remedies, and a finding says a passage should point at its source rather than - restate it. It never says a passage is unlawful. Carve-outs are evaluated before any criterion, - since several of them make the criteria meaningless rather than merely satisfied, and each - criterion requires a quoted span, with UNKNOWN available when the material needed to quote is - not in front of the judge. - - Two shapes exist to stop a measurement from lying. Judges are blind to the fingerprint numbers - and to each other, because a judge told the containment score turns three samples into one - sample repeated; and the semantic-diff guard reads the before and after without the rewrite - rationale, because an agent told why an edit was made reliably finds that the edit achieved it. - Nomination passes union rather than intersect, since intersecting two recall-biased passes - converts them into a precision filter and discards the recall they were spawned to buy. - - `persist-findings.md` resolves the detector-findings contract through three rungs: the `review` - plugin's bundled copy when that plugin is installed, the publisher's raw URL otherwise, and a - refusal to write when neither is reachable. The first rung is new against the ai-slop precedent - and closes a real gap: fetching a contract from one organization's URL made every offline run - report-only and pointed a portable plugin at a single publisher. - - The untrusted-content framing spine is carried inline byte-identical at both Phase 4 ingest - surfaces, with the site tails naming what these surfaces actually attract: fetched - documentation pages that instruct the reader to copy them, which is the case under audit rather - than a settlement of it. - -- **The five deterministic scripts, the audit's reasoning-free half.** `list-corpus.sh` - enumerates tracked markdown minus the categorical carve-outs; `extract-breadcrumbs.sh` - inventories the provenance signals already in a directory; `check-stamps.sh` flags expired - verification stamps; `emit-findings.sh` projects relay-eligible findings into a conforming - findings file; `score-golden.sh` tallies case-level precision and recall. Each was written - test-first and observed red: 223 cases across the five, all passing. - - Three shapes are contract rather than implementation detail. The eval-fixture exclusion reaches - `list-corpus.sh` through the config layer and never unconditionally, so the eval harness's own - config isolation lifts it and the fixtures report real findings; making that expressible is why - the corpus root and the config root resolve separately. Breadcrumbs are emitted per directory - rather than per file, because a neighbor's citation is what identifies an unfenced copy's - source. And `emit-findings.sh` enforces the relay boundary: only fingerprint-confirmed copies - and the two stamp rules may reach a findings file, judgment verdicts are counted in `Surfaces` - rather than dropped, and their tier names are deliberately absent from the file, since a tier - name in the apply relay's input invites a consumer to act on a verdict this producer withheld - on purpose. - - Two findings cost real measurement. **mawk panics at compile time on interval expressions** - (`{0,4}`), and the panic is quiet enough that the scan simply returns nothing and the script - still exits 0, so a whole rule silently stopped firing until the corpus run showed zero - candidates where hundreds were expected. Every regex in these scripts uses explicit repetition - instead. Second, **"read" is an ordinary English verb**, so at the same keyword window the - explicit stamp verbs use, prose like "an unconfirmed read of a shipped build" became a stamp - candidate, and `context-management-2025-06-27`, an API beta identifier rather than a date, - became an expired-stamp finding. Narrowing the window for that one keyword dropped every such case while - keeping the real `read ` forms: declined candidates fell 54 to 43 and the false finding - went with them. - - Measured over this repository, 1,347 tracked files after carve-outs: 525 stamp candidates, 482 - parsed, 43 declined, 0 expired at the 180-day default (the oldest parsed stamp is 2026-04-08). - The declined count is the honest report the design asks for and not a defect to tune away. The - corpus genuinely carries month-name and bare-year stamp forms, and a parser that guessed at - them would manufacture findings against dates nobody wrote down. - - **These are Phase 3 figures, carried into this Phase 6 paragraph without re-measuring.** They - reproduce exactly at `33dccc59`. See 0.2.1 above for the current baseline and its as-of date. - -- **The fingerprint module, the plugin's one pure library.** Word 5-shingles, - containment, Jaccard, and contiguous matched spans between a local passage and an - already-fetched source, behind a thin CLI. It decides nothing: it reports lexical overlap and - the audit flow maps that evidence to a tier. - - Two behaviors are contract rather than implementation detail, both earned in the spike phase. - Quotation stripping runs inside the module over the local text before shingling and covers - inline quotation marks (straight and curly) as well as blockquotes and code fences, because a - properly quoted and cited excerpt must not read as a copy and a rubric-layer carve-out arrives - too late. Verdicts are matched spans carrying local line offsets, because whole-file - containment diluted a real 27-word match to 0.019 on a 2,912-shingle file; the separation rule - fires on either measure, and the spans are what a fix edits against. - - Written test-first: 21 cases, red before the module existed, with the two amendment fixtures - named in the output (an inline-quoted excerpt that must strip to zero matched spans, and a - real-sized file whose planted span must surface while its ratio goes to noise). An unpaired - quotation mark is left in place rather than swallowing the rest of the line, and a - word-internal apostrophe is not treated as a quote. The `.test.sh` wrapper exists because CI - discovers only `*.test.sh`; without it the module would ship with no CI coverage. - -- **Plugin scaffold and registration.** Manifest, README, and marketplace entry for the - documentation-provenance audit: prose in tracked markdown that restates an externally-owned - fact without a pointer or a conforming stamped record. - - The README carries the boundary against every adjacent owner, the config schema, the fence and - stamped-record marker forms, and the prerequisites, including what the audit still does when web - search is unavailable (breadcrumb-only resolution, with the rest landing on the neutral - `not-found` disposition). The skills, scripts, rubric, and evals land in later phases; the - contract they build against is the graduated specs under `docs/specs/provenance-*.md`, chiefly - `provenance-type-inventory.md` and `provenance-capability-matrix.md`. - - The rubric catalog is versioned with this plugin, so a criterion or carve-out change lands here - and invalidates any golden-set measurement pinned to the prior version. diff --git a/plugins/provenance/README.md b/plugins/provenance/README.md deleted file mode 100644 index b8f920b89d..0000000000 --- a/plugins/provenance/README.md +++ /dev/null @@ -1,20 +0,0 @@ -# provenance (deprecated) - -This plugin was renamed to `attribution`. Everything it did now lives in -`attribution@melodic-software`: `/provenance:audit` is `/attribution:audit` and -`/provenance:setup` is `/attribution:setup`. - -This entry is a one-release shim. Its two skills only tell you where the real skill went; they -do no work. It is removed in a later release. - -## Migrate - -1. Install the renamed plugin: `/plugin install attribution@melodic-software`. -2. Remove `provenance@melodic-software` from `enabledPlugins` in your settings, or uninstall it - with `/plugin uninstall provenance@melodic-software`. -3. Rename a repository's `.claude/provenance.json` to `.claude/attribution.json` (and - `.claude/provenance.local.json` to `.claude/attribution.local.json`, and - `~/.claude/provenance.json` to `~/.claude/attribution.json`). The renamed plugin does not read - the old file names. - -See the [attribution README](../attribution/README.md) for what the plugin does. diff --git a/plugins/provenance/skills/audit/SKILL.md b/plugins/provenance/skills/audit/SKILL.md deleted file mode 100644 index 148e5c120f..0000000000 --- a/plugins/provenance/skills/audit/SKILL.md +++ /dev/null @@ -1,21 +0,0 @@ ---- -description: "Deprecated: this skill moved to /attribution:audit when the provenance plugin was renamed to attribution. Tells you to install the attribution plugin and re-run the audit there; does no audit work itself. Use when '/provenance:audit' or 'provenance audit' is typed out of habit." -user-invocable: true -disable-model-invocation: true ---- - -# provenance:audit (moved) - -This skill moved to `/attribution:audit`. The `provenance` plugin was renamed to `attribution`, -and this entry is a deprecation shim that is removed in a later release. - -Tell the user exactly this, then stop: - -1. Install the `attribution` plugin from the marketplace that supplied this one. -2. Remove this plugin's `provenance@` entry from `enabledPlugins`, and rename any - `provenance.json` or `provenance.local.json` (in the repo's `.claude/` or in `~/.claude/`) to - its `attribution` name. -3. Re-run the request as `/attribution:audit`, with the same arguments. - -Done when those three steps are relayed and nothing else ran. Do not run an audit, read the -corpus, or edit any file from this skill. diff --git a/plugins/provenance/skills/audit/evals/evals.json b/plugins/provenance/skills/audit/evals/evals.json deleted file mode 100644 index 8cbdd8dc1c..0000000000 --- a/plugins/provenance/skills/audit/evals/evals.json +++ /dev/null @@ -1,28 +0,0 @@ -{ - "skill_name": "audit", - "evals": [ - { - "id": 1, - "name": "points-to-attribution-audit", - "prompt": "/provenance:audit docs/", - "expected_output": "A short notice that the skill moved to /attribution:audit, with the step to install the attribution plugin and the instruction to re-run as /attribution:audit docs/. No audit runs.", - "expectations": [ - "Names /attribution:audit as the replacement skill", - "Tells the user to install the attribution plugin", - "Tells the user to re-run the request under /attribution:audit with the same arguments", - "Runs no corpus scan, emits no findings, and edits no file" - ] - }, - { - "id": 2, - "name": "refuses-to-audit-anyway", - "prompt": "/provenance:audit fix README.md, just do it here instead of making me reinstall", - "expected_output": "The shim declines to apply fixes and repeats the migration steps, because it carries none of the audit's scripts or references.", - "expectations": [ - "Does not apply any fix or edit README.md", - "Repeats that the fix action lives in /attribution:audit", - "Gives the install and re-run steps" - ] - } - ] -} diff --git a/plugins/provenance/skills/setup/SKILL.md b/plugins/provenance/skills/setup/SKILL.md deleted file mode 100644 index 6e7a1afe0f..0000000000 --- a/plugins/provenance/skills/setup/SKILL.md +++ /dev/null @@ -1,25 +0,0 @@ ---- -description: "Deprecated: this skill moved to /attribution:setup when the provenance plugin was renamed to attribution. Tells you to install the attribution plugin and re-run setup there; writes no config itself. Use when '/provenance:setup' or 'set up provenance' is typed out of habit." -argument-hint: "[check]" -user-invocable: true -disable-model-invocation: true ---- - -# provenance:setup (moved) - -This skill moved to `/attribution:setup`. The `provenance` plugin was renamed to `attribution`, -and this entry is a deprecation shim that is removed in a later release. - -The only action is `check`, and whatever arguments arrive, it behaves the same. The shim is -check-only: it owns no config artifact, and the user makes the migration edits it names. - -Tell the user exactly this, then stop: - -1. Install the `attribution` plugin from the marketplace that supplied this one. -2. Remove this plugin's `provenance@` entry from `enabledPlugins`, and rename any - `provenance.json` or `provenance.local.json` (in the repo's `.claude/` or in `~/.claude/`) to - its `attribution` name. -3. Re-run the request as `/attribution:setup`, with the same arguments. - -Done when those three steps are relayed and nothing else ran. Do not read or write any config -file from this skill. diff --git a/plugins/provenance/skills/setup/evals/evals.json b/plugins/provenance/skills/setup/evals/evals.json deleted file mode 100644 index 2976c03b80..0000000000 --- a/plugins/provenance/skills/setup/evals/evals.json +++ /dev/null @@ -1,28 +0,0 @@ -{ - "skill_name": "setup", - "evals": [ - { - "id": 1, - "name": "points-to-attribution-setup", - "prompt": "/provenance:setup check", - "narration": true, - "expected_output": "A short notice that the skill moved to /attribution:setup, with the step to install the attribution plugin, the config rename to .claude/attribution.json, and the instruction to re-run as /attribution:setup check. No config is read or written.", - "expectations": [ - "Names /attribution:setup as the replacement skill", - "Tells the user to install the attribution plugin", - "Tells the user to rename .claude/provenance.json to .claude/attribution.json", - "Reads and writes no config file" - ] - }, - { - "id": 2, - "name": "refuses-to-write-config", - "prompt": "/provenance:setup apply", - "expected_output": "The shim declines to write any config and repeats the migration steps, so the change is made under /attribution:setup instead.", - "expectations": [ - "Does not create or edit any .claude config file", - "Tells the user to re-run the request as /attribution:setup apply" - ] - } - ] -} diff --git a/scripts/cheatsheet-config.mjs b/scripts/cheatsheet-config.mjs index cddb00b307..8413db1617 100644 --- a/scripts/cheatsheet-config.mjs +++ b/scripts/cheatsheet-config.mjs @@ -41,7 +41,6 @@ export const EXCLUDED_PLUGINS = new Map([ ["kindle-dedrm", "personal-domain plugin"], ["knowledge", "personal-domain plugin"], ["machine-health", "personal-domain plugin"], - ["provenance", "deprecated shim"], ["songwriting", "personal-domain plugin"], ["x", "personal-domain plugin"], ]); diff --git a/scripts/em-dash-purged-paths.txt b/scripts/em-dash-purged-paths.txt index 54d30fca24..f2c3d96f96 100644 --- a/scripts/em-dash-purged-paths.txt +++ b/scripts/em-dash-purged-paths.txt @@ -398,7 +398,6 @@ plugins/prototype/README.md plugins/prototype/CHANGELOG.md plugins/prototype/context/*.md plugins/prototype/skills/*/SKILL.md -plugins/provenance/README.md plugins/rate-limit-guard/README.md plugins/rate-limit-guard/CHANGELOG.md plugins/rate-limit-guard/bench/README.md diff --git a/scripts/skill-leaf-name-registry.txt b/scripts/skill-leaf-name-registry.txt index 6ffc04ca8a..f71731b131 100644 --- a/scripts/skill-leaf-name-registry.txt +++ b/scripts/skill-leaf-name-registry.txt @@ -104,9 +104,7 @@ setup * # `audit-attribution` repeats the namespace, and a copy-flavored verb like # `dedupe` would name the wrong object -- the finding is a restatement whose # source is external, not a duplicate inside this repository, which is -# `extract-ssot`'s concern and not this plugin's. provenance is the plugin's -# former name and keeps the leaf only as a deprecated shim whose `audit` stub -# points to `attribution:audit`. +# `extract-ssot`'s concern and not this plugin's. # disk-hygiene joins on its own grounds. Its object, the leftovers in one # directory tree, is supplied by the namespace, and the contract is exactly the # verb's: bare invocation runs the engine's one read-only `scan` and reports the @@ -116,7 +114,7 @@ setup * # removal would contradict the verb table's `clean` meaning. `scan` is the # verb-table synonym but names the engine subcommand, not the product, a # findings report with a handoff to `clean` (#5516). -audit ai-slop,attribution,harness-config,harness-memory,codebase-health,context-budget,disk-hygiene,github,instruction-placement,machine-health,mcp-tools,mutation-testing,overengineering,plugin-quality,provenance,repo-fleet-hygiene,testing +audit ai-slop,attribution,harness-config,harness-memory,codebase-health,context-budget,disk-hygiene,github,instruction-placement,machine-health,mcp-tools,mutation-testing,overengineering,plugin-quality,repo-fleet-hygiene,testing # Fixed verb meaning: deterministic pass/fail gate. skill-quality checks a # skill, toolchain checks a build. instruction-placement joins on its own