From 115a39a64ded14d0270c515e784f9e19fb380e05 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Fri, 2 Oct 2026 11:05:46 -0400 Subject: [PATCH] fix(harness-ops): read the native index from the registration detect scores build_native_index replaced a filtered held entry only with a built-in plugin component and walked the lanes in a different order from native_surfaces. It now holds the first unfiltered entry in LANE_ORDER, the registration native_surfaces scores first. A name scored in two lanes now keeps the pair from its first lane, the one the index holds, instead of the last. On the 2.1.287 extraction every detect count is unchanged; the seeded fork pair's score moves from 0.0448 to 0.057, now read from the built-in command its verdict names. Closes #5799 Co-Authored-By: Claude Opus 5.5 --- .../harness-ops/.claude-plugin/plugin.json | 2 +- plugins/harness-ops/CHANGELOG.md | 15 ++++++ .../audit-native-overlap/scripts/overlap.py | 39 +++++--------- .../scripts/test_overlap.py | 52 +++++++++++++++++++ 4 files changed, 82 insertions(+), 26 deletions(-) diff --git a/plugins/harness-ops/.claude-plugin/plugin.json b/plugins/harness-ops/.claude-plugin/plugin.json index 02c08fcd33..4ad8d264a4 100644 --- a/plugins/harness-ops/.claude-plugin/plugin.json +++ b/plugins/harness-ops/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "harness-ops", - "version": "1.2.0", + "version": "1.2.1", "description": "Claude Code operations toolkit. Fifteen skills: audit-skill-visibility (audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used: a skill is invisible when its description is dropped by Claude Code's skill-listing context budget, which sheds descriptions lowest-score-first so an unused skill loses the keywords that would let it be matched, from skills genuinely not wanted, from skills the run cannot observe at all; computes whether the listing overflows from documented settings, and withholds every cold verdict the data cannot support rather than reporting absence of data as absence of use), inventory (read-only enumeration of the complete invocable surface: every built-in CLI command with aliases and hidden/gated status, every bundled skill, every built-in subagent and tool, every built-in plugin with its components, and every component of every installed plugin across all marketplaces; reads the shipped binary because upstream publishes no built-in command list, and carries an integrity verdict so a drifted build reports counts as floors rather than silently short totals), audit-install-state (read-only audit of the machine-scope ~/.claude installation directory and ~/.claude.json: full inventory split into an authored surface and rolled-up bulk trees, product-managed retention vs genuinely unmanaged state, filename-scheme resolution before any process-liveness check, and deliberate/mid-experiment detection; reports, never deletes), audit-performance (read-only slowness-diagnostic capture run at the moment the machine or a session feels slow: CLI version, retention-sweep health including the unparsable-settings pause, which warns in /status, a timed census walk of the install tree as a sweep-cost proxy, active-session and plugin-fleet counts, a process census, and the fan-out layer, which covers a load-labeled no-op spawn baseline, every hook that will fire bucketed per-tool-call versus per-turn with its invocation shape, the configured statusline, subagent concurrency and spawn-depth ceilings against documented defaults, whether running sessions predate the settings file they are judged by, and orphan attribution by parent liveness rather than age, plus on Windows a kernel-object census (Token objects against uptime, paged pool) that names a host-level leak beneath all four suspects; read against a bundled known-performance-issues reference that also records the causes tested and cleared; separates the four documented suspects of accumulated state, version regression, component bloat, and per-spawn fan-out cost, and routes remediation out; reports, never mutates, and never executes a discovered hook or statusline command), audit-native-overlap (map native Claude Code surfaces, namely built-in CLI commands, bundled skills, plugin-backed built-ins, and session-provided skills, against the current repo's plugin skills and agents, so a custom component never silently duplicates what Claude Code itself ships; bare invocation is a read-only overlap report carrying the extraction's integrity floors and a shared-listing-budget exposure section, verdicts are human-gated in a committed store rendered into a generated registry whose every row carries an observable recheck trigger, and only an explicit apply step bakes presence-gated native references into descriptions and Boundary sections), observability (read locally captured telemetry from the OTEL store, the collector, the per-session hook event log and hook-event JSONL, and ccusage, with trend reports, a per-session report of what fired, what was blocked and the event timeline, and store pruning), known-issues (search known Claude product GitHub bugs, check service health, maintain a persistent tracked-issue registry), changelog (ingest Claude Code changelog entries and turn them into decisions: apply executes those in scope one PR per owner plugin and hands larger ones off as work items, then re-extract the native surface and file its drift as work items), prerequisites (read-only table of external binaries declared by enabled plugins; never installs), check (read-only check that node and jq resolve for the harness-ops hooks; never installs), machine-profile (discover this machine's facts and per-tree identity domains, store them as a re-runnable profile with the observation behind every value, and diff the stored profile against the host now; read-only unless the operator confirms a write, never installs and never reapplies a stored value on its own), plugins (bring a machine's plugin fleet current on demand: marketplace refresh, effective-scope updates including in-repo project/local installs, new-plugin install per policy, scope-divergence detection and explicit convergence), morning-brief (read-only gh-based operator morning view: queue-label counts, merge-ready PRs, parked decisions with their RECOMMENDED lines, and loop-lane telemetry freshness), lanes (start/restart/stop/status loop lanes as named background Claude Code sessions seeded from canonical prompt files, with per-lane model/effort, a repo-pull + marketplace-refresh launch step, and a consume-restarts action, an OS-schedulable reader that relaunches stopped lanes whose telemetry carries a restart_request), and a re-runnable setup action that settles where the known-issues registry, the skill-usage log and the hook log root live, places the root's self-ignoring guard, and detects retired conventions. Plus an opt-in, default-off per-session hook event log (one JSON line per hook event on every event the generated registry marks observable, written to /sessions/.jsonl, with SessionEnd retention by session count or age and an optional detached pre-prune command), a family of eight advisory *-audit hooks (API errors, config changes, instruction loads, permission denials, pre-compaction, skill usage, tool failures, and unsurfaced hook failures. The last also warns the user via systemMessage, since a hook that fails to launch enforces nothing and Claude Code surfaces the failure to nobody) that emit the shared hook-telemetry envelope, and a reference sink that routes envelopes under the same root: per session when the envelope carries a session id, else into the shared hook-events.jsonl the observability skill reads.", "author": { "name": "Melodic Software", diff --git a/plugins/harness-ops/CHANGELOG.md b/plugins/harness-ops/CHANGELOG.md index b38d91f4af..1f265977c9 100644 --- a/plugins/harness-ops/CHANGELOG.md +++ b/plugins/harness-ops/CHANGELOG.md @@ -3,6 +3,21 @@ All notable changes to the `harness-ops` plugin are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); this plugin uses semantic versioning. +## [1.2.1] - 2026-10-02 + +### Fixed + +- **`audit-native-overlap` detect reads seeded pairs and dismissals against the registration it + scores.** The native index that seeded pairs and dismissal drift read replaced a filtered + (empty or `internal`) entry only with a built-in plugin component, and walked the lanes in a + different order from scoring. It now holds the first entry that is not filtered, in scoring + lane order, so an internal bundled skill followed by a built-in command and a built-in plugin + component of the same name resolves to the command in both. A name scored in two lanes now + reports the pair from its first lane, the one the index holds, instead of the last. On the + 2.1.287 extraction the only change is the seeded `fork` pair's score (0.0448 to 0.057), now + read from the built-in command its verdict names rather than the built-in agent of the same + name; every count is unchanged. + ## [1.2.0] - 2026-10-02 ### Added diff --git a/plugins/harness-ops/skills/audit-native-overlap/scripts/overlap.py b/plugins/harness-ops/skills/audit-native-overlap/scripts/overlap.py index 6b7775c0a0..50277942e3 100755 --- a/plugins/harness-ops/skills/audit-native-overlap/scripts/overlap.py +++ b/plugins/harness-ops/skills/audit-native-overlap/scripts/overlap.py @@ -1403,28 +1403,18 @@ def check_presence_mentions(repo: Path) -> list[str]: def build_native_index( inventory: dict[str, Any], lane_payloads: dict[str, Any] ) -> dict[str, dict[str, Any]]: - """Native name -> {class, lane, entry}, first lane wins in workflow, skill, - command, agent, tool, built-in plugin order; plugin-backed built-ins + """Native name -> {class, lane, entry}: the first entry that is not + filtered, in LANE_ORDER, the registration native_surfaces scores first; a + name whose every entry is filtered keeps its first. Plugin-backed built-ins override. `lane` is the lane the entry was read from: two lanes share the plugin-backed-builtin class, so the class alone cannot name it.""" native_index: dict[str, dict[str, Any]] = {} - for lane in ( - "bundled_workflows", - "bundled_skills", - "builtin_commands", - "builtin_agents", - "builtin_tools", - PLUGIN_COMPONENT_LANE, - ): + for lane in LANE_ORDER: + if lane == "plugin_backed": + continue for name, entry in (lane_payloads.get(lane) or {}).items(): held = native_index.get(name) - # A built-in plugin component takes a name only a filtered entry - # held, the same selection native_surfaces scores. - if held is None or ( - lane == PLUGIN_COMPONENT_LANE - and is_filtered(held["entry"]) - and not is_filtered(entry) - ): + if held is None or (is_filtered(held["entry"]) and not is_filtered(entry)): native_index[name] = { "class": CLASS_OF_LANE[lane], "lane": lane, @@ -1674,14 +1664,13 @@ def resurfaced_evidence(resurfaced: dict[str, Any] | None) -> list[str]: ] def keyed(scored: list[discover.Scored]) -> dict[tuple[str, str, str, str], Any]: - return { - key_of(s.name, {"plugin": c.plugin, "skill": c.name, "kind": c.kind}): ( - s, - score, - matched, - ) - for s, c, score, matched in scored - } + # A name scored in two lanes keeps its first surface's pair, the + # registration the native index holds. + out: dict[tuple[str, str, str, str], Any] = {} + for s, c, score, matched in scored: + key = key_of(s.name, {"plugin": c.plugin, "skill": c.name, "kind": c.kind}) + out.setdefault(key, (s, score, matched)) + return out # Seeds read their score from every scored pair, so a seed below the # discovery cut still shows how far lexical evidence alone would carry it. diff --git a/plugins/harness-ops/skills/audit-native-overlap/scripts/test_overlap.py b/plugins/harness-ops/skills/audit-native-overlap/scripts/test_overlap.py index 5758c27402..15a5fb87da 100755 --- a/plugins/harness-ops/skills/audit-native-overlap/scripts/test_overlap.py +++ b/plugins/harness-ops/skills/audit-native-overlap/scripts/test_overlap.py @@ -12,6 +12,7 @@ import contextlib import io +import itertools import json import sys import tempfile @@ -1932,6 +1933,18 @@ def test_a_discovered_candidate_carries_score_tokens_and_no_verdict(self): self.assertIsNone(candidate["store_verdict"]) self.assertEqual(report["discovery"]["discovered"], 1) + def test_a_name_scored_in_two_lanes_reports_the_indexed_surface(self): + commit = {"commit": {"name": "commit", "description": "Create a git commit"}} + self.write_inventory( + builtin_commands=commit, bundled_skills={}, builtin_agents=commit + ) + [candidate] = self.discovered(self.detect()[1]) + index = overlap.build_native_index( + {}, overlap._lane_payloads(json.loads(self.inventory_path.read_text())) + ) + self.assertEqual(candidate["native"]["class"], index["commit"]["class"]) + self.assertEqual(candidate["native"]["class"], "builtin-command") + def test_a_high_threshold_or_zero_top_k_discovers_nothing(self): self.write_inventory() self.assertEqual(self.discovered(self.detect("--threshold", "1.01")[1]), []) @@ -2277,6 +2290,45 @@ def test_a_filtered_earlier_entry_does_not_claim_a_plugin_name(self) -> None: index = overlap.build_native_index({}, payloads) self.assertEqual(index["diff"]["lane"], "builtin_plugins") + def test_a_filtered_entry_yields_to_a_later_ordinary_lane(self) -> None: + payloads = overlap._lane_payloads( + { + "bundled_skills": { + "diff": {"name": "diff", "description": "Diff", "internal": True} + }, + "builtin_commands": {"diff": {"name": "diff", "description": "Diff"}}, + "builtin_plugins": BUILTIN_PLUGINS, + } + ) + [scored] = [s for s in overlap.native_surfaces(payloads) if s.name == "diff"] + self.assertEqual(scored.lane, "builtin_commands") + index = overlap.build_native_index({}, payloads) + self.assertEqual(index["diff"]["lane"], scored.lane) + + def test_the_index_holds_the_surface_scored_first_in_every_lane_mix( + self, + ) -> None: + entries = { + "valid": [{"name": "x", "description": "X"}], + "internal": [{"name": "x", "description": "X", "internal": True}], + } + lanes = [lane for lane in overlap.LANE_ORDER if lane != "plugin_backed"] + for states in itertools.product((None, "valid", "internal"), repeat=len(lanes)): + present = { + lane: {"x": entries[state]} + for lane, state in zip(lanes, states, strict=True) + if state is not None + } + if not present: + continue + with self.subTest( + present={lane: s for lane, s in zip(lanes, states, strict=True) if s} + ): + index = overlap.build_native_index({}, present) + scored = [s for s in overlap.native_surfaces(present) if s.name == "x"] + expected = scored[0].lane if scored else next(iter(present)) + self.assertEqual(index["x"]["lane"], expected) + def test_an_internal_plugin_backed_name_gets_no_fallback_surface(self) -> None: payloads = overlap._lane_payloads( {