diff --git a/plugins/harness-ops/.claude-plugin/plugin.json b/plugins/harness-ops/.claude-plugin/plugin.json index 3287316fb2..7f842afd8c 100644 --- a/plugins/harness-ops/.claude-plugin/plugin.json +++ b/plugins/harness-ops/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "harness-ops", - "version": "1.1.0", + "version": "1.1.1", "description": "Claude Code operations toolkit. Fifteen skills: audit-skill-visibility (audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used: a skill is invisible when its description is dropped by Claude Code's skill-listing context budget, which sheds descriptions lowest-score-first so an unused skill loses the keywords that would let it be matched, from skills genuinely not wanted, from skills the run cannot observe at all; computes whether the listing overflows from documented settings, and withholds every cold verdict the data cannot support rather than reporting absence of data as absence of use), inventory (read-only enumeration of the complete invocable surface: every built-in CLI command with aliases and hidden/gated status, every bundled skill, every built-in subagent and tool, every built-in plugin with its components, and every component of every installed plugin across all marketplaces; reads the shipped binary because upstream publishes no built-in command list, and carries an integrity verdict so a drifted build reports counts as floors rather than silently short totals), audit-install-state (read-only audit of the machine-scope ~/.claude installation directory and ~/.claude.json: full inventory split into an authored surface and rolled-up bulk trees, product-managed retention vs genuinely unmanaged state, filename-scheme resolution before any process-liveness check, and deliberate/mid-experiment detection; reports, never deletes), audit-performance (read-only slowness-diagnostic capture run at the moment the machine or a session feels slow: CLI version, retention-sweep health including the unparsable-settings pause, which warns in /status, a timed census walk of the install tree as a sweep-cost proxy, active-session and plugin-fleet counts, a process census, and the fan-out layer, which covers a load-labeled no-op spawn baseline, every hook that will fire bucketed per-tool-call versus per-turn with its invocation shape, the configured statusline, subagent concurrency and spawn-depth ceilings against documented defaults, whether running sessions predate the settings file they are judged by, and orphan attribution by parent liveness rather than age, plus on Windows a kernel-object census (Token objects against uptime, paged pool) that names a host-level leak beneath all four suspects; read against a bundled known-performance-issues reference that also records the causes tested and cleared; separates the four documented suspects of accumulated state, version regression, component bloat, and per-spawn fan-out cost, and routes remediation out; reports, never mutates, and never executes a discovered hook or statusline command), audit-native-overlap (map native Claude Code surfaces, namely built-in CLI commands, bundled skills, plugin-backed built-ins, and session-provided skills, against the current repo's plugin skills and agents, so a custom component never silently duplicates what Claude Code itself ships; bare invocation is a read-only overlap report carrying the extraction's integrity floors and a shared-listing-budget exposure section, verdicts are human-gated in a committed store rendered into a generated registry whose every row carries an observable recheck trigger, and only an explicit apply step bakes presence-gated native references into descriptions and Boundary sections), observability (read locally captured telemetry from the OTEL store, the collector, the per-session hook event log and hook-event JSONL, and ccusage, with trend reports, a per-session report of what fired, what was blocked and the event timeline, and store pruning), known-issues (search known Claude product GitHub bugs, check service health, maintain a persistent tracked-issue registry), changelog (ingest Claude Code changelog entries and turn them into decisions: apply executes those in scope one PR per owner plugin and hands larger ones off as work items, then re-extract the native surface and file its drift as work items), prerequisites (read-only table of external binaries declared by enabled plugins; never installs), check (read-only check that node and jq resolve for the harness-ops hooks; never installs), machine-profile (discover this machine's facts and per-tree identity domains, store them as a re-runnable profile with the observation behind every value, and diff the stored profile against the host now; read-only unless the operator confirms a write, never installs and never reapplies a stored value on its own), plugins (bring a machine's plugin fleet current on demand: marketplace refresh, effective-scope updates including in-repo project/local installs, new-plugin install per policy, scope-divergence detection and explicit convergence), morning-brief (read-only gh-based operator morning view: queue-label counts, merge-ready PRs, parked decisions with their RECOMMENDED lines, and loop-lane telemetry freshness), lanes (start/restart/stop/status loop lanes as named background Claude Code sessions seeded from canonical prompt files, with per-lane model/effort, a repo-pull + marketplace-refresh launch step, and a consume-restarts action, an OS-schedulable reader that relaunches stopped lanes whose telemetry carries a restart_request), and a re-runnable setup action that settles where the known-issues registry, the skill-usage log and the hook log root live, places the root's self-ignoring guard, and detects retired conventions. Plus an opt-in, default-off per-session hook event log (one JSON line per hook event on every event the generated registry marks observable, written to /sessions/.jsonl, with SessionEnd retention by session count or age and an optional detached pre-prune command), a family of eight advisory *-audit hooks (API errors, config changes, instruction loads, permission denials, pre-compaction, skill usage, tool failures, and unsurfaced hook failures. The last also warns the user via systemMessage, since a hook that fails to launch enforces nothing and Claude Code surfaces the failure to nobody) that emit the shared hook-telemetry envelope, and a reference sink that routes envelopes under the same root: per session when the envelope carries a session id, else into the shared hook-events.jsonl the observability skill reads.", "author": { "name": "Melodic Software", diff --git a/plugins/harness-ops/CHANGELOG.md b/plugins/harness-ops/CHANGELOG.md index a66054b2e0..80fa324bda 100644 --- a/plugins/harness-ops/CHANGELOG.md +++ b/plugins/harness-ops/CHANGELOG.md @@ -3,6 +3,19 @@ All notable changes to the `harness-ops` plugin are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); this plugin uses semantic versioning. +## [1.1.1] - 2026-10-02 + +### Fixed + +- **`audit-native-overlap` detect scores a built-in plugin surface whose name an earlier lane + entry held but was filtered out.** Building the native index marked a name seen before + dropping an empty or `internal` entry, so a later built-in plugin or plugin component of the + same name was skipped as a duplicate and its overlap candidates were never produced. A name + is now marked seen only when its entry is kept; a filtered name still gets no fallback + `plugin_backed` surface, so an internal plugin-backed command stays unscored. The native index + that seeded pairs and dismissal drift read applies the same selection. On the 2.1.287 + extraction the candidate report is unchanged, since no name there collides that way. + ## [1.1.0] - 2026-10-02 ### Added diff --git a/plugins/harness-ops/skills/audit-native-overlap/scripts/overlap.py b/plugins/harness-ops/skills/audit-native-overlap/scripts/overlap.py index ca6775ef89..6b7775c0a0 100755 --- a/plugins/harness-ops/skills/audit-native-overlap/scripts/overlap.py +++ b/plugins/harness-ops/skills/audit-native-overlap/scripts/overlap.py @@ -453,22 +453,32 @@ def load_components(repo: Path) -> list[discover.Component]: return corpus +def is_filtered(entry: Any) -> bool: + """An entry with no registration, or an `internal` one, is never scored.""" + registrations = registrations_of(entry) + return not registrations or any(r.get("internal") for r in registrations) + + def native_surfaces(lane_payloads: dict[str, Any]) -> list[discover.Surface]: """Every scorable native surface. An `internal` registration is skipped: it is plumbing the product never offers anyone to type or call.""" plugin_backed = lane_payloads.get("plugin_backed") or {} surfaces: list[discover.Surface] = [] seen: set[str] = set() + # A filtered name still keeps the plugin_backed fallback from fabricating + # an unmarked surface for it, but does not claim a built-in plugin's name. + filtered: set[str] = set() for lane, payload in lane_payloads.items(): if lane == "plugin_backed": continue for name, entry in payload.items(): if lane == PLUGIN_COMPONENT_LANE and name in seen: continue - seen.add(name) - registrations = registrations_of(entry) - if not registrations or any(r.get("internal") for r in registrations): + if is_filtered(entry): + filtered.add(name) continue + registrations = registrations_of(entry) + seen.add(name) # A plugin-backed name the extractor enriched in this lane is one # surface, scored once, under the plugin-backed class. klass, source = ( @@ -478,7 +488,7 @@ def native_surfaces(lane_payloads: dict[str, Any]) -> list[discover.Surface]: ) surfaces.append(discover.Surface.build(name, klass, source, registrations)) for name, plugin in plugin_backed.items(): - if name not in seen: + if name not in seen and name not in filtered: registrations = [{"name": name, "plugin_name": plugin}] surfaces.append( discover.Surface.build( @@ -1407,9 +1417,19 @@ def build_native_index( PLUGIN_COMPONENT_LANE, ): for name, entry in (lane_payloads.get(lane) or {}).items(): - native_index.setdefault( - name, {"class": CLASS_OF_LANE[lane], "lane": lane, "entry": entry} - ) + held = native_index.get(name) + # A built-in plugin component takes a name only a filtered entry + # held, the same selection native_surfaces scores. + if held is None or ( + lane == PLUGIN_COMPONENT_LANE + and is_filtered(held["entry"]) + and not is_filtered(entry) + ): + native_index[name] = { + "class": CLASS_OF_LANE[lane], + "lane": lane, + "entry": entry, + } for name, plugin in (inventory.get("plugin_backed") or {}).items(): # The extractor enriches a same-named command or skill with the plugin; # reclassify that registration rather than replace it with a bare one. diff --git a/plugins/harness-ops/skills/audit-native-overlap/scripts/test_overlap.py b/plugins/harness-ops/skills/audit-native-overlap/scripts/test_overlap.py index 2ad3d80d4f..5758c27402 100755 --- a/plugins/harness-ops/skills/audit-native-overlap/scripts/test_overlap.py +++ b/plugins/harness-ops/skills/audit-native-overlap/scripts/test_overlap.py @@ -2258,6 +2258,37 @@ def test_a_name_another_lane_holds_is_not_scored_twice(self) -> None: self.assertEqual(index["diff"]["class"], "builtin-command") self.assertEqual(index["author"]["class"], "plugin-backed-builtin") + def test_a_filtered_earlier_entry_does_not_claim_a_plugin_name(self) -> None: + for label, entry in ( + ("internal", {"name": "diff", "description": "Diff", "internal": True}), + ("empty", []), + ): + with self.subTest(label): + payloads = overlap._lane_payloads( + { + "builtin_commands": {"diff": entry}, + "builtin_plugins": BUILTIN_PLUGINS, + } + ) + diff = [ + s for s in overlap.native_surfaces(payloads) if s.name == "diff" + ] + self.assertEqual([s.lane for s in diff], ["builtin_plugins"]) + index = overlap.build_native_index({}, payloads) + self.assertEqual(index["diff"]["lane"], "builtin_plugins") + + def test_an_internal_plugin_backed_name_gets_no_fallback_surface(self) -> None: + payloads = overlap._lane_payloads( + { + "builtin_commands": { + "scan": {"name": "scan", "description": "Scan", "internal": True} + }, + "plugin_backed": {"scan": "scanner"}, + } + ) + names = [s.name for s in overlap.native_surfaces(payloads)] + self.assertNotIn("scan", names) + def make_dismissal(**overrides): entry = {