Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion plugins/harness-ops/.claude-plugin/plugin.json
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
{
"$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
"name": "harness-ops",
"version": "2.0.0",
"version": "2.0.1",
"description": "Claude Code operations toolkit. Fifteen skills: audit-skill-visibility (audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used: a skill is invisible when its description is dropped by Claude Code's skill-listing context budget, which sheds descriptions lowest-score-first so an unused skill loses the keywords that would let it be matched, from skills genuinely not wanted, from skills the run cannot observe at all; computes whether the listing overflows from documented settings, and withholds every cold verdict the data cannot support rather than reporting absence of data as absence of use), inventory (read-only enumeration of the complete invocable surface: every built-in CLI command with aliases and hidden/gated status, every bundled skill, every built-in subagent and tool, every built-in plugin with its components, and every component of every installed plugin across all marketplaces; reads the shipped binary because upstream publishes no built-in command list, and carries an integrity verdict so a drifted build reports counts as floors rather than silently short totals), audit-install-state (read-only audit of the machine-scope ~/.claude installation directory and ~/.claude.json: full inventory split into an authored surface and rolled-up bulk trees, product-managed retention vs genuinely unmanaged state, filename-scheme resolution before any process-liveness check, and deliberate/mid-experiment detection; reports, never deletes), audit-performance (read-only slowness-diagnostic capture run at the moment the machine or a session feels slow: CLI version, retention-sweep health including the unparsable-settings pause, which warns in /status, a timed census walk of the install tree as a sweep-cost proxy, active-session and plugin-fleet counts, a process census, and the fan-out layer, which covers a load-labeled no-op spawn baseline, every hook that will fire bucketed per-tool-call versus per-turn with its invocation shape, the configured statusline, subagent concurrency and spawn-depth ceilings against documented defaults, whether running sessions predate the settings file they are judged by, and orphan attribution by parent liveness rather than age, plus on Windows a kernel-object census (Token objects against uptime, paged pool) that names a host-level leak beneath all four suspects; read against a bundled known-performance-issues reference that also records the causes tested and cleared; separates the four documented suspects of accumulated state, version regression, component bloat, and per-spawn fan-out cost, and routes remediation out; reports, never mutates, and never executes a discovered hook or statusline command), audit-native-overlap (map native Claude Code surfaces, namely built-in CLI commands, bundled skills, plugin-backed built-ins, and session-provided skills, against the current repo's plugin skills and agents, so a custom component never silently duplicates what Claude Code itself ships; bare invocation is a read-only overlap report carrying the extraction's integrity floors and a shared-listing-budget exposure section, verdicts are human-gated in a committed store rendered into a generated registry whose every row carries an observable recheck trigger, and only an explicit apply step bakes presence-gated native references into descriptions and Boundary sections), observability (read locally captured telemetry from the OTEL store, the collector, the per-session hook event log and hook-event JSONL, and ccusage, with trend reports, a per-session report of what fired, what was blocked and the event timeline, and store pruning), known-issues (search known Claude product GitHub bugs, check service health, maintain a persistent tracked-issue registry), changelog (ingest Claude Code changelog entries and turn them into decisions: apply executes those in scope one PR per owner plugin and hands larger ones off as work items, then re-extract the native surface and file its drift as work items), prerequisites (read-only table of external binaries declared by enabled plugins; never installs), check (read-only check that node and jq resolve for the harness-ops hooks; never installs), machine-profile (discover this machine's facts and per-tree identity domains, store them as a re-runnable profile with the observation behind every value, and diff the stored profile against the host now; read-only unless the operator confirms a write, never installs and never reapplies a stored value on its own), plugins (bring a machine's plugin fleet current on demand: marketplace refresh, effective-scope updates including in-repo project/local installs, new-plugin install per policy, scope-divergence detection and explicit convergence), morning-brief (read-only gh-based operator morning view: queue-label counts, merge-ready PRs, parked decisions with their RECOMMENDED lines, and loop-lane telemetry freshness), lanes (start/restart/stop/status loop lanes as named background Claude Code sessions seeded from canonical prompt files, with per-lane model/effort, a repo-pull + marketplace-refresh launch step, and a consume-restarts action, an OS-schedulable reader that relaunches stopped lanes whose telemetry carries a restart_request), and a re-runnable setup action that settles where the known-issues registry, the skill-usage log and the hook log root live, places the root's self-ignoring guard, and detects retired conventions. Plus an opt-in, default-off per-session hook event log (one JSON line per hook event on every event the generated registry marks observable, written to <root>/sessions/<session_id>.jsonl, with SessionEnd retention by session count or age and an optional detached pre-prune command), a family of eight advisory *-audit hooks (API errors, config changes, instruction loads, permission denials, pre-compaction, skill usage, tool failures, and unsurfaced hook failures. The last also warns the user via systemMessage, since a hook that fails to launch enforces nothing and Claude Code surfaces the failure to nobody) that emit the shared hook-telemetry envelope, and a reference sink that routes envelopes under the same root: per session when the envelope carries a session id, else into the shared hook-events.jsonl the observability skill reads.",
"author": {
"name": "Melodic Software",
Expand Down
15 changes: 15 additions & 0 deletions plugins/harness-ops/CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,21 @@
All notable changes to the `harness-ops` plugin are documented here. Format follows
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/); this plugin uses semantic versioning.

## [2.0.1] - 2026-10-02

### Fixed

- **`audit-native-overlap` detect reads seeded pairs and dismissals against the registration it
scores.** The native index that seeded pairs and dismissal drift read replaced a filtered
(empty or `internal`) entry only with a built-in plugin component, and walked the lanes in a
different order from scoring. It now holds the first entry that is not filtered, in scoring
lane order, so an internal bundled skill followed by a built-in command and a built-in plugin
component of the same name resolves to the command in both. A name scored in two lanes now
reports the pair from its first lane, the one the index holds, instead of the last. On the
2.1.287 extraction the only change is the seeded `fork` pair's score (0.0448 to 0.057), now
read from the built-in command its verdict names rather than the built-in agent of the same
name; every count is unchanged.

## [2.0.0] - 2026-10-02

### Changed
Expand Down
39 changes: 14 additions & 25 deletions plugins/harness-ops/skills/audit-native-overlap/scripts/overlap.py
Original file line number Diff line number Diff line change
Expand Up @@ -1403,28 +1403,18 @@ def check_presence_mentions(repo: Path) -> list[str]:
def build_native_index(
inventory: dict[str, Any], lane_payloads: dict[str, Any]
) -> dict[str, dict[str, Any]]:
"""Native name -> {class, lane, entry}, first lane wins in workflow, skill,
command, agent, tool, built-in plugin order; plugin-backed built-ins
"""Native name -> {class, lane, entry}: the first entry that is not
filtered, in LANE_ORDER, the registration native_surfaces scores first; a
name whose every entry is filtered keeps its first. Plugin-backed built-ins
override. `lane` is the lane the entry was read from: two lanes share the
plugin-backed-builtin class, so the class alone cannot name it."""
native_index: dict[str, dict[str, Any]] = {}
for lane in (
"bundled_workflows",
"bundled_skills",
"builtin_commands",
"builtin_agents",
"builtin_tools",
PLUGIN_COMPONENT_LANE,
):
for lane in LANE_ORDER:
if lane == "plugin_backed":
continue
for name, entry in (lane_payloads.get(lane) or {}).items():
held = native_index.get(name)
# A built-in plugin component takes a name only a filtered entry
# held, the same selection native_surfaces scores.
if held is None or (
lane == PLUGIN_COMPONENT_LANE
and is_filtered(held["entry"])
and not is_filtered(entry)
):
if held is None or (is_filtered(held["entry"]) and not is_filtered(entry)):
native_index[name] = {
"class": CLASS_OF_LANE[lane],
"lane": lane,
Expand Down Expand Up @@ -1674,14 +1664,13 @@ def resurfaced_evidence(resurfaced: dict[str, Any] | None) -> list[str]:
]

def keyed(scored: list[discover.Scored]) -> dict[tuple[str, str, str, str], Any]:
return {
key_of(s.name, {"plugin": c.plugin, "skill": c.name, "kind": c.kind}): (
s,
score,
matched,
)
for s, c, score, matched in scored
}
# A name scored in two lanes keeps its first surface's pair, the
# registration the native index holds.
out: dict[tuple[str, str, str, str], Any] = {}
for s, c, score, matched in scored:
key = key_of(s.name, {"plugin": c.plugin, "skill": c.name, "kind": c.kind})
out.setdefault(key, (s, score, matched))
return out

# Seeds read their score from every scored pair, so a seed below the
# discovery cut still shows how far lexical evidence alone would carry it.
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@

import contextlib
import io
import itertools
import json
import sys
import tempfile
Expand Down Expand Up @@ -1932,6 +1933,18 @@ def test_a_discovered_candidate_carries_score_tokens_and_no_verdict(self):
self.assertIsNone(candidate["store_verdict"])
self.assertEqual(report["discovery"]["discovered"], 1)

def test_a_name_scored_in_two_lanes_reports_the_indexed_surface(self):
commit = {"commit": {"name": "commit", "description": "Create a git commit"}}
self.write_inventory(
builtin_commands=commit, bundled_skills={}, builtin_agents=commit
)
[candidate] = self.discovered(self.detect()[1])
index = overlap.build_native_index(
{}, overlap._lane_payloads(json.loads(self.inventory_path.read_text()))
)
self.assertEqual(candidate["native"]["class"], index["commit"]["class"])
self.assertEqual(candidate["native"]["class"], "builtin-command")

def test_a_high_threshold_or_zero_top_k_discovers_nothing(self):
self.write_inventory()
self.assertEqual(self.discovered(self.detect("--threshold", "1.01")[1]), [])
Expand Down Expand Up @@ -2277,6 +2290,45 @@ def test_a_filtered_earlier_entry_does_not_claim_a_plugin_name(self) -> None:
index = overlap.build_native_index({}, payloads)
self.assertEqual(index["diff"]["lane"], "builtin_plugins")

def test_a_filtered_entry_yields_to_a_later_ordinary_lane(self) -> None:
payloads = overlap._lane_payloads(
{
"bundled_skills": {
"diff": {"name": "diff", "description": "Diff", "internal": True}
},
"builtin_commands": {"diff": {"name": "diff", "description": "Diff"}},
"builtin_plugins": BUILTIN_PLUGINS,
}
)
[scored] = [s for s in overlap.native_surfaces(payloads) if s.name == "diff"]
self.assertEqual(scored.lane, "builtin_commands")
index = overlap.build_native_index({}, payloads)
self.assertEqual(index["diff"]["lane"], scored.lane)

def test_the_index_holds_the_surface_scored_first_in_every_lane_mix(
self,
) -> None:
entries = {
"valid": [{"name": "x", "description": "X"}],
"internal": [{"name": "x", "description": "X", "internal": True}],
}
lanes = [lane for lane in overlap.LANE_ORDER if lane != "plugin_backed"]
for states in itertools.product((None, "valid", "internal"), repeat=len(lanes)):
present = {
lane: {"x": entries[state]}
for lane, state in zip(lanes, states, strict=True)
if state is not None
}
if not present:
continue
with self.subTest(
present={lane: s for lane, s in zip(lanes, states, strict=True) if s}
):
index = overlap.build_native_index({}, present)
scored = [s for s in overlap.native_surfaces(present) if s.name == "x"]
expected = scored[0].lane if scored else next(iter(present))
self.assertEqual(index["x"]["lane"], expected)

def test_an_internal_plugin_backed_name_gets_no_fallback_surface(self) -> None:
payloads = overlap._lane_payloads(
{
Expand Down
Loading