From 467fe9065b64af000701267e6086fceb0eb4c9ca Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:31:20 -0400 Subject: [PATCH 1/9] feat(disk-hygiene): add deep-inventory categories and KEEP-reason validator Read-only stdlib module with the shared row schema (name, ext, size, mtime, owner, producer, category, disposition, reason, evidence) and named categories: superseded versioned dirs, unreferenced plugin cache versions, /tmp by producer prefix, transcript dirs whose source path is gone, and dangling symlinks. validate_report fails a KEEP whose reason is empty or only a category phrase unless its evidence shows the named tool still references the entry. Refs #5221 Co-Authored-By: Claude Opus 5.5 --- .../skills/clean/scripts/deep_inventory.py | 563 ++++++++++++++++++ .../clean/scripts/deep_inventory.test.sh | 33 + .../clean/scripts/test_deep_inventory.py | 469 +++++++++++++++ 3 files changed, 1065 insertions(+) create mode 100644 plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py create mode 100755 plugins/disk-hygiene/skills/clean/scripts/deep_inventory.test.sh create mode 100644 plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py diff --git a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py new file mode 100644 index 0000000000..0aed9457fc --- /dev/null +++ b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py @@ -0,0 +1,563 @@ +#!/usr/bin/env python3 +"""Deep-inventory categories and KEEP-reason validator for ``/disk-hygiene:clean``. + +Read-only. Every function takes its paths, clock and process table as arguments +and only reads; nothing here deletes, moves or writes. A candidate row is a +finding, not a deletion plan: removal stays behind the engine's own gates. + +Row schema (one dict per inventoried entry, shared by every producer of a +machine listing): + + name absolute path of the entry + ext file suffix, "" for a directory + size bytes; a directory is the sum of the files beneath it, links not followed + mtime modification time, ISO-8601 UTC + owner owning user name, or the numeric uid when it has no name + producer the tool that made the entry ("unknown" when no rule attributes it) + category the named category that produced the row + disposition KEEP | CANDIDATE | UNKNOWN + reason who produced it, what uses it, why it stays or may go + evidence optional dict of proof. {"tool": ..., "references": ...} shows the named + tool still references the entry, which is what lets a KEEP rest on a + category phrase such as "managed by " + +``validate_report`` fails a report whose KEEP row has an empty reason, or one +that is only a category phrase without such evidence. +""" + +from __future__ import annotations + +import datetime as dt +import json +import os +import re +from collections.abc import Iterable +from pathlib import Path +from typing import Any + +try: + import pwd +except ImportError: # Windows + pwd = None + +ROW_COLUMNS = ( + "name", + "ext", + "size", + "mtime", + "owner", + "producer", + "category", + "disposition", + "reason", + "evidence", +) +DISPOSITIONS = ("KEEP", "CANDIDATE", "UNKNOWN") +DAY = 86400.0 +# A directory's mtime moves only when its direct children change, so an entry +# untouched for this long may still be in use; the running-process check and +# the reader's judgment cover that gap. +TMP_MIN_AGE_DAYS = 7.0 +# (name prefix, producer, reason the entry must stay or None when age decides). +TMP_PRODUCERS = ( + ("pytest-of-", "pytest", None), + ("claude-", "claude-code", None), + ("npm-", "npm", None), + ("uv-", "uv", None), + ("playwright", "playwright", None), + ("codex-", "codex", None), + ("cursor-", "cursor-agent", None), + ( + "systemd-private-", + "systemd", + "private /tmp of a service unit; removing it breaks the running unit until it restarts", + ), + ( + ".X11-unix", + "xorg", + "socket directory of the running X server; clients cannot connect without it", + ), + ("ssh-", "ssh-agent", "socket directory of a running ssh-agent"), +) +# Substring of a dangling link's path -> producer. +LINK_PRODUCERS = ( + ("/mise/", "mise"), + ("/.claude/", "claude-code"), + ("/.npm/", "npm"), + ("/uv/", "uv"), +) +VERSION_RE = re.compile(r"^v?(\d+(?:\.\d+)*)(?:[-+][\w.+-]+)?$") +_TOKEN = r"[\w.@/+-]+" +_CATEGORY_PHRASE = re.compile( + rf"(?:(?:tool|os|system|app|vendor)[- ]?(?:managed|owned)" + rf"|(?:managed|owned) by (?:the )?{_TOKEN}(?: {_TOKEN}){{0,2}})" +) +_MANAGED_BY = re.compile(r"managed by ") + + +def _iso(epoch: float) -> str: + return dt.datetime.fromtimestamp(epoch, dt.timezone.utc).isoformat( + timespec="seconds" + ) + + +def _owner(uid: int) -> str: + if pwd is not None: + try: + return pwd.getpwuid(uid).pw_name + except KeyError: + pass + return str(uid) + + +def _tree_size(path: Path) -> int: + total, stack = 0, [path] + while stack: + current = stack.pop() + try: + with os.scandir(current) as entries: + for entry in entries: + if entry.is_dir(follow_symlinks=False): + stack.append(Path(entry.path)) + else: + try: + total += entry.stat(follow_symlinks=False).st_size + except OSError: + pass + except OSError: + pass + return total + + +def make_row( + path: Path, + *, + producer: str, + category: str, + disposition: str, + reason: str, + evidence: dict[str, Any] | None = None, +) -> dict[str, Any]: + """One schema row for ``path``, measured with lstat so a link is never followed.""" + st = os.lstat(path) + is_dir = os.path.isdir(path) and not os.path.islink(path) + row: dict[str, Any] = { + "name": str(path), + "ext": "" if is_dir else path.suffix, + "size": _tree_size(path) if is_dir else st.st_size, + "mtime": _iso(st.st_mtime), + "owner": _owner(st.st_uid), + "producer": producer, + "category": category, + "disposition": disposition, + "reason": reason, + } + if evidence is not None: + row["evidence"] = evidence + return row + + +def _child_dirs(parent: Path) -> list[Path]: + try: + with os.scandir(parent) as entries: + return sorted( + Path(e.path) for e in entries if e.is_dir(follow_symlinks=False) + ) + except OSError: + return [] + + +def running_paths(proc_root: Path = Path("/proc")) -> set[str]: + """Resolved executable and working-directory targets of every process under ``proc_root``. + + A deleted executable reads `` (deleted)``; the suffix is dropped so the + superseded directory it came from still matches. + """ + found: set[str] = set() + try: + pids = [p for p in proc_root.iterdir() if p.name.isdigit()] + except OSError: + return found + for pid in pids: + for link in ("exe", "cwd"): + try: + found.add(os.readlink(pid / link).removesuffix(" (deleted)")) + except OSError: + pass + return found + + +def _in_use(entry: Path, running: Iterable[str]) -> str | None: + prefix = str(entry) + return next( + (r for r in running if r == prefix or r.startswith(prefix + os.sep)), None + ) + + +def superseded_versions( + parents: Iterable[Path], running: Iterable[str] = () +) -> list[dict[str, Any]]: + """Sibling entries under each parent whose names parse as versions. + + Keeps the newest and any version a running process executes; the rest are + candidates. A parent with fewer than two version entries yields no rows. + """ + rows: list[dict[str, Any]] = [] + running = set(running) + for parent in parents: + versions: list[tuple[tuple[int, ...], str, Path]] = [] + try: + with os.scandir(parent) as entries: + for entry in entries: + match = VERSION_RE.match(entry.name) + if match and not entry.is_symlink(): + key = tuple(int(n) for n in match.group(1).split(".")) + versions.append((key, entry.name, Path(entry.path))) + except OSError: + continue + if len(versions) < 2: + continue + versions.sort() + newest = versions[-1][1] + holder = ( + parent.parent.name + if parent.name in {"versions", "releases", "bin"} + else parent.name + ) + for _, name, path in versions: + used = _in_use(path, running) + if name == newest: + disposition, reason, evidence = ( + "KEEP", + f"newest of {len(versions)} versions beside it in {parent}", + None, + ) + elif used: + disposition, reason, evidence = ( + "KEEP", + f"a running process executes {used}", + {"running": used}, + ) + else: + disposition, reason, evidence = ( + "CANDIDATE", + f"superseded by {newest}; no running process executes it", + None, + ) + rows.append( + make_row( + path, + producer=holder, + category="superseded-version", + disposition=disposition, + reason=reason, + evidence=evidence, + ) + ) + return rows + + +def _install_paths(data: object) -> list[str]: + if isinstance(data, dict): + own = data.get("installPath") + found = [own] if isinstance(own, str) else [] + return found + [p for v in data.values() for p in _install_paths(v)] + if isinstance(data, list): + return [p for v in data for p in _install_paths(v)] + return [] + + +def _resolve(path: str | Path) -> Path | None: + try: + return Path(path).expanduser().resolve() + except (OSError, RuntimeError): + return None + + +def plugin_cache_versions(claude_dir: Path) -> list[dict[str, Any]]: + """Rows for ``plugins/cache///`` under ``claude_dir``. + + A version some ``installPath`` in ``plugins/installed_plugins.json`` names is + KEEP with that registry as evidence; one no path names is a candidate. When + the registry cannot vouch for this cache (unreadable, no ``plugins`` object, + or no path in it under this cache) every row is UNKNOWN: a missing registry + is not evidence that a version is unreferenced. + """ + cache = claude_dir / "plugins" / "cache" + paths = [ + version + for marketplace in _child_dirs(cache) + for plugin in _child_dirs(marketplace) + for version in _child_dirs(plugin) + ] + if not paths: + return [] + registry = claude_dir / "plugins" / "installed_plugins.json" + referenced: set[Path] = set() + doubt = "" + try: + data = json.loads(registry.read_text(encoding="utf-8")) + except (OSError, ValueError) as exc: + data, doubt = None, f"{registry.name} unreadable ({type(exc).__name__})" + if not doubt and not ( + isinstance(data, dict) and isinstance(data.get("plugins"), dict) + ): + doubt = f"{registry.name} has no `plugins` object" + if not doubt: + referenced = {p for p in map(_resolve, _install_paths(data)) if p is not None} + cache_resolved = _resolve(cache) + if data["plugins"] and not any(cache_resolved in p.parents for p in referenced): + doubt = f"no installPath in {registry.name} lies under {cache}" + rows = [] + for path in paths: + producer = path.parent.name + common = {"producer": producer, "category": "plugin-cache-version"} + if doubt: + rows.append( + make_row( + path, + disposition="UNKNOWN", + reason=f"{doubt}; the registry cannot show whether this version is installed", + **common, + ) + ) + elif _resolve(path) in referenced: + rows.append( + make_row( + path, + disposition="KEEP", + reason=f"an installPath in {registry.name} points at this installed version", + evidence={"tool": "claude-code", "references": str(registry)}, + **common, + ) + ) + else: + rows.append( + make_row( + path, + disposition="CANDIDATE", + reason=f"no installPath in {registry.name} references this version", + **common, + ) + ) + return rows + + +def tmp_entries( + tmp_dir: Path, + now: float, + running: Iterable[str] = (), + min_age_days: float = TMP_MIN_AGE_DAYS, + producers: tuple[tuple[str, str, str | None], ...] = TMP_PRODUCERS, +) -> list[dict[str, Any]]: + """One row per top-level entry of ``tmp_dir``, attributed by producer name prefix. + + An unattributed entry is UNKNOWN. An attributed one stays when its producer + rule names a reason, when a running process uses it, or when it changed + inside ``min_age_days``; otherwise it is a candidate. + """ + rows: list[dict[str, Any]] = [] + running = set(running) + try: + entries = sorted(tmp_dir.iterdir()) + except OSError: + return rows + for path in entries: + match = next((p for p in producers if path.name.startswith(p[0])), None) + try: + age = (now - os.lstat(path).st_mtime) / DAY + used = _in_use(path, running) + except OSError: + continue + common = {"category": "tmp-producer"} + if match is None: + row = make_row( + path, + producer="unknown", + disposition="UNKNOWN", + reason="no producer prefix rule matches this name", + **common, + ) + else: + _, producer, keep_reason = match + if keep_reason: + verdict = ("KEEP", keep_reason) + elif used: + verdict = ("KEEP", f"a running process uses {used}") + elif age < min_age_days: + verdict = ( + "KEEP", + f"changed {age:.1f} days ago, inside the {min_age_days:g}-day window " + f"in which a live {producer} run may still use it", + ) + else: + verdict = ( + "CANDIDATE", + f"{producer} leftover unchanged for {age:.0f} days; no running process uses it", + ) + row = make_row( + path, + producer=producer, + disposition=verdict[0], + reason=verdict[1], + **common, + ) + rows.append(row) + return rows + + +def _encode_project(path: str) -> str: + return re.sub(r"[^A-Za-z0-9]", "-", path) + + +def decode_project(encoded: str, fs_root: Path = Path("/")) -> str | None: + """The existing path under ``fs_root`` whose encoding is ``encoded``, or None. + + The encoding replaces every non-alphanumeric character with ``-``, so a name + has no unique decoding; the directory tree settles which reading exists. + """ + + def walk(base: Path, rest: str) -> Path | None: + try: + names = os.listdir(base) + except OSError: + return None + for name in names: + code = _encode_project(name) + if rest == code: + return base / name + if rest.startswith(code + "-") and (base / name).is_dir(): + found = walk(base / name, rest[len(code) + 1 :]) + if found: + return found + return None + + found = walk(fs_root, encoded[1:]) + return None if found is None else "/" + found.relative_to(fs_root).as_posix() + + +# Claude Code caps an encoded directory name near this length and appends a hash, +# after which the source path cannot be recovered from the name. +PROJECT_NAME_CAP = 200 + + +def project_transcripts( + projects_dir: Path, fs_root: Path = Path("/") +) -> list[dict[str, Any]]: + """Rows for ``~/.claude/projects/``; a source path that is gone makes a candidate.""" + rows = [] + for path in _child_dirs(projects_dir): + name = path.name + common = {"producer": "claude-code", "category": "transcript-dir"} + if not name.startswith("-") or len(name) > PROJECT_NAME_CAP: + rows.append( + make_row( + path, + disposition="UNKNOWN", + reason="the name is not a decodable POSIX path encoding", + **common, + ) + ) + continue + source = decode_project(name, fs_root) + if source is None: + rows.append( + make_row( + path, + disposition="CANDIDATE", + reason="transcripts of a project whose source path no longer exists", + **common, + ) + ) + else: + rows.append( + make_row( + path, + disposition="KEEP", + reason=f"transcripts of {source}, which still exists", + evidence={"tool": "claude-code", "references": source}, + **common, + ) + ) + return rows + + +def dangling_symlinks( + roots: Iterable[Path], max_depth: int = 8 +) -> list[dict[str, Any]]: + """Symlinks under ``roots`` (to ``max_depth`` levels) whose target does not exist.""" + rows: list[dict[str, Any]] = [] + for root in roots: + base_depth = len(root.parts) + for current, dirs, files in os.walk(root, followlinks=False): + if len(Path(current).parts) - base_depth >= max_depth: + dirs[:] = [] + for name in dirs + files: + path = Path(current) / name + if not path.is_symlink() or path.exists(): + continue + producer = next( + (p for hint, p in LINK_PRODUCERS if hint in path.as_posix()), + "unknown", + ) + rows.append( + make_row( + path, + producer=producer, + category="dangling-symlink", + disposition="CANDIDATE", + reason=f"points to {os.readlink(path)}, which does not exist", + ) + ) + return rows + + +def _category_only(reason: str) -> bool: + parts = [ + p for p in re.split(r"\s*(?:[;,.]|\band\b)\s*", reason.strip().lower()) if p + ] + return not parts or all(_CATEGORY_PHRASE.fullmatch(p) for p in parts) + + +def _shows_reference(reason: str, evidence: object) -> bool: + if not ( + isinstance(evidence, dict) + and str(evidence.get("tool") or "").strip() + and str(evidence.get("references") or "").strip() + ): + return False + return not _MANAGED_BY.search(reason.lower()) or ( + str(evidence["tool"]).strip().lower() in reason.lower() + ) + + +def validate_report(rows: Iterable[dict[str, Any]]) -> list[str]: + """Failures of a report; an empty list means it passes. + + Every row needs the schema columns except the optional ``evidence`` and a + valid disposition. A KEEP row fails when its reason is empty or only a + category phrase ("tool-managed", "OS-owned", "managed by ") unless its + evidence shows the named tool still references the entry. + """ + failures = [] + for index, row in enumerate(rows): + label = str(row.get("name") or f"row {index}") + missing = [c for c in ROW_COLUMNS if c != "evidence" and c not in row] + if missing: + failures.append(f"{label}: missing columns {', '.join(missing)}") + continue + if row["disposition"] not in DISPOSITIONS: + failures.append( + f"{label}: disposition {row['disposition']!r} is not one of {DISPOSITIONS}" + ) + continue + if row["disposition"] != "KEEP": + continue + reason = str(row["reason"] or "") + if _category_only(reason) and not _shows_reference(reason, row.get("evidence")): + failures.append( + f"{label}: KEEP reason {reason!r} is " + f"{'empty' if not reason.strip() else 'only a category phrase'} " + "and no evidence shows the named tool still references the entry" + ) + return failures diff --git a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.test.sh b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.test.sh new file mode 100755 index 0000000000..43772bd531 --- /dev/null +++ b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.test.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash +# Cross-platform contract wrapper for the deep-inventory test suite. +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +# shellcheck source=../../../scripts/test-wrapper-lib.sh +source "$SCRIPT_DIR/../../../scripts/test-wrapper-lib.sh" + +ENGINE="$SCRIPT_DIR/hygiene.py" +FLOOR="" +test_wrapper::floor_to FLOOR "$ENGINE" +if [[ -z "$FLOOR" ]]; then + echo "FAIL: could not parse MIN_PYTHON from $ENGINE" >&2 + exit 1 +fi + +PYTHON="" +test_wrapper::interpreter_to PYTHON +if [[ -z "$PYTHON" ]]; then + echo "SKIP: Python ${FLOOR}+ not found" >&2 + exit 0 +fi + +FLOOR_CHECK="" +test_wrapper::floor_check_to FLOOR_CHECK "$FLOOR" +"$PYTHON" -c "$FLOOR_CHECK" || { + echo "SKIP: Python ${FLOOR}+ required" >&2 + exit 0 +} +PYFILE="" +test_wrapper::python_file_to PYFILE "$SCRIPT_DIR/test_deep_inventory.py" +"$PYTHON" -m unittest -v "$PYFILE" diff --git a/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py b/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py new file mode 100644 index 0000000000..7c53c884ec --- /dev/null +++ b/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py @@ -0,0 +1,469 @@ +#!/usr/bin/env python3 +"""Tests for the read-only deep-inventory categories and KEEP-reason validator.""" + +from __future__ import annotations + +import json +import os +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT_DIR = Path(__file__).resolve().parent +sys.path.insert(0, str(SCRIPT_DIR)) + +import deep_inventory as di # noqa: E402 (path set above) + +NOW = 1_800_000_000.0 + + +class TempTree(unittest.TestCase): + def setUp(self) -> None: + temporary = tempfile.TemporaryDirectory() + self.addCleanup(temporary.cleanup) + self.root = Path(temporary.name).resolve() + + def write(self, relative: str, text: str = "x", age_days: float = 0.0) -> Path: + path = self.root / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text, encoding="utf-8") + self.age(path, age_days) + return path + + def mkdir(self, relative: str, age_days: float = 0.0) -> Path: + path = self.root / relative + path.mkdir(parents=True, exist_ok=True) + self.age(path, age_days) + return path + + @staticmethod + def age(path: Path, age_days: float) -> None: + stamp = NOW - age_days * di.DAY + os.utime(path, (stamp, stamp), follow_symlinks=False) + + +def by_name(rows: list[dict]) -> dict[str, dict]: + return {Path(r["name"]).name: r for r in rows} + + +class RowSchemaTest(TempTree): + def test_row_carries_every_column_and_measures_a_directory(self) -> None: + directory = self.mkdir("d") + (directory / "a.bin").write_bytes(b"12345") + (directory / "sub").mkdir() + (directory / "sub" / "b.bin").write_bytes(b"678") + row = di.make_row( + directory, producer="p", category="c", disposition="KEEP", reason="r" + ) + self.assertEqual(set(row), set(di.ROW_COLUMNS) - {"evidence"}) + self.assertEqual((row["size"], row["ext"]), (8, "")) + self.assertEqual(di.validate_report([row]), []) + + def test_file_row_reports_extension_and_evidence(self) -> None: + path = self.write("f.tar.gz", "abc") + row = di.make_row( + path, + producer="p", + category="c", + disposition="UNKNOWN", + reason="r", + evidence={"k": 1}, + ) + self.assertEqual( + (row["ext"], row["size"], row["evidence"]), (".gz", 3, {"k": 1}) + ) + + +class ValidatorTest(unittest.TestCase): + @staticmethod + def row(disposition: str, reason: str, evidence=None) -> dict: + row = {c: "x" for c in di.ROW_COLUMNS if c != "evidence"} + row.update(disposition=disposition, reason=reason) + if evidence is not None: + row["evidence"] = evidence + return row + + def failures( + self, reason: str, evidence=None, disposition: str = "KEEP" + ) -> list[str]: + return di.validate_report([self.row(disposition, reason, evidence)]) + + def test_specific_keep_reason_passes(self) -> None: + self.assertEqual( + self.failures("codex 1.4 is the running binary; pid 42 executes it"), [] + ) + + def test_empty_keep_reason_fails(self) -> None: + for reason in ("", " ", None): + with self.subTest(reason=reason): + self.assertEqual(len(self.failures(reason)), 1) + + def test_category_only_reasons_fail(self) -> None: + for reason in ( + "tool-managed", + "Tool managed.", + "OS-owned", + "os owned", + "managed by mise", + "Managed by the claude code", + "tool-managed; OS-owned", + "OS-owned and tool-managed", + ): + with self.subTest(reason=reason): + failures = self.failures(reason) + self.assertEqual(len(failures), 1) + self.assertIn("category phrase", failures[0]) + + def test_category_phrase_with_more_detail_passes(self) -> None: + for reason in ( + "managed by mise; trusted-config link for /repo, target exists", + "OS-owned socket the running X server accepts clients on", + ): + with self.subTest(reason=reason): + self.assertEqual(self.failures(reason), []) + + def test_evidence_that_the_tool_references_the_entry_rescues_a_category_reason( + self, + ) -> None: + evidence = {"tool": "mise", "references": "trusted-configs/abc"} + self.assertEqual(self.failures("managed by mise", evidence), []) + self.assertEqual(self.failures("tool-managed", evidence), []) + + def test_evidence_for_a_different_tool_does_not_rescue(self) -> None: + evidence = {"tool": "npm", "references": "package.json"} + self.assertEqual(len(self.failures("managed by mise", evidence)), 1) + + def test_incomplete_evidence_does_not_rescue(self) -> None: + for evidence in ( + {"tool": "mise"}, + {"references": "x"}, + {"tool": "", "references": ""}, + "mise", + ): + with self.subTest(evidence=evidence): + self.assertEqual(len(self.failures("tool-managed", evidence)), 1) + + def test_only_keep_rows_need_a_reason(self) -> None: + for disposition in ("CANDIDATE", "UNKNOWN"): + self.assertEqual(self.failures("", disposition=disposition), []) + + def test_bad_disposition_and_missing_columns_fail(self) -> None: + self.assertEqual(len(self.failures("r", disposition="DELETE")), 1) + self.assertEqual( + len(di.validate_report([{"name": "n", "disposition": "KEEP"}])), 1 + ) + + def test_every_failure_is_reported(self) -> None: + rows = [ + self.row("KEEP", ""), + self.row("KEEP", "OS-owned"), + self.row("KEEP", "ok reason here"), + ] + self.assertEqual(len(di.validate_report(rows)), 2) + + +class RunningPathsTest(TempTree): + def test_reads_exe_and_cwd_and_drops_the_deleted_suffix(self) -> None: + proc = self.mkdir("proc") + for pid, exe, cwd in ( + ("10", "/opt/tool/1.0/bin (deleted)", "/home/u"), + ("11", "/opt/tool/2.0/bin", "/srv"), + ): + (proc / pid).mkdir() + os.symlink(exe, proc / pid / "exe") + os.symlink(cwd, proc / pid / "cwd") + (proc / "self-note").mkdir() + self.assertEqual( + di.running_paths(proc), + {"/opt/tool/1.0/bin", "/home/u", "/opt/tool/2.0/bin", "/srv"}, + ) + + def test_missing_proc_root_is_empty(self) -> None: + self.assertEqual(di.running_paths(self.root / "absent"), set()) + + +class SupersededVersionsTest(TempTree): + def setUp(self) -> None: + super().setUp() + self.parent = self.mkdir("share/codex/versions") + for name in ("1.9.0", "1.10.0", "1.2.0", "v1.10.1"): + (self.parent / name).mkdir() + (self.parent / name / "bin").write_text("x", encoding="utf-8") + (self.parent / "current").mkdir() + (self.parent / "notes.txt").write_text("x", encoding="utf-8") + + def test_keeps_newest_by_numeric_order_and_marks_the_rest(self) -> None: + rows = by_name(di.superseded_versions([self.parent])) + self.assertEqual(set(rows), {"1.9.0", "1.10.0", "1.2.0", "v1.10.1"}) + self.assertEqual(rows["v1.10.1"]["disposition"], "KEEP") + for name in ("1.9.0", "1.10.0", "1.2.0"): + self.assertEqual(rows[name]["disposition"], "CANDIDATE", name) + self.assertEqual(rows["1.2.0"]["producer"], "codex") + self.assertEqual(rows["1.2.0"]["category"], "superseded-version") + + def test_a_version_a_running_process_executes_is_kept(self) -> None: + exe = str(self.parent / "1.9.0" / "bin") + rows = by_name(di.superseded_versions([self.parent], {exe, "/elsewhere"})) + self.assertEqual(rows["1.9.0"]["disposition"], "KEEP") + self.assertEqual(rows["1.9.0"]["evidence"], {"running": exe}) + self.assertEqual(rows["1.10.0"]["disposition"], "CANDIDATE") + + def test_every_keep_passes_the_validator(self) -> None: + exe = str(self.parent / "1.2.0") + rows = di.superseded_versions([self.parent], {exe}) + self.assertEqual(di.validate_report(rows), []) + + def test_a_lone_version_yields_no_rows(self) -> None: + lone = self.mkdir("share/solo") + (lone / "3.0.0").mkdir() + self.assertEqual(di.superseded_versions([lone, self.root / "absent"]), []) + + def test_file_versions_and_symlinks(self) -> None: + parent = self.mkdir("share/claude/versions") + (parent / "2.1.283").write_text("x", encoding="utf-8") + (parent / "2.1.284").write_text("x", encoding="utf-8") + os.symlink(parent / "2.1.284", parent / "9.9.9") + rows = by_name(di.superseded_versions([parent])) + self.assertEqual(set(rows), {"2.1.283", "2.1.284"}) + self.assertEqual(rows["2.1.283"]["producer"], "claude") + + +class PluginCacheTest(TempTree): + def registry(self, install_paths: list[str], plugins: dict | None = None) -> None: + data = plugins + if data is None: + data = { + f"p{i}@m": [{"scope": "user", "installPath": p}] + for i, p in enumerate(install_paths) + } + path = self.root / "plugins" / "installed_plugins.json" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps({"version": 2, "plugins": data}), encoding="utf-8") + + def cache(self, *versions: str) -> None: + for v in versions: + (self.mkdir(f"plugins/cache/{v}") / "plugin.json").write_text( + "{}", encoding="utf-8" + ) + + def test_referenced_version_is_kept_with_evidence_and_the_rest_are_candidates( + self, + ) -> None: + self.cache("mkt/alpha/1.0.0", "mkt/alpha/1.1.0", "mkt/beta/0.3.0") + self.registry( + [ + str(self.root / "plugins/cache/mkt/alpha/1.1.0"), + str(self.root / "plugins/cache/mkt/beta/0.3.0"), + ] + ) + rows = di.plugin_cache_versions(self.root) + by_path = {Path(r["name"]).relative_to(self.root).as_posix(): r for r in rows} + old = by_path["plugins/cache/mkt/alpha/1.0.0"] + kept = by_path["plugins/cache/mkt/alpha/1.1.0"] + self.assertEqual(old["disposition"], "CANDIDATE") + self.assertEqual(old["producer"], "alpha") + self.assertEqual(kept["disposition"], "KEEP") + self.assertEqual(kept["evidence"]["tool"], "claude-code") + self.assertEqual(by_path["plugins/cache/mkt/beta/0.3.0"]["disposition"], "KEEP") + self.assertEqual(di.validate_report(rows), []) + + def test_nested_install_paths_count(self) -> None: + self.cache("mkt/alpha/1.0.0") + self.registry( + [], + { + "alpha@mkt": [ + { + "scope": "project", + "meta": { + "installPath": str( + self.root / "plugins/cache/mkt/alpha/1.0.0" + ) + }, + } + ] + }, + ) + self.assertEqual(di.plugin_cache_versions(self.root)[0]["disposition"], "KEEP") + + def test_registry_that_cannot_vouch_leaves_every_version_unknown(self) -> None: + self.cache("mkt/alpha/1.0.0") + for prepare in ( + lambda: None, + lambda: (self.root / "plugins" / "installed_plugins.json").write_text( + "{", encoding="utf-8" + ), + lambda: (self.root / "plugins" / "installed_plugins.json").write_text( + "[]", encoding="utf-8" + ), + lambda: self.registry(["/somewhere/else/plugins/cache/mkt/alpha/1.0.0"]), + ): + with self.subTest(): + prepare() + rows = di.plugin_cache_versions(self.root) + self.assertEqual([r["disposition"] for r in rows], ["UNKNOWN"]) + + def test_empty_registry_leaves_every_version_a_candidate(self) -> None: + self.cache("mkt/alpha/1.0.0") + self.registry([], {}) + self.assertEqual( + di.plugin_cache_versions(self.root)[0]["disposition"], "CANDIDATE" + ) + + def test_no_cache_yields_no_rows(self) -> None: + self.assertEqual(di.plugin_cache_versions(self.root), []) + + +class TmpEntriesTest(TempTree): + def setUp(self) -> None: + super().setUp() + self.tmp = self.mkdir("tmp") + self.mkdir("tmp/pytest-of-kyle", age_days=30) + self.mkdir("tmp/pytest-of-old", age_days=2) + self.write("tmp/claude-1000-cache", age_days=40) + self.mkdir("tmp/systemd-private-abc-svc", age_days=90) + self.write("tmp/mystery.log", age_days=90) + + def rows(self, running=()) -> dict[str, dict]: + return by_name(di.tmp_entries(self.tmp, NOW, running)) + + def test_old_attributed_entries_are_candidates_by_producer(self) -> None: + rows = self.rows() + self.assertEqual(rows["pytest-of-kyle"]["disposition"], "CANDIDATE") + self.assertEqual(rows["pytest-of-kyle"]["producer"], "pytest") + self.assertEqual(rows["claude-1000-cache"]["producer"], "claude-code") + self.assertEqual(rows["claude-1000-cache"]["category"], "tmp-producer") + + def test_recent_entry_is_kept_with_a_reason_naming_the_window(self) -> None: + row = self.rows()["pytest-of-old"] + self.assertEqual(row["disposition"], "KEEP") + self.assertIn("7-day window", row["reason"]) + + def test_producer_rule_reason_keeps_regardless_of_age(self) -> None: + row = self.rows()["systemd-private-abc-svc"] + self.assertEqual((row["disposition"], row["producer"]), ("KEEP", "systemd")) + + def test_unattributed_entry_is_unknown(self) -> None: + row = self.rows()["mystery.log"] + self.assertEqual((row["disposition"], row["producer"]), ("UNKNOWN", "unknown")) + + def test_a_running_process_inside_an_old_entry_keeps_it(self) -> None: + cwd = str(self.tmp / "pytest-of-kyle" / "run-1") + self.assertEqual(self.rows({cwd})["pytest-of-kyle"]["disposition"], "KEEP") + # a sibling whose name only shares the prefix is not the same entry + self.assertEqual( + self.rows({str(self.tmp / "pytest-of-kyle-2")})["pytest-of-kyle"][ + "disposition" + ], + "CANDIDATE", + ) + + def test_every_keep_passes_the_validator(self) -> None: + self.assertEqual(di.validate_report(di.tmp_entries(self.tmp, NOW)), []) + + def test_missing_tmp_dir_is_empty(self) -> None: + self.assertEqual(di.tmp_entries(self.root / "absent", NOW), []) + + +class ProjectTranscriptsTest(TempTree): + def setUp(self) -> None: + super().setUp() + self.fs = self.root / "fs" + self.mkdir("fs/home/kyle/my.repo/sub") + self.mkdir("fs/home/kyle/.config") + self.projects = self.mkdir("claude/projects") + + def rows(self, *names: str) -> dict[str, dict]: + for name in names: + self.mkdir(f"claude/projects/{name}") + return by_name(di.project_transcripts(self.projects, self.fs)) + + def test_decodes_by_the_directory_tree_not_by_guessing_separators(self) -> None: + rows = self.rows( + "-home-kyle-my-repo-sub", "-home-kyle--config", "-home-kyle-my-repo" + ) + for name, source in ( + ("-home-kyle-my-repo-sub", "/home/kyle/my.repo/sub"), + ("-home-kyle--config", "/home/kyle/.config"), + ("-home-kyle-my-repo", "/home/kyle/my.repo"), + ): + self.assertEqual(rows[name]["disposition"], "KEEP", name) + self.assertEqual(rows[name]["evidence"]["references"], source) + self.assertEqual(di.validate_report(rows.values()), []) + + def test_a_gone_source_path_is_a_candidate(self) -> None: + rows = self.rows("-tmp-harness-run-7", "-home-kyle-deleted-repo") + for name in ("-tmp-harness-run-7", "-home-kyle-deleted-repo"): + self.assertEqual(rows[name]["disposition"], "CANDIDATE", name) + self.assertEqual(rows[name]["producer"], "claude-code") + + def test_a_prefix_only_match_is_not_a_source(self) -> None: + rows = self.rows("-home-kyle-my-repo-sub-gone") + self.assertEqual( + rows["-home-kyle-my-repo-sub-gone"]["disposition"], "CANDIDATE" + ) + + def test_undecodable_names_are_unknown(self) -> None: + rows = self.rows("C--Users-kyle", "-" + "a" * di.PROJECT_NAME_CAP) + self.assertEqual({r["disposition"] for r in rows.values()}, {"UNKNOWN"}) + + def test_decode_project_returns_none_for_missing(self) -> None: + self.assertIsNone(di.decode_project("-nope", self.fs)) + self.assertEqual( + di.decode_project("-home-kyle-my-repo", self.fs), "/home/kyle/my.repo" + ) + + +class DanglingSymlinksTest(TempTree): + def test_reports_only_links_whose_target_is_gone(self) -> None: + home = self.mkdir("home") + live = self.write("home/real.toml") + trusted = self.mkdir("home/.local/state/mise/trusted-configs") + os.symlink(live, trusted / "ok") + os.symlink(self.root / "gone" / "mise.toml", trusted / "broken") + os.symlink(self.root / "gone", home / "loose") + rows = by_name(di.dangling_symlinks([home])) + self.assertEqual(set(rows), {"broken", "loose"}) + self.assertEqual(rows["broken"]["producer"], "mise") + self.assertEqual(rows["loose"]["producer"], "unknown") + self.assertEqual(rows["broken"]["disposition"], "CANDIDATE") + self.assertIn("does not exist", rows["broken"]["reason"]) + self.assertEqual(di.validate_report(rows.values()), []) + + def test_depth_limit_and_links_are_not_followed(self) -> None: + deep = self.mkdir("root/a/b/c") + os.symlink(self.root / "gone", deep / "far") + os.symlink(self.root / "root", self.root / "root" / "a" / "loop") + self.assertEqual(di.dangling_symlinks([self.root / "root"], max_depth=2), []) + self.assertEqual( + len(di.dangling_symlinks([self.root / "root"], max_depth=8)), 1 + ) + + +class ReadOnlyTest(TempTree): + def test_categories_leave_the_tree_unchanged(self) -> None: + self.mkdir("tmp/pytest-of-x", age_days=30) + self.mkdir("v/1.0.0") + self.mkdir("v/2.0.0") + self.mkdir("claude/projects/-gone") + self.mkdir("claude/plugins/cache/m/p/1.0.0") + os.symlink(self.root / "nowhere", self.root / "tmp" / "dangling") + + def snapshot() -> list[tuple[str, float]]: + return sorted( + (str(p), os.lstat(p).st_mtime) + for p in [self.root, *self.root.rglob("*")] + ) + + before = snapshot() + di.tmp_entries(self.root / "tmp", NOW) + di.superseded_versions([self.root / "v"]) + di.project_transcripts(self.root / "claude" / "projects", self.root) + di.plugin_cache_versions(self.root / "claude") + di.dangling_symlinks([self.root]) + self.assertEqual(snapshot(), before) + + +if __name__ == "__main__": + unittest.main() From d91385654f51768a049bae319e8df04336e4ee67 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:41:41 -0400 Subject: [PATCH 2/9] feat(disk-hygiene): add read-only inventory subcommand with deep mode Declare an `inventory` subcommand in engine_grammar (--target, --data-root, optional --deep), so the engine parses it and the guard admits it from one declaration; apply and preview grammar are unchanged. The guard allows it alongside scan, preview and handoff-verify through one named read-only set. Inventory streams one row per entry to a JSONL report under the data root, with no entry cap. Deep mode lists every level with bottom-up directory sizes and is the default when the target is the home directory; otherwise only immediate children are listed. Category rows (plugin cache versions, transcript dirs, /tmp producers, dotted superseded versions, dangling links) replace the unclassified row at their path; an unreadable or mounted subtree is one UNKNOWN row. Each row runs through validate_report; a failure marks the summary inventory-failed and exits 5. The summary is neither a snapshot nor a plan, and preview refuses it. Refs #5221 Co-Authored-By: Claude Opus 5.5 --- plugins/disk-hygiene/lib/engine_grammar.py | 20 ++ plugins/disk-hygiene/skills/clean/SKILL.md | 4 +- .../skills/clean/scripts/deep_inventory.py | 201 ++++++++++++++-- .../skills/clean/scripts/destructive_guard.py | 10 +- .../skills/clean/scripts/hygiene.py | 71 ++++++ .../skills/clean/scripts/test_hygiene.py | 224 ++++++++++++++++++ 6 files changed, 508 insertions(+), 22 deletions(-) diff --git a/plugins/disk-hygiene/lib/engine_grammar.py b/plugins/disk-hygiene/lib/engine_grammar.py index 151cafdba0..14c0ee256d 100644 --- a/plugins/disk-hygiene/lib/engine_grammar.py +++ b/plugins/disk-hygiene/lib/engine_grammar.py @@ -220,6 +220,26 @@ def _data_root_flag() -> Flag: ), help="inventory a target without mutating it", ), + Subcommand( + "inventory", + ( + Flag("--target", required=True, example="target-dir"), + _data_root_flag(), + Flag( + "--deep", + takes_value=False, + help=( + "list every level of the target instead of its immediate " + "children; the default when the target is the user's home " + "directory" + ), + ), + ), + help=( + "report each entry's producer, disposition and reason; writes a " + "report that preview and apply never accept" + ), + ), Subcommand( "preview", ( diff --git a/plugins/disk-hygiene/skills/clean/SKILL.md b/plugins/disk-hygiene/skills/clean/SKILL.md index 2ef2d309ad..e80a79cd9e 100644 --- a/plugins/disk-hygiene/skills/clean/SKILL.md +++ b/plugins/disk-hygiene/skills/clean/SKILL.md @@ -447,8 +447,8 @@ and what the guard does when no Python resolves → "Hook launch form". snapshot token exists. - `allowed-tools` would pre-approve rather than restrict tools, so this destructive skill intentionally grants none. Consumer permission policy remains authoritative. -- The Bash lane is deny-by-default: only the literal-word bundled scan, preview, handoff-verify, and - apply shapes (plus the argument-free kill-switch probe) pass, using the hook runtime's own absolute +- The Bash lane is deny-by-default: only the literal-word bundled scan, inventory, preview, + handoff-verify, and apply shapes (plus the argument-free kill-switch probe) pass, using the hook runtime's own absolute interpreter. The same denial text also admits literal-form read-only supporting commands whose heads are absolute paths under a trusted system directory: `[`, `basename`, `dirname`, `du`, `file`, `find`, `ls`, `pwd`, `stat`, `test` (`[` only as a complete `/usr/bin/[ ... ]` diff --git a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py index 0aed9457fc..88a69ea2f7 100644 --- a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py +++ b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py @@ -31,6 +31,7 @@ import json import os import re +import stat from collections.abc import Iterable from pathlib import Path from typing import Any @@ -140,11 +141,33 @@ def make_row( ) -> dict[str, Any]: """One schema row for ``path``, measured with lstat so a link is never followed.""" st = os.lstat(path) - is_dir = os.path.isdir(path) and not os.path.islink(path) + return _stat_row( + path, + st, + _tree_size(path) if stat.S_ISDIR(st.st_mode) else st.st_size, + producer=producer, + category=category, + disposition=disposition, + reason=reason, + evidence=evidence, + ) + + +def _stat_row( + path: Path, + st: os.stat_result, + size: int, + *, + producer: str, + category: str, + disposition: str, + reason: str, + evidence: dict[str, Any] | None = None, +) -> dict[str, Any]: row: dict[str, Any] = { "name": str(path), - "ext": "" if is_dir else path.suffix, - "size": _tree_size(path) if is_dir else st.st_size, + "ext": "" if stat.S_ISDIR(st.st_mode) else path.suffix, + "size": size, "mtime": _iso(st.st_mtime), "owner": _owner(st.st_uid), "producer": producer, @@ -494,24 +517,166 @@ def dangling_symlinks( dirs[:] = [] for name in dirs + files: path = Path(current) / name - if not path.is_symlink() or path.exists(): - continue - producer = next( - (p for hint, p in LINK_PRODUCERS if hint in path.as_posix()), - "unknown", - ) - rows.append( - make_row( - path, - producer=producer, - category="dangling-symlink", - disposition="CANDIDATE", - reason=f"points to {os.readlink(path)}, which does not exist", - ) - ) + if path.is_symlink() and not path.exists(): + rows.append(dangling_row(path)) return rows +def dangling_row(path: Path) -> dict[str, Any]: + """The candidate row for one symlink whose target does not exist.""" + producer = next( + (p for hint, p in LINK_PRODUCERS if hint in path.as_posix()), "unknown" + ) + return make_row( + path, + producer=producer, + category="dangling-symlink", + disposition="CANDIDATE", + reason=f"points to {os.readlink(path)}, which does not exist", + ) + + +FILE_ATTRIBUTE_REPARSE_POINT = 0x0400 +UNCLASSIFIED = { + "producer": "unknown", + "category": "unclassified", + "disposition": "UNKNOWN", + "reason": "no category rule attributes this entry", +} + + +def _descends(st: os.stat_result) -> bool: + """A real directory: not a symlink, junction or other reparse point.""" + return stat.S_ISDIR(st.st_mode) and not ( + getattr(st, "st_file_attributes", 0) & FILE_ATTRIBUTE_REPARSE_POINT + ) + + +def _within(path: Path, root: Path) -> bool: + return path == root or root in path.parents + + +def _dotted_versions(names: Iterable[str]) -> bool: + """Two or more dotted version names, so a year-named folder pair never qualifies.""" + dotted = (VERSION_RE.match(name) for name in names) + return sum(1 for m in dotted if m and "." in m.group(1)) >= 2 + + +def category_rows( + target: Path, *, home: Path, tmp_dir: Path, now: float, running: set[str] +) -> dict[str, dict[str, Any]]: + """Rows of each category whose root lies inside ``target``, keyed by name.""" + rows: list[dict[str, Any]] = [] + claude_dir = home / ".claude" + if _within(claude_dir, target): + rows += plugin_cache_versions(claude_dir) + rows += project_transcripts(claude_dir / "projects") + if _within(tmp_dir, target): + rows += tmp_entries(tmp_dir, now, running) + return {row["name"]: row for row in rows} + + +def inventory_rows( + target: Path, + *, + deep: bool, + home: Path, + tmp_dir: Path, + now: float, + running: Iterable[str] = (), + skip: frozenset[str] = frozenset(), +) -> Iterable[dict[str, Any]]: + """Yield one row per entry of ``target``, the target itself last. + + ``deep`` lists every level, each directory after its contents with the + sum of their sizes; otherwise only the immediate children, each directory + sized by its own walk. A category row replaces the unclassified row at its + path. A directory that cannot be read, or that is another filesystem's + mount point, is one UNKNOWN row and is not entered. Paths in ``skip`` (the + report being written) are left out. + """ + running = set(running) + overrides = category_rows( + target, home=home, tmp_dir=tmp_dir, now=now, running=running + ) + + def children(directory: Path) -> list[tuple[Path, os.stat_result | None, str]]: + with os.scandir(directory) as it: + entries = sorted(it, key=lambda e: e.name, reverse=True) + if _dotted_versions(e.name for e in entries): + for row in superseded_versions([directory], running): + overrides.setdefault(row["name"], row) + found: list[tuple[Path, os.stat_result | None, str]] = [] + for entry in entries: + if entry.path in skip: + continue + try: + found.append((Path(entry.path), entry.stat(follow_symlinks=False), "")) + except OSError as exc: + found.append((Path(entry.path), None, f"{type(exc).__name__}: {exc}")) + return found + + def row(path: Path, st: os.stat_result, size: int) -> dict[str, Any]: + found = overrides.get(str(path)) + if found is not None: + return found + if stat.S_ISLNK(st.st_mode) and not path.exists(): + return dangling_row(path) + return _stat_row(path, st, size, **UNCLASSIFIED) + + def not_entered(path: Path, st: os.stat_result, why: str) -> dict[str, Any]: + return _stat_row( + path, + st, + 0, + producer="unknown", + category="not-walked", + disposition="UNKNOWN", + reason=f"{why}; its contents and size are not counted", + ) + + root_st = os.lstat(target) + try: + stack = [(target, root_st, children(target), [0])] + except OSError as exc: + yield not_entered(target, root_st, f"unreadable ({type(exc).__name__}: {exc})") + return + while stack: + directory, dir_st, pending, total = stack[-1] + if not pending: + stack.pop() + if stack: + stack[-1][3][0] += total[0] + yield row(directory, dir_st, total[0]) + continue + path, st, error = pending.pop() + if st is None: + yield { + "name": str(path), + "ext": path.suffix, + "size": 0, + "mtime": None, + "owner": None, + **UNCLASSIFIED, + "category": "not-walked", + "reason": f"cannot be read ({error})", + } + continue + # Windows DirEntry stats carry st_dev 0, so only a real device id compares. + if _descends(st) and st.st_dev and st.st_dev != root_st.st_dev: + yield not_entered(path, st, "another filesystem is mounted here") + continue + if deep and _descends(st): + try: + stack.append((path, st, children(path), [0])) + except OSError as exc: + yield not_entered(path, st, f"unreadable ({type(exc).__name__}: {exc})") + continue + size = _tree_size(path) if _descends(st) else st.st_size + total[0] += size + yield row(path, st, size) + + def _category_only(reason: str) -> bool: parts = [ p for p in re.split(r"\s*(?:[;,.]|\band\b)\s*", reason.strip().lower()) if p diff --git a/plugins/disk-hygiene/skills/clean/scripts/destructive_guard.py b/plugins/disk-hygiene/skills/clean/scripts/destructive_guard.py index 9ac35ad21c..6b5249a368 100755 --- a/plugins/disk-hygiene/skills/clean/scripts/destructive_guard.py +++ b/plugins/disk-hygiene/skills/clean/scripts/destructive_guard.py @@ -569,6 +569,10 @@ def _within_plugin_cache_family(value: str) -> bool: # before it matches flags, so an unknown subcommand fails closed. _ALLOWED_ENGINE_SUBCOMMANDS = engine_grammar.SUBCOMMAND_NAMES +# The subcommands allowed without a prompt. Named here rather than derived from +# the grammar, so a subcommand added there is denied until it is listed. +_READ_ONLY_ENGINE_SUBCOMMANDS = ("scan", "inventory", "preview", "handoff-verify") + def _engine_script_path() -> Path: """The one bundled engine path both the classifier and the denial disclose.""" @@ -2510,7 +2514,7 @@ def _decide(command: str, tool_name: str, start: float) -> int: "(disk-hygiene belt inspection allowlist).", ) command_kind = classify_exact_engine_command(command, authority) - if command_kind in {"scan", "preview", "handoff-verify"}: + if command_kind in _READ_ONLY_ENGINE_SUBCOMMANDS: return _settle( command, tool_name, @@ -2537,7 +2541,9 @@ def _decide(command: str, tool_name: str, start: float) -> int: "kill-switch-disabled-apply" if denied_by_kill_switch else "not-exact-engine-command", - "Disk-hygiene execution is disabled; only exact bundled scan, preview, and handoff-verify invocations are permitted." + "Disk-hygiene execution is disabled; only exact bundled " + + ", ".join(_READ_ONLY_ENGINE_SUBCOMMANDS) + + " invocations are permitted." if denied_by_kill_switch else _bash_denial_guidance(authority), ) diff --git a/plugins/disk-hygiene/skills/clean/scripts/hygiene.py b/plugins/disk-hygiene/skills/clean/scripts/hygiene.py index 10eea0d956..5458de5edd 100755 --- a/plugins/disk-hygiene/skills/clean/scripts/hygiene.py +++ b/plugins/disk-hygiene/skills/clean/scripts/hygiene.py @@ -29,6 +29,11 @@ import engine_grammar # noqa: E402 (path set above; plugin-bundled module) +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) + +import deep_inventory # noqa: E402 (path set above; sibling module) + MIN_PYTHON = (3, 11) SCHEMA_VERSION = 1 MAX_SNAPSHOT_ENTRIES = 250_000 @@ -4277,6 +4282,70 @@ def apply_plan(snapshot: dict[str, Any], plan: dict[str, Any]) -> dict[str, Any] } +INVENTORY_REPORT_KIND = "deep-inventory-report" +# Distinct from 2 (invalid), 3 (blocked) and 4 (apply skipped paths). +INVENTORY_VALIDATION_FAILED = 5 + + +def run_inventory(target_arg: str, deep_flag: bool) -> int: + """List a target into a JSONL report under the data root, then validate it. + + Report only: nothing here deletes, and the summary is neither a snapshot + nor a plan, so preview and apply refuse it. Rows stream to the file, so no + entry cap applies. Deep is the default when the target is the home + directory. + """ + target_input = Path(target_arg).expanduser().absolute() + if not target_input.is_dir() or has_linkish_component(target_input): + raise HygieneError( + "target must be an existing directory with no link or reparse-point component" + ) + target = target_input.resolve(strict=True) + if is_os_managed_target(target): + raise HygieneError("OS-managed roots are not valid inventory targets") + home = Path.home().resolve() + deep = deep_flag or target == home + stamp = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S%fZ") + rows_path = state_output_path( + Path(DATA_ROOT_OVERRIDE or "") / "inventory" / f"inventory-{stamp}.jsonl" + ) + summary_path = state_output_path(rows_path.with_suffix(".json")) + rows_path.parent.mkdir(parents=True, exist_ok=True) + dispositions: dict[str, int] = {} + failures: list[str] = [] + with rows_path.open("w", encoding="utf-8") as out: + for row in deep_inventory.inventory_rows( + target, + deep=deep, + home=home, + tmp_dir=Path(tempfile.gettempdir()).resolve(), + now=dt.datetime.now(dt.timezone.utc).timestamp(), + running=deep_inventory.running_paths(), + skip=frozenset({str(rows_path), str(summary_path)}), + ): + out.write(json.dumps(row, sort_keys=True) + "\n") + key = str(row.get("disposition")) + dispositions[key] = dispositions.get(key, 0) + 1 + failures.extend(deep_inventory.validate_report([row])) + summary = { + "kind": INVENTORY_REPORT_KIND, + "status": "inventory-failed" if failures else "inventory-complete", + "target": str(target), + "deep": deep, + "rows": str(rows_path), + "row_count": sum(dispositions.values()), + "dispositions": dispositions, + "validation_failures": failures, + "note": ( + "Report only: rows are findings, never a deletion plan, and preview " + "and apply do not accept this report. Removing a candidate goes " + "through scan, preview and apply with their confirmation gates." + ), + } + write_json(summary_path, summary) + return emit(summary, INVENTORY_VALIDATION_FAILED if failures else 0) + + _PARSER_VALUE_TYPES = {"int": int} @@ -4592,6 +4661,8 @@ def main(argv: list[str] | None = None) -> int: args.quiet, ) ) + if args.command == "inventory": + return run_inventory(args.target, args.deep) snapshot = load_json(Path(args.snapshot)) if args.command == "handoff-verify": approved = validate_handoff_paths( diff --git a/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py b/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py index b7d62ffd17..cdb704bcc3 100755 --- a/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py +++ b/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py @@ -11976,6 +11976,7 @@ def resolve_enabled(self) -> bool: ENGINE_TAILS = { "scan": "scan --target t --output s", + "inventory": "inventory --target t --deep", "preview": "preview --snapshot s --plan p", "handoff-verify": "handoff-verify --snapshot s --paths q", "apply": ( @@ -12397,6 +12398,7 @@ def test_engine_call_with_data_root_keeps_its_verdict(self) -> None: self.set_env_data_root() verdicts = { "scan": "allow", + "inventory": "allow", "preview": "allow", "handoff-verify": "allow", "apply": "ask", @@ -12841,6 +12843,228 @@ def test_grammar_refuses_a_subcommand_it_does_not_declare(self) -> None: self.assertIsNone(self.grammar.subcommand("summarize")) self.assertFalse(self.grammar.match_invocation("summarize", [])) + def test_apply_grammar_is_unchanged(self) -> None: + apply_spec = self.grammar.subcommand("apply") + assert apply_spec is not None + self.assertEqual("apply", self.grammar.SUBCOMMAND_NAMES[-1]) + self.assertEqual( + [ + "--execute", + "--snapshot", + "--plan", + "--confirm-tier", + "--approval-token", + "--report", + "--data-root", + ], + [flag.name for flag in apply_spec.flags], + ) + self.assertEqual( + ["--snapshot", "--plan", "--data-root"], + [flag.name for flag in self.grammar.subcommand("preview").flags], + ) + + def test_only_apply_is_left_off_the_read_only_allowance(self) -> None: + self.assertEqual( + set(self.grammar.SUBCOMMAND_NAMES) - {"apply"}, + set(guard._READ_ONLY_ENGINE_SUBCOMMANDS), + ) + + def test_inventory_takes_deep_but_never_an_execute_flag(self) -> None: + head = ["--target", "target-dir", "--data-root", self.AUTHORITY] + self.assertTrue(self.parse("inventory", [*head, "--deep"]).deep) + self.assertFalse(self.parse("inventory", head).deep) + self.assertEqual("inventory", self.classify("inventory", [*head, "--deep"])) + for extra in (["--execute"], ["--report", "report.json"], ["--plan", "p"]): + with self.subTest(extra=extra): + self.assertIsNone(self.classify("inventory", [*head, *extra])) + self.refuse_parse("inventory", [*head, *extra]) + + +class InventoryCommandTests(unittest.TestCase): + """The read-only ``inventory`` subcommand and the report it writes.""" + + def setUp(self) -> None: + temporary = tempfile.TemporaryDirectory() + self.addCleanup(temporary.cleanup) + base = Path(temporary.name).resolve() + self.target = base / "target" + self.data_root = base / "data" + self.target.mkdir() + self.data_root.mkdir() + # HOME points away from the target unless a test says otherwise. + self.enterContext(mock.patch.dict(os.environ, {"HOME": str(base / "home")})) + + def write(self, relative: str, text: str = "x") -> Path: + path = self.target / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text, encoding="utf-8") + return path + + def run_inventory(self, *extra: str) -> tuple[int, dict[str, Any]]: + output = io.StringIO() + with redirect_stdout(output): + code = hygiene.main( + [ + "inventory", + "--target", + str(self.target), + "--data-root", + str(self.data_root), + *extra, + ] + ) + return code, json.loads(output.getvalue()) + + def rows(self, summary: dict[str, Any]) -> dict[str, dict[str, Any]]: + lines = Path(summary["rows"]).read_text(encoding="utf-8").splitlines() + return {row["name"]: row for row in map(json.loads, lines)} + + def test_deep_lists_every_level_with_bottom_up_sizes(self) -> None: + self.write("a/b/c.txt", "12345") + self.write("a/d.log", "123") + (self.target / "gone").symlink_to(self.target / "missing") + self.write("tool/1.2.0/bin", "old") + self.write("tool/1.10.0/bin", "new") + code, summary = self.run_inventory("--deep") + self.assertEqual(0, code, summary) + self.assertEqual("deep-inventory-report", summary["kind"]) + self.assertEqual("inventory-complete", summary["status"]) + self.assertTrue(summary["deep"]) + rows = self.rows(summary) + self.assertEqual(len(rows), summary["row_count"]) + self.assertTrue(hygiene.is_within(Path(summary["rows"]), self.data_root)) + nested = rows[str(self.target / "a" / "b" / "c.txt")] + self.assertEqual( + (5, ".txt", "UNKNOWN"), + (nested["size"], nested["ext"], nested["disposition"]), + ) + self.assertEqual(8, rows[str(self.target / "a")]["size"]) + self.assertEqual( + "dangling-symlink", rows[str(self.target / "gone")]["category"] + ) + old = rows[str(self.target / "tool" / "1.2.0")] + self.assertEqual( + ("superseded-version", "CANDIDATE"), (old["category"], old["disposition"]) + ) + self.assertEqual( + "KEEP", rows[str(self.target / "tool" / "1.10.0")]["disposition"] + ) + lines = Path(summary["rows"]).read_text(encoding="utf-8").splitlines() + self.assertEqual(str(self.target), json.loads(lines[-1])["name"]) + for row in rows.values(): + self.assertLessEqual( + set(hygiene.deep_inventory.ROW_COLUMNS) - {"evidence"}, set(row) + ) + + def test_year_named_folders_are_not_versions(self) -> None: + self.write("Pictures/2024/a.jpg") + self.write("Pictures/2025/b.jpg") + _, summary = self.run_inventory("--deep") + row = self.rows(summary)[str(self.target / "Pictures" / "2024")] + self.assertEqual("unclassified", row["category"]) + + def test_without_deep_lists_only_immediate_children(self) -> None: + self.write("a/b/c.txt", "12345") + code, summary = self.run_inventory() + self.assertEqual(0, code, summary) + self.assertFalse(summary["deep"]) + rows = self.rows(summary) + self.assertEqual({str(self.target), str(self.target / "a")}, set(rows)) + self.assertEqual(5, rows[str(self.target / "a")]["size"]) + + def test_home_target_is_deep_by_default_and_runs_claude_categories(self) -> None: + self.write("notes/todo.md") + cache = self.target / ".claude" / "plugins" / "cache" / "mkt" / "plug" + (cache / "1.0.0").mkdir(parents=True) + (cache / "2.0.0").mkdir(parents=True) + (self.target / ".claude" / "plugins" / "installed_plugins.json").write_text( + json.dumps( + {"plugins": {"plug@mkt": [{"installPath": str(cache / "2.0.0")}]}} + ), + encoding="utf-8", + ) + with mock.patch.dict(os.environ, {"HOME": str(self.target)}): + code, summary = self.run_inventory() + self.assertEqual(0, code, summary) + self.assertTrue(summary["deep"]) + rows = self.rows(summary) + self.assertIn(str(self.target / "notes" / "todo.md"), rows) + self.assertEqual("CANDIDATE", rows[str(cache / "1.0.0")]["disposition"]) + self.assertEqual("KEEP", rows[str(cache / "2.0.0")]["disposition"]) + + def test_report_inside_the_target_is_not_listed(self) -> None: + self.data_root = self.target / "data" + self.data_root.mkdir() + _, summary = self.run_inventory("--deep") + self.assertNotIn(summary["rows"], self.rows(summary)) + + @unittest.skipIf( + os.name == "nt" or os.geteuid() == 0, "needs POSIX modes as non-root" + ) + def test_unreadable_subtree_is_an_unknown_row(self) -> None: + locked = self.target / "locked" + self.write("locked/secret.txt") + locked.chmod(0) + self.addCleanup(locked.chmod, 0o700) + code, summary = self.run_inventory("--deep") + self.assertEqual(0, code, summary) + row = self.rows(summary)[str(locked)] + self.assertEqual( + ("not-walked", "UNKNOWN"), (row["category"], row["disposition"]) + ) + self.assertIn("PermissionError", row["reason"]) + self.assertNotIn(str(locked / "secret.txt"), self.rows(summary)) + + def test_a_failing_validator_fails_the_report(self) -> None: + bad = self.write("keep.bin") + row = hygiene.deep_inventory.make_row( + bad, + producer="tool", + category="test", + disposition="KEEP", + reason="tool-managed", + ) + with mock.patch.object( + hygiene.deep_inventory, "category_rows", return_value={str(bad): row} + ): + code, summary = self.run_inventory("--deep") + self.assertEqual(hygiene.INVENTORY_VALIDATION_FAILED, code) + self.assertEqual("inventory-failed", summary["status"]) + self.assertEqual(1, len(summary["validation_failures"])) + written = json.loads(Path(summary["rows"]).with_suffix(".json").read_text()) + self.assertEqual("inventory-failed", written["status"]) + + def test_report_is_never_accepted_as_a_snapshot_or_plan(self) -> None: + self.write("a.tmp") + _, summary = self.run_inventory("--deep") + report = Path(summary["rows"]).with_suffix(".json") + for path in (report, Path(summary["rows"])): + output = io.StringIO() + with redirect_stdout(output): + code = hygiene.main( + [ + "preview", + "--snapshot", + str(path), + "--plan", + str(path), + "--data-root", + str(self.data_root), + ] + ) + with self.subTest(path=path.name): + self.assertNotEqual(0, code) + self.assertNotIn("ready-for-explicit-approval", output.getvalue()) + + def test_inventory_needs_a_data_root(self) -> None: + output = io.StringIO() + with redirect_stdout(output): + code = hygiene.main(["inventory", "--target", str(self.target)]) + self.assertEqual(2, code) + self.assertIn("--data-root", output.getvalue()) + self.assertEqual([], list(self.data_root.iterdir())) + if __name__ == "__main__": unittest.main() From c598437b73db7980abeb3c3f0b6f8f89cb3e22ba Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:44:30 -0400 Subject: [PATCH 3/9] test(disk-hygiene): make inventory tests host-independent and cache owner lookups Patch Path.home and the OS-managed check in the inventory tests so they hold on macOS temp paths and Windows homes, and cache uid-to-name lookups so a home-directory walk does one password-database lookup per owner. Refs #5221 Co-Authored-By: Claude Opus 5.5 --- .../skills/clean/scripts/deep_inventory.py | 2 ++ .../skills/clean/scripts/test_hygiene.py | 14 ++++++++++---- 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py index 88a69ea2f7..123e7f5aa9 100644 --- a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py +++ b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py @@ -28,6 +28,7 @@ from __future__ import annotations import datetime as dt +import functools import json import os import re @@ -102,6 +103,7 @@ def _iso(epoch: float) -> str: ) +@functools.lru_cache(maxsize=None) def _owner(uid: int) -> str: if pwd is not None: try: diff --git a/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py b/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py index cdb704bcc3..abbe81c64c 100755 --- a/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py +++ b/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py @@ -12892,8 +12892,14 @@ def setUp(self) -> None: self.data_root = base / "data" self.target.mkdir() self.data_root.mkdir() - # HOME points away from the target unless a test says otherwise. - self.enterContext(mock.patch.dict(os.environ, {"HOME": str(base / "home")})) + # Home points away from the target unless a test says otherwise, and a + # macOS temp dir sits under /private, which is OS-managed. + self.home = self.enterContext( + mock.patch.object(Path, "home", return_value=base / "home") + ) + self.enterContext( + mock.patch.object(hygiene, "is_os_managed_target", return_value=False) + ) def write(self, relative: str, text: str = "x") -> Path: path = self.target / relative @@ -12984,8 +12990,8 @@ def test_home_target_is_deep_by_default_and_runs_claude_categories(self) -> None ), encoding="utf-8", ) - with mock.patch.dict(os.environ, {"HOME": str(self.target)}): - code, summary = self.run_inventory() + self.home.return_value = self.target + code, summary = self.run_inventory() self.assertEqual(0, code, summary) self.assertTrue(summary["deep"]) rows = self.rows(summary) From 9d342c56f5e82a313c92d65dba3d13368241d4cc Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:52:40 -0400 Subject: [PATCH 4/9] docs(disk-hygiene): document attended deep inventory and add repo-hygiene pointer Document `--deep` as an attended, report-only mode: default for a whole-home target, forced elsewhere. List the shared row columns and named categories, state that every KEEP needs a specific reason checked by the validator, and that --execute, the low-signal rule and the confirmation gates are unchanged. Add an eval case, and a one-line pointer in repo-hygiene:clean to machine-level listing. Refs #5221 Co-Authored-By: Claude Opus 5.5 --- plugins/disk-hygiene/skills/clean/SKILL.md | 28 ++++++++++++++++--- .../skills/clean/evals/evals.json | 13 +++++++++ .../skills/clean/reference/safety-model.md | 17 +++++++++++ .../skills/clean/reference/scan-flags.md | 26 ++++++++++++++++- plugins/repo-hygiene/skills/clean/SKILL.md | 2 ++ 5 files changed, 81 insertions(+), 5 deletions(-) diff --git a/plugins/disk-hygiene/skills/clean/SKILL.md b/plugins/disk-hygiene/skills/clean/SKILL.md index e80a79cd9e..6417dd60fc 100644 --- a/plugins/disk-hygiene/skills/clean/SKILL.md +++ b/plugins/disk-hygiene/skills/clean/SKILL.md @@ -1,6 +1,6 @@ --- description: "Audit an arbitrary directory tree for orphaned, temporary, stale-lock, failed-write, partial-download, and empty leftover artifacts; classify evidence into confidence tiers; and optionally remove exact validated paths after explicit per-tier approval. Read-only by default and manual-only. Use when: 'audit this directory', 'find orphaned files', 'what junk can I clean up', 'reclaim disk space', 'find temp or lock leftovers', 'clean up my home directory'. Skip when: repository cache/build cleanup belongs to repo-hygiene, a product has its own prune/GC command, or the target is an OS-managed root." -argument-hint: "[--execute] [--max-depth ] [--sizes-only] [--policy ] [options] " +argument-hint: "[--execute] [--deep] [--max-depth ] [--sizes-only] [--policy ] [options] " user-invocable: true disable-model-invocation: true hooks: @@ -27,7 +27,7 @@ metadata: summary: Audit a directory tree for stale leftovers and remove validated paths --- -**Arguments.** `[--execute] [--max-depth ] [--sizes-only] [--policy ] [options] `. Full form: `[--execute] [--policy ] [--max-depth ] [--confirmed-large-scan] [--sizes-only] [--quiet] [--root-children [--root-child ]...] ` +**Arguments.** `[--execute] [--deep] [--max-depth ] [--sizes-only] [--policy ] [options] `. Full form: `[--execute] [--deep] [--policy ] [--max-depth ] [--confirmed-large-scan] [--sizes-only] [--quiet] [--root-children [--root-child ]...] ` # Disk hygiene @@ -41,8 +41,7 @@ optional execution lane. ## Arguments and boundaries -Parse `$ARGUMENTS` as the complete user-facing surface: optional `--execute`, optional -`--policy `, optional `--max-depth `, optional `--confirmed-large-scan`, optional +Parse `$ARGUMENTS` as the complete user-facing surface: optional `--execute`, optional `--deep` ([deep inventory](#deep-inventory)), optional `--policy `, optional `--max-depth `, optional `--confirmed-large-scan`, optional `--quiet`, optional `--root-children` with zero or more `--root-child `, and one target directory. Remaining engine flags (`--output`, `--project-dir`, `--data-root` on scan; `--snapshot`, `--plan`, `--report`, `--confirm-tier`, `--approval-token`, `--paths`, `--path`, and @@ -158,6 +157,27 @@ naming what the question never presented cannot be met. and has no entry cap. Detail: [scan-flags.md](reference/scan-flags.md#--sizes-only). +## Deep inventory + +`--deep` runs the read-only `inventory` subcommand (`hygiene.py inventory --target +--data-root [--deep]`, the Bash-lane shape listed in Gotchas). It is the default when +the target is the user's home directory and is forced elsewhere with `--deep`; without either, only +the target's immediate children are listed. It runs only in an attended session and only reports: +it never deletes, prepares an approval, or produces a snapshot or plan, and `preview` refuses its +output. It asks no question, so it passes no confirmation-gate row. The row columns, named +categories, and shared schema are in [scan-flags.md](reference/scan-flags.md#--deep). + +Every `KEEP` row needs a specific reason: who produced the entry and what still uses it. The +validator fails an empty reason and a reason that is only a category phrase ("tool-managed", +"OS-owned", "managed by ") unless `evidence` shows the named tool still references the +entry. Present the report grouped by category, `CANDIDATE` rows first, and report `UNKNOWN` rows as +coverage gaps. + +Nothing else in this skill changes: `--execute` still means only that deletion may be offered, a +`CANDIDATE` row is a finding and not a tier, the tiers and the low-signal rule in §3 decide what +may be offered, and removal still needs `scan`, a fresh `preview`, and the confirmation gate's +removal row. + ## 1. Create a read-only snapshot Create a unique run directory under `${CLAUDE_PLUGIN_DATA}/runs/`; snapshots, plans, and reports must diff --git a/plugins/disk-hygiene/skills/clean/evals/evals.json b/plugins/disk-hygiene/skills/clean/evals/evals.json index 62785e6828..2f644255b3 100644 --- a/plugins/disk-hygiene/skills/clean/evals/evals.json +++ b/plugins/disk-hygiene/skills/clean/evals/evals.json @@ -158,6 +158,19 @@ "Does not treat 'go, execute these' as the removal approval; the approval must name exactly one tier and its path list", "Deletes nothing before that approval" ] + }, + { + "id": 14, + "name": "deep-inventory-lists-with-justified-keep-and-deletes-nothing", + "prompt": "/disk-hygiene:clean --deep ~ — list everything in my home directory and tell me what is safe to remove.", + "expected_output": "Runs the read-only inventory subcommand in deep mode and presents a report grouped by category with CANDIDATE rows first. Every KEEP row carries a specific reason. Nothing is deleted; removal of any candidate goes through scan, a fresh preview, and one-tier approval.", + "files": [], + "expectations": [ + "Runs hygiene.py inventory with --deep and does not run apply", + "Rows use the shared schema columns including producer, category, disposition, reason, and evidence", + "Gives every KEEP row a specific reason and does not accept a bare category phrase such as 'tool-managed' without evidence the tool still references the entry", + "Treats a CANDIDATE row as a finding, not approval: offers removal only through scan, preview, and the confirmation gate's one-tier path list" + ] } ] } diff --git a/plugins/disk-hygiene/skills/clean/reference/safety-model.md b/plugins/disk-hygiene/skills/clean/reference/safety-model.md index cce0b88cb2..861dee72fc 100644 --- a/plugins/disk-hygiene/skills/clean/reference/safety-model.md +++ b/plugins/disk-hygiene/skills/clean/reference/safety-model.md @@ -4,6 +4,7 @@ - [Trust boundaries](#trust-boundaries) - [Tidiness, not emergency](#tidiness-not-emergency) +- [Deep inventory is report-only](#deep-inventory-is-report-only) - [Non-overridable checks](#non-overridable-checks) - [Live agent scratchpads](#live-agent-scratchpads) - [Handle semantics and honest scope](#handle-semantics-and-honest-scope) @@ -48,6 +49,22 @@ the current defaults rather than funding a proportionality rebuild. **As of:** 2 **Recheck:** reopening #3855, or a funded design that names which rule yields and under what bounded conditions. +## Deep inventory is report-only + +The `inventory` subcommand (the [deep inventory](../SKILL.md#deep-inventory) mode) is attended and +read-only. It writes only its own JSONL report and summary under the data root. Neither is a +snapshot or a plan, so `preview` refuses them and no approval token can derive from them. A +`CANDIDATE` disposition is a finding: it grants no tier, and the guard admits the subcommand +because it cannot mutate, not because its output authorizes anything. + +Every `KEEP` row must carry a specific reason (who produced the entry and what still uses it). The +validator rejects an empty reason and a bare category phrase unless `evidence` shows the named tool +still references the entry, and a rejected row fails the report with exit 5. + +`--execute`, the low-signal rule (Low is kept unless the human separately reviews exact paths), and +every confirmation gate apply exactly as before. Removing anything the inventory lists goes through +`scan`, a fresh `preview`, and the removal approval. + ## Non-overridable checks - target containment; an OS-managed root (per `system_roots()`: the OS drive holding an existing diff --git a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md index 80da0df13a..8da96ce8c5 100644 --- a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md +++ b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md @@ -1,7 +1,7 @@ # Scan flag detail Flag-by-flag behavior of the `scan` flags `--quiet`, `--root-children` with `--root-child`, and -`--sizes-only` for `/disk-hygiene:clean`. The parse contract, the rejection rules, and the +`--sizes-only`, plus the inventory flag `--deep`, for `/disk-hygiene:clean`. The parse contract, the rejection rules, and the confirmation gate stay in [SKILL.md](../SKILL.md#arguments-and-boundaries). ## `--quiet` @@ -29,6 +29,30 @@ more explicit `--root-child ` flags, after the human clears the confirmati root-children row, it audits only those admitted children into one snapshot. A general "clean everything" is not selection. +## `--deep` + +`--deep` selects the deep mode of the read-only `inventory` subcommand: every level of the target +is listed with bottom-up directory sizes instead of only its immediate children. It is the default +when the target is the user's home directory, so the flag matters for any other target. It is not a +`scan` flag: it takes no `--max-depth`, snapshot or entry cap, and it produces a JSONL report that +`preview` and `apply` refuse. The `KEEP` reason rule and the unchanged gates: +[SKILL.md](../SKILL.md#deep-inventory). + +Rows stream to `/inventory/inventory-.jsonl` with no entry cap, beside a `.json` +summary (`deep-inventory-report`; status `inventory-failed` and exit 5 when the validator rejects a +row). The shared listing schema is defined in `scripts/deep_inventory.py`: one row per entry with +the columns `name`, `ext`, `size`, `mtime`, `owner`, `producer`, `category`, `disposition` +(`KEEP`, `CANDIDATE`, `UNKNOWN`), `reason`, and `evidence`. A directory's `size` is the sum of the +files beneath it. Named categories: `superseded-version` (dotted versioned directories), +`plugin-cache-version` (cache versions no installed plugin references), `tmp-producer` (`/tmp` +entries by producer prefix), `transcript-dir` (project transcript directories whose source path is +gone), `dangling-symlink`, and `not-walked` (an unreadable or mounted subtree, one `UNKNOWN` row). +Every other entry is `unclassified`. Other read-only listings that need the same columns reuse this +schema instead of defining their own +([#5214](https://github.com/melodic-software/claude-code-plugins/issues/5214), +[#4006](https://github.com/melodic-software/claude-code-plugins/issues/4006)); their scope stays +their own. + ## `--sizes-only` `--sizes-only` as implemented: it does not ask the large-scan question, so a known-large root walks diff --git a/plugins/repo-hygiene/skills/clean/SKILL.md b/plugins/repo-hygiene/skills/clean/SKILL.md index edacebc90c..3df192446b 100644 --- a/plugins/repo-hygiene/skills/clean/SKILL.md +++ b/plugins/repo-hygiene/skills/clean/SKILL.md @@ -79,6 +79,8 @@ contains git. The dated record for that composition claim is the `source-control Return the repo toward a known-good state. **Selective tiers** (`scan`, `caches`, `build`, `git`, `all`) remove *artifacts* while preserving secrets, runtime deps, and skill data. **`tree`** is the destructive tier. `reset --hard` + `clean -fdx`, but **safe-by-default**: it preserves the same secrets / runtime-deps / skill-data classes unless you opt in via `--include-deps` / `--include-secrets`. +Machine-level listing (a whole home directory, not one repository) is `/disk-hygiene:clean` deep mode; this skill adds no scanner for it. + Bare invocation never mutates silently: resolve intent → dry-run → user confirmation → `--apply`. Full menu, aliases, and confirmation matrix: [context/action-router.md](context/action-router.md). Bundled-script invocation uses two deliberate forms. Paired `${CLAUDE_SKILL_DIR}` in this file (matches `allowed-tools`) and interpreter-led `${CLAUDE_PLUGIN_ROOT}` in routed `context/*.md` detail files. Rationale: [reference/invocation-forms.md](reference/invocation-forms.md). From 13f28339c960fae457ef3ad8f195b17ab85f8551 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:54:50 -0400 Subject: [PATCH 5/9] chore(disk-hygiene): bump to 0.30.0 and repo-hygiene to 0.15.1 with changelog entries Refs #5221 Co-Authored-By: Claude Opus 5.5 --- plugins/disk-hygiene/.claude-plugin/plugin.json | 2 +- plugins/disk-hygiene/CHANGELOG.md | 12 ++++++++++++ plugins/repo-hygiene/.claude-plugin/plugin.json | 2 +- plugins/repo-hygiene/CHANGELOG.md | 9 +++++++++ 4 files changed, 23 insertions(+), 2 deletions(-) diff --git a/plugins/disk-hygiene/.claude-plugin/plugin.json b/plugins/disk-hygiene/.claude-plugin/plugin.json index 5231faba04..f14b9b97a1 100644 --- a/plugins/disk-hygiene/.claude-plugin/plugin.json +++ b/plugins/disk-hygiene/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "disk-hygiene", - "version": "0.29.2", + "version": "0.30.0", "description": "Context-aware disk hygiene for arbitrary directory trees: inventories orphaned and temporary artifacts, classifies evidence into review tiers, and offers exact-path cleanup only after a fresh safety preview and explicit per-tier approval. The target is read-only by default; OS-managed paths, links and mount points, VCS-tracked content without the complete checkout evidence bundle, changed entries, and live-handle uncertainty fail closed.", "author": { "name": "Melodic Software", diff --git a/plugins/disk-hygiene/CHANGELOG.md b/plugins/disk-hygiene/CHANGELOG.md index 11ae4cd3d2..07ffb3c943 100644 --- a/plugins/disk-hygiene/CHANGELOG.md +++ b/plugins/disk-hygiene/CHANGELOG.md @@ -3,6 +3,18 @@ All notable changes to the `disk-hygiene` plugin are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); this plugin uses semantic versioning. +## [0.30.0] - 2026-09-30 + +### Added + +- **Read-only `inventory` subcommand with a deep mode** + ([#5221](https://github.com/melodic-software/claude-code-plugins/issues/5221)). `hygiene.py inventory + --target [--data-root ] [--deep]` lists what is under a target and writes a JSONL report and + a JSON summary under `/inventory/`. Deep mode adds per-category entries, each with a + validated KEEP reason (exit 5 when the validator fails). The engine grammar declares the subcommand + read-only, so the destructive guard admits it. `skills/clean/SKILL.md` and its references document the + attended workflow. + ## [0.29.2] - 2026-09-30 ### Fixed diff --git a/plugins/repo-hygiene/.claude-plugin/plugin.json b/plugins/repo-hygiene/.claude-plugin/plugin.json index 51391df9ed..9a4cfab13e 100644 --- a/plugins/repo-hygiene/.claude-plugin/plugin.json +++ b/plugins/repo-hygiene/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "repo-hygiene", - "version": "0.15.0", + "version": "0.15.1", "description": "Repo hygiene action-router: /repo-hygiene:clean sweeps reclaimable caches, build artifacts, and stale git metadata, and can realign the working tree to a fresh-pull state, dry-run-first, with destructive tiers gated behind explicit confirmation and a session-scoped destructive-command guard. Ecosystem targets are detected at runtime; secrets, runtime dependencies, and skill data are preserved by default.", "author": { "name": "Melodic Software", diff --git a/plugins/repo-hygiene/CHANGELOG.md b/plugins/repo-hygiene/CHANGELOG.md index 3e574e66e3..21a24b41fc 100644 --- a/plugins/repo-hygiene/CHANGELOG.md +++ b/plugins/repo-hygiene/CHANGELOG.md @@ -3,6 +3,15 @@ All notable changes to the `repo-hygiene` plugin are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); this plugin uses semantic versioning. +## [0.15.1] - 2026-09-30 + +### Changed + +- **`clean` points machine-level listing at the disk-hygiene deep inventory** + ([#5221](https://github.com/melodic-software/claude-code-plugins/issues/5221)). + `skills/clean/SKILL.md` names `/disk-hygiene:clean` deep mode for a whole home directory and states + this skill adds no scanner for it. + ## [0.15.0] - 2026-09-30 ### Added From fc25ef3f381e7796754df2725ca733564c3362f8 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Wed, 30 Sep 2026 02:13:17 -0400 Subject: [PATCH 6/9] fix(disk-hygiene): report UNKNOWN when the process table was not read running_paths returned an empty set when /proc was missing, so on macOS and Windows every non-newest version and old /tmp entry became a CANDIDATE with the reason "no running process uses it", which nothing had checked. It now returns None for an unreadable process table, and the rows that rest on it are UNKNOWN with a reason that says the table was not read. Also drops dangling_symlinks, which only tests called (the walk uses dangling_row), moving its coverage to the walk, and trims SKILL.md back under the 500-line cap. Refs #5221 Co-Authored-By: Claude Sonnet 5.5 --- plugins/disk-hygiene/CHANGELOG.md | 3 +- plugins/disk-hygiene/skills/clean/SKILL.md | 16 ++--- .../skills/clean/reference/scan-flags.md | 6 +- .../skills/clean/scripts/deep_inventory.py | 70 ++++++++++--------- .../clean/scripts/test_deep_inventory.py | 66 ++++++++++++----- .../skills/clean/scripts/test_hygiene.py | 20 ++++++ 6 files changed, 121 insertions(+), 60 deletions(-) diff --git a/plugins/disk-hygiene/CHANGELOG.md b/plugins/disk-hygiene/CHANGELOG.md index 07ffb3c943..9d18fd9d5f 100644 --- a/plugins/disk-hygiene/CHANGELOG.md +++ b/plugins/disk-hygiene/CHANGELOG.md @@ -11,7 +11,8 @@ All notable changes to the `disk-hygiene` plugin are documented here. Format fol ([#5221](https://github.com/melodic-software/claude-code-plugins/issues/5221)). `hygiene.py inventory --target [--data-root ] [--deep]` lists what is under a target and writes a JSONL report and a JSON summary under `/inventory/`. Deep mode adds per-category entries, each with a - validated KEEP reason (exit 5 when the validator fails). The engine grammar declares the subcommand + validated KEEP reason (exit 5 when the validator fails); without a readable `/proc`, rows that depend + on the process table are UNKNOWN, not CANDIDATE. The engine grammar declares the subcommand read-only, so the destructive guard admits it. `skills/clean/SKILL.md` and its references document the attended workflow. diff --git a/plugins/disk-hygiene/skills/clean/SKILL.md b/plugins/disk-hygiene/skills/clean/SKILL.md index d4a1992048..155e4f7747 100644 --- a/plugins/disk-hygiene/skills/clean/SKILL.md +++ b/plugins/disk-hygiene/skills/clean/SKILL.md @@ -168,15 +168,13 @@ output. It asks no question, so it passes no confirmation-gate row. The row colu categories, and shared schema are in [scan-flags.md](reference/scan-flags.md#--deep). Every `KEEP` row needs a specific reason: who produced the entry and what still uses it. The -validator fails an empty reason and a reason that is only a category phrase ("tool-managed", -"OS-owned", "managed by ") unless `evidence` shows the named tool still references the -entry. Present the report grouped by category, `CANDIDATE` rows first, and report `UNKNOWN` rows as -coverage gaps. - -Nothing else in this skill changes: `--execute` still means only that deletion may be offered, a -`CANDIDATE` row is a finding and not a tier, the tiers and the low-signal rule in §3 decide what -may be offered, and removal still needs `scan`, a fresh `preview`, and the confirmation gate's -removal row. +validator fails an empty reason or one that is only a category phrase ("tool-managed", "OS-owned", +"managed by ") unless `evidence` shows the named tool still references the entry. Present the +report grouped by category, `CANDIDATE` rows first, and report `UNKNOWN` rows as coverage gaps. + +A `CANDIDATE` row is a finding, not a tier. `--execute` still means only that deletion may be +offered, the tiers and the low-signal rule in §3 decide what, and removal still needs `scan`, a +fresh `preview`, and the confirmation gate's removal row. ## 1. Create a read-only snapshot diff --git a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md index 8da96ce8c5..aaa3aac178 100644 --- a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md +++ b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md @@ -47,8 +47,10 @@ files beneath it. Named categories: `superseded-version` (dotted versioned direc `plugin-cache-version` (cache versions no installed plugin references), `tmp-producer` (`/tmp` entries by producer prefix), `transcript-dir` (project transcript directories whose source path is gone), `dangling-symlink`, and `not-walked` (an unreadable or mounted subtree, one `UNKNOWN` row). -Every other entry is `unclassified`. Other read-only listings that need the same columns reuse this -schema instead of defining their own +Every other entry is `unclassified`. `superseded-version` and `tmp-producer` keep an entry a running +process uses; where `/proc` cannot be read (macOS, Windows), a row that would be a `CANDIDATE` on +that basis is `UNKNOWN`, because nothing checked the process table. Other read-only listings that +need the same columns reuse this schema instead of defining their own ([#5214](https://github.com/melodic-software/claude-code-plugins/issues/5214), [#4006](https://github.com/melodic-software/claude-code-plugins/issues/4006)); their scope stays their own. diff --git a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py index 123e7f5aa9..d828eeeeed 100644 --- a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py +++ b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py @@ -192,17 +192,19 @@ def _child_dirs(parent: Path) -> list[Path]: return [] -def running_paths(proc_root: Path = Path("/proc")) -> set[str]: +def running_paths(proc_root: Path = Path("/proc")) -> set[str] | None: """Resolved executable and working-directory targets of every process under ``proc_root``. A deleted executable reads `` (deleted)``; the suffix is dropped so the - superseded directory it came from still matches. + superseded directory it came from still matches. None means the process table + could not be read (no ``/proc``, as on macOS and Windows), which is not the + same as an empty set: nothing was checked. """ found: set[str] = set() try: pids = [p for p in proc_root.iterdir() if p.name.isdigit()] except OSError: - return found + return None for pid in pids: for link in ("exe", "cwd"): try: @@ -220,15 +222,16 @@ def _in_use(entry: Path, running: Iterable[str]) -> str | None: def superseded_versions( - parents: Iterable[Path], running: Iterable[str] = () + parents: Iterable[Path], running: Iterable[str] | None = () ) -> list[dict[str, Any]]: """Sibling entries under each parent whose names parse as versions. Keeps the newest and any version a running process executes; the rest are - candidates. A parent with fewer than two version entries yields no rows. + candidates, or UNKNOWN when ``running`` is None (process table not read). A + parent with fewer than two version entries yields no rows. """ rows: list[dict[str, Any]] = [] - running = set(running) + running = None if running is None else set(running) for parent in parents: versions: list[tuple[tuple[int, ...], str, Path]] = [] try: @@ -250,7 +253,7 @@ def superseded_versions( else parent.name ) for _, name, path in versions: - used = _in_use(path, running) + used = _in_use(path, running or ()) if name == newest: disposition, reason, evidence = ( "KEEP", @@ -263,6 +266,13 @@ def superseded_versions( f"a running process executes {used}", {"running": used}, ) + elif running is None: + disposition, reason, evidence = ( + "UNKNOWN", + f"superseded by {newest}; the process table was not read, " + "so whether a process executes it is not known", + None, + ) else: disposition, reason, evidence = ( "CANDIDATE", @@ -371,7 +381,7 @@ def plugin_cache_versions(claude_dir: Path) -> list[dict[str, Any]]: def tmp_entries( tmp_dir: Path, now: float, - running: Iterable[str] = (), + running: Iterable[str] | None = (), min_age_days: float = TMP_MIN_AGE_DAYS, producers: tuple[tuple[str, str, str | None], ...] = TMP_PRODUCERS, ) -> list[dict[str, Any]]: @@ -379,10 +389,11 @@ def tmp_entries( An unattributed entry is UNKNOWN. An attributed one stays when its producer rule names a reason, when a running process uses it, or when it changed - inside ``min_age_days``; otherwise it is a candidate. + inside ``min_age_days``; otherwise it is a candidate, or UNKNOWN when + ``running`` is None (process table not read). """ rows: list[dict[str, Any]] = [] - running = set(running) + running = None if running is None else set(running) try: entries = sorted(tmp_dir.iterdir()) except OSError: @@ -391,7 +402,7 @@ def tmp_entries( match = next((p for p in producers if path.name.startswith(p[0])), None) try: age = (now - os.lstat(path).st_mtime) / DAY - used = _in_use(path, running) + used = _in_use(path, running or ()) except OSError: continue common = {"category": "tmp-producer"} @@ -415,6 +426,12 @@ def tmp_entries( f"changed {age:.1f} days ago, inside the {min_age_days:g}-day window " f"in which a live {producer} run may still use it", ) + elif running is None: + verdict = ( + "UNKNOWN", + f"{producer} leftover unchanged for {age:.0f} days; the process table " + "was not read, so whether a process uses it is not known", + ) else: verdict = ( "CANDIDATE", @@ -507,23 +524,6 @@ def project_transcripts( return rows -def dangling_symlinks( - roots: Iterable[Path], max_depth: int = 8 -) -> list[dict[str, Any]]: - """Symlinks under ``roots`` (to ``max_depth`` levels) whose target does not exist.""" - rows: list[dict[str, Any]] = [] - for root in roots: - base_depth = len(root.parts) - for current, dirs, files in os.walk(root, followlinks=False): - if len(Path(current).parts) - base_depth >= max_depth: - dirs[:] = [] - for name in dirs + files: - path = Path(current) / name - if path.is_symlink() and not path.exists(): - rows.append(dangling_row(path)) - return rows - - def dangling_row(path: Path) -> dict[str, Any]: """The candidate row for one symlink whose target does not exist.""" producer = next( @@ -565,7 +565,12 @@ def _dotted_versions(names: Iterable[str]) -> bool: def category_rows( - target: Path, *, home: Path, tmp_dir: Path, now: float, running: set[str] + target: Path, + *, + home: Path, + tmp_dir: Path, + now: float, + running: set[str] | None, ) -> dict[str, dict[str, Any]]: """Rows of each category whose root lies inside ``target``, keyed by name.""" rows: list[dict[str, Any]] = [] @@ -585,7 +590,7 @@ def inventory_rows( home: Path, tmp_dir: Path, now: float, - running: Iterable[str] = (), + running: Iterable[str] | None = (), skip: frozenset[str] = frozenset(), ) -> Iterable[dict[str, Any]]: """Yield one row per entry of ``target``, the target itself last. @@ -595,9 +600,10 @@ def inventory_rows( sized by its own walk. A category row replaces the unclassified row at its path. A directory that cannot be read, or that is another filesystem's mount point, is one UNKNOWN row and is not entered. Paths in ``skip`` (the - report being written) are left out. + report being written) are left out. ``running`` is the process table; None + means it was not read, so rows that rest on it are UNKNOWN. """ - running = set(running) + running = None if running is None else set(running) overrides = category_rows( target, home=home, tmp_dir=tmp_dir, now=now, running=running ) diff --git a/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py b/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py index 7c53c884ec..bdeb632d19 100644 --- a/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py +++ b/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py @@ -179,8 +179,9 @@ def test_reads_exe_and_cwd_and_drops_the_deleted_suffix(self) -> None: {"/opt/tool/1.0/bin", "/home/u", "/opt/tool/2.0/bin", "/srv"}, ) - def test_missing_proc_root_is_empty(self) -> None: - self.assertEqual(di.running_paths(self.root / "absent"), set()) + def test_unreadable_proc_root_is_none_not_empty(self) -> None: + self.assertIsNone(di.running_paths(self.root / "absent")) + self.assertEqual(di.running_paths(self.mkdir("proc")), set()) class SupersededVersionsTest(TempTree): @@ -209,6 +210,14 @@ def test_a_version_a_running_process_executes_is_kept(self) -> None: self.assertEqual(rows["1.9.0"]["evidence"], {"running": exe}) self.assertEqual(rows["1.10.0"]["disposition"], "CANDIDATE") + def test_unread_process_table_leaves_the_superseded_unknown(self) -> None: + rows = by_name(di.superseded_versions([self.parent], None)) + self.assertEqual(rows["v1.10.1"]["disposition"], "KEEP") + for name in ("1.9.0", "1.10.0", "1.2.0"): + self.assertEqual(rows[name]["disposition"], "UNKNOWN", name) + self.assertIn("process table was not read", rows[name]["reason"]) + self.assertNotIn("no running process", rows[name]["reason"]) + def test_every_keep_passes_the_validator(self) -> None: exe = str(self.parent / "1.2.0") rows = di.superseded_versions([self.parent], {exe}) @@ -359,6 +368,14 @@ def test_a_running_process_inside_an_old_entry_keeps_it(self) -> None: "CANDIDATE", ) + def test_unread_process_table_leaves_an_old_entry_unknown(self) -> None: + rows = by_name(di.tmp_entries(self.tmp, NOW, None)) + self.assertEqual(rows["pytest-of-kyle"]["disposition"], "UNKNOWN") + self.assertIn("process table was not read", rows["pytest-of-kyle"]["reason"]) + # an entry kept for other reasons stays kept + self.assertEqual(rows["pytest-of-old"]["disposition"], "KEEP") + self.assertEqual(rows["systemd-private-abc-svc"]["disposition"], "KEEP") + def test_every_keep_passes_the_validator(self) -> None: self.assertEqual(di.validate_report(di.tmp_entries(self.tmp, NOW)), []) @@ -415,30 +432,47 @@ def test_decode_project_returns_none_for_missing(self) -> None: ) -class DanglingSymlinksTest(TempTree): - def test_reports_only_links_whose_target_is_gone(self) -> None: +class DanglingLinksTest(TempTree): + def walk(self, target: Path) -> dict[str, dict]: + rows = di.inventory_rows( + target, + deep=True, + home=self.root / "no-home", + tmp_dir=self.root / "no-tmp", + now=NOW, + ) + return by_name(list(rows)) + + def test_a_walk_marks_only_links_whose_target_is_gone(self) -> None: home = self.mkdir("home") live = self.write("home/real.toml") trusted = self.mkdir("home/.local/state/mise/trusted-configs") os.symlink(live, trusted / "ok") os.symlink(self.root / "gone" / "mise.toml", trusted / "broken") os.symlink(self.root / "gone", home / "loose") - rows = by_name(di.dangling_symlinks([home])) - self.assertEqual(set(rows), {"broken", "loose"}) + rows = self.walk(home) + self.assertEqual(rows["ok"]["category"], "unclassified") self.assertEqual(rows["broken"]["producer"], "mise") self.assertEqual(rows["loose"]["producer"], "unknown") - self.assertEqual(rows["broken"]["disposition"], "CANDIDATE") - self.assertIn("does not exist", rows["broken"]["reason"]) + for name in ("broken", "loose"): + self.assertEqual(rows[name]["category"], "dangling-symlink") + self.assertEqual(rows[name]["disposition"], "CANDIDATE") + self.assertIn("does not exist", rows[name]["reason"]) self.assertEqual(di.validate_report(rows.values()), []) - def test_depth_limit_and_links_are_not_followed(self) -> None: - deep = self.mkdir("root/a/b/c") - os.symlink(self.root / "gone", deep / "far") - os.symlink(self.root / "root", self.root / "root" / "a" / "loop") - self.assertEqual(di.dangling_symlinks([self.root / "root"], max_depth=2), []) - self.assertEqual( - len(di.dangling_symlinks([self.root / "root"], max_depth=8)), 1 + def test_a_link_to_a_directory_is_not_followed(self) -> None: + root = self.mkdir("root") + self.write("root/a/file.txt") + os.symlink(root, root / "a" / "loop") + rows = di.inventory_rows( + root, + deep=True, + home=self.root / "no-home", + tmp_dir=self.root / "no-tmp", + now=NOW, ) + names = [Path(r["name"]).relative_to(root).as_posix() for r in rows] + self.assertEqual(sorted(names), [".", "a", "a/file.txt", "a/loop"]) class ReadOnlyTest(TempTree): @@ -461,7 +495,7 @@ def snapshot() -> list[tuple[str, float]]: di.superseded_versions([self.root / "v"]) di.project_transcripts(self.root / "claude" / "projects", self.root) di.plugin_cache_versions(self.root / "claude") - di.dangling_symlinks([self.root]) + di.dangling_row(self.root / "tmp" / "dangling") self.assertEqual(snapshot(), before) diff --git a/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py b/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py index abbe81c64c..77af863a31 100755 --- a/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py +++ b/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py @@ -12900,6 +12900,12 @@ def setUp(self) -> None: self.enterContext( mock.patch.object(hygiene, "is_os_managed_target", return_value=False) ) + # The process table is /proc, which macOS and Windows lack. + self.running = self.enterContext( + mock.patch.object( + hygiene.deep_inventory, "running_paths", return_value=set() + ) + ) def write(self, relative: str, text: str = "x") -> Path: path = self.target / relative @@ -12963,6 +12969,20 @@ def test_deep_lists_every_level_with_bottom_up_sizes(self) -> None: set(hygiene.deep_inventory.ROW_COLUMNS) - {"evidence"}, set(row) ) + def test_unread_process_table_leaves_superseded_versions_unknown(self) -> None: + self.write("tool/1.2.0/bin", "old") + self.write("tool/1.10.0/bin", "new") + self.running.return_value = None + code, summary = self.run_inventory("--deep") + self.assertEqual(0, code, summary) + rows = self.rows(summary) + old = rows[str(self.target / "tool" / "1.2.0")] + self.assertEqual("UNKNOWN", old["disposition"]) + self.assertIn("process table was not read", old["reason"]) + self.assertEqual( + "KEEP", rows[str(self.target / "tool" / "1.10.0")]["disposition"] + ) + def test_year_named_folders_are_not_versions(self) -> None: self.write("Pictures/2024/a.jpg") self.write("Pictures/2025/b.jpg") From d0f85a3fb35aa45d95101bb428a0d2fb0c11f61b Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Wed, 30 Sep 2026 13:13:30 -0400 Subject: [PATCH 7/9] docs(disk-hygiene): document --deep in the README and scope the /tmp category The README lists the scan flags but not --deep. scan-flags.md now says the tmp-producer category yields rows only when the target is /tmp or contains it, so a home inventory has none and /tmp is inventoried as its own target. Co-Authored-By: Claude Opus 5.5 --- plugins/disk-hygiene/README.md | 6 ++++++ plugins/disk-hygiene/skills/clean/reference/scan-flags.md | 7 ++++--- 2 files changed, 10 insertions(+), 3 deletions(-) diff --git a/plugins/disk-hygiene/README.md b/plugins/disk-hygiene/README.md index 41f9c4bdbd..a039a8ea85 100644 --- a/plugins/disk-hygiene/README.md +++ b/plugins/disk-hygiene/README.md @@ -240,6 +240,12 @@ gets the relaxed directory listing. large-scan confirmation, sums through VCS and protected directories read-only, and has no entry cap. +`--deep`, and a home-directory target without it, runs the read-only deep inventory before any +scan: every entry with its producer, a disposition and a reason, where each `KEEP` names who +produced the entry and what still uses it. It reports only and prepares no deletion; removing +anything it lists still goes through scan, preview and the removal approval. Columns and +categories: `skills/clean/reference/scan-flags.md`. + The skill stores snapshots, plans, and reports under `${CLAUDE_PLUGIN_DATA}`. It never writes generated state into the installed plugin directory or the audited target. diff --git a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md index 99b1ef0454..adb1b0b5e4 100644 --- a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md +++ b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md @@ -70,9 +70,10 @@ rather than a symlink (an nvm alias, `.tool-versions`) is not seen, so its row c sweep window (`ORPHAN_SWEEP_DAYS` in `scripts/deep_inventory.py`), in `evidence` and in the reason, so a version Claude Code removes itself reads differently from one it has not. -`tmp-producer` covers the entries of `/tmp` itself, not `$TMPDIR`, and runs only where `/tmp` is an -ordinary directory a target can name (Linux): macOS rejects `/tmp` (a link) and `/private/tmp` (an -OS-managed root), so the category has no rows there. +`tmp-producer` covers the entries of `/tmp` itself, not `$TMPDIR`, and produces rows only when the +target is `/tmp` or contains it, so a home inventory has none: run `--deep /tmp` as its own +target. It runs only where `/tmp` is an ordinary directory a target can name (Linux): macOS rejects +`/tmp` (a link) and `/private/tmp` (an OS-managed root), so the category has no rows there. `superseded-version` and `tmp-producer` also keep an entry a running process uses; where `/proc` cannot be read (macOS, Windows), a row that would be a `CANDIDATE` on that basis is `UNKNOWN`, From e14439470bea467fb00d435dc39f83253e138686 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Wed, 30 Sep 2026 13:59:38 -0400 Subject: [PATCH 8/9] fix(disk-hygiene): reject an empty KEEP reason that only carries evidence, and pass lint The deep-inventory validator let a KEEP row with an empty reason pass whenever its evidence named any tool and a reference, and let "tool-managed" pass with evidence for any tool. An empty reason now always fails, and a category-only reason passes only when it names the tool ("managed by ") that the evidence shows still references the entry. Also: mark deep_inventory.py and test_deep_inventory.py executable (they carry shebangs), move the test fixture paths outside /home, drop the duplicate sys.path insert in the engine, and give ORPHAN_SWEEP_DAYS and PROJECT_NAME_CAP a verification stamp and recheck trigger. Refs #5221 Co-Authored-By: Claude Opus 5.5 --- .../skills/clean/reference/safety-model.md | 4 +- .../skills/clean/reference/scan-flags.md | 5 +- .../skills/clean/scripts/deep_inventory.py | 52 ++++++++++++------- .../skills/clean/scripts/hygiene.py | 4 -- .../clean/scripts/test_deep_inventory.py | 51 +++++++++++------- 5 files changed, 71 insertions(+), 45 deletions(-) mode change 100644 => 100755 plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py mode change 100644 => 100755 plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py diff --git a/plugins/disk-hygiene/skills/clean/reference/safety-model.md b/plugins/disk-hygiene/skills/clean/reference/safety-model.md index 31449f5e1a..1033db574e 100644 --- a/plugins/disk-hygiene/skills/clean/reference/safety-model.md +++ b/plugins/disk-hygiene/skills/clean/reference/safety-model.md @@ -61,8 +61,8 @@ snapshot or a plan, so `preview` refuses them and no approval token can derive f because it cannot mutate, not because its output authorizes anything. Every `KEEP` row must carry a specific reason (who produced the entry and what still uses it). The -validator rejects an empty reason and a bare category phrase unless `evidence` shows the named tool -still references the entry, and a rejected row fails the report with exit 5. +validator rejects an empty reason, and a bare category phrase unless it names a tool and `evidence` +shows that tool still references the entry. A rejected row fails the report with exit 5. `--execute`, the low-signal rule (Low is kept unless the human separately reviews exact paths), and every confirmation gate apply exactly as before. Removing anything the inventory lists goes through diff --git a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md index 4c1546fcf9..67cd83ed8d 100644 --- a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md +++ b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md @@ -45,8 +45,9 @@ JSONL report that `preview` and `apply` refuse. The gates are unchanged: [safety-model.md](safety-model.md#deep-inventory-is-report-only). Every `KEEP` row needs a specific reason: who produced the entry and what still uses it. The -validator fails an empty reason or one that is only a category phrase ("tool-managed", "OS-owned", -"managed by ") unless `evidence` shows the named tool still references the entry. +validator always fails an empty reason. It fails a reason that is only a category phrase +("tool-managed", "OS-owned", "managed by ") unless the phrase names the tool ("managed by +") and `evidence` shows that tool still references the entry. Rows stream to `/inventory/inventory-.jsonl` with no entry cap, beside a `.json` summary (`deep-inventory-report`; status `inventory-failed` and exit 5 when the validator rejects a diff --git a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py old mode 100644 new mode 100755 index 8b7d16a5e0..5880e4ad3e --- a/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py +++ b/plugins/disk-hygiene/skills/clean/scripts/deep_inventory.py @@ -93,14 +93,17 @@ VERSION_RE = re.compile(r"^v?(\d+(?:\.\d+)*)(?:([-+])[\w.+-]+)?$") # Days after an update or uninstall that Claude Code removes an orphaned plugin # version, counted from its `.orphaned_at` marker. Basis: -# https://code.claude.com/docs/en/plugins/loading.md ("Cleanup of previous versions"). +# https://code.claude.com/docs/en/plugins/loading.md ("Cleanup of previous versions"), +# verified 2026-09-30; recheck when that section or a Claude Code changelog entry +# changes the window. ORPHAN_SWEEP_DAYS = 14 _TOKEN = r"[\w.@/+-]+" +_NAMED_TOOL = re.compile( + rf"(?:managed|owned) by (?:the )?({_TOKEN}(?: {_TOKEN}){{0,2}})" +) _CATEGORY_PHRASE = re.compile( - rf"(?:(?:tool|os|system|app|vendor)[- ]?(?:managed|owned)" - rf"|(?:managed|owned) by (?:the )?{_TOKEN}(?: {_TOKEN}){{0,2}})" + rf"(?:(?:tool|os|system|app|vendor)[- ]?(?:managed|owned)|{_NAMED_TOOL.pattern})" ) -_MANAGED_BY = re.compile(r"managed by ") def _iso(epoch: float) -> str: @@ -554,8 +557,12 @@ def walk(base: Path, rest: str) -> Path | None: return None if found is None else "/" + found.relative_to(fs_root).as_posix() -# Claude Code caps an encoded directory name near this length and appends a hash, -# after which the source path cannot be recovered from the name. +# Claude Code keeps the first 200 characters of an encoded project directory name +# and appends `-`, after which the source path cannot be recovered from the +# name. Basis: the path-sanitizing function in the Claude Code 2.1.285 binary +# (non-alphanumerics become `-`, names over 200 characters are cut and hashed), +# verified 2026-09-30; recheck when a Claude Code changelog entry mentions +# project directory naming or a release changes the encoding. PROJECT_NAME_CAP = 200 @@ -780,14 +787,16 @@ def _category_only(reason: str) -> bool: def _shows_reference(reason: str, evidence: object) -> bool: - if not ( - isinstance(evidence, dict) - and str(evidence.get("tool") or "").strip() - and str(evidence.get("references") or "").strip() - ): + """True when the reason names a tool ("managed by ") that evidence shows references the entry.""" + if not isinstance(evidence, dict): return False - return not _MANAGED_BY.search(reason.lower()) or ( - str(evidence["tool"]).strip().lower() in reason.lower() + tool = str(evidence.get("tool") or "").strip().lower() + return bool( + tool + and str(evidence.get("references") or "").strip() + and any( + tool in m.group(1).split() for m in _NAMED_TOOL.finditer(reason.lower()) + ) ) @@ -795,9 +804,11 @@ def validate_report(rows: Iterable[dict[str, Any]]) -> list[str]: """Failures of a report; an empty list means it passes. Every row needs the schema columns except the optional ``evidence`` and a - valid disposition. A KEEP row fails when its reason is empty or only a - category phrase ("tool-managed", "OS-owned", "managed by ") unless its - evidence shows the named tool still references the entry. + valid disposition. A KEEP row fails when its reason is empty, which names no + tool, so no evidence can stand in for it. It also fails when its reason is + only a category phrase ("tool-managed", "OS-owned", "managed by "), + unless the phrase names a tool ("managed by ") and the row's evidence + shows that tool still references the entry. """ failures = [] for index, row in enumerate(rows): @@ -814,10 +825,13 @@ def validate_report(rows: Iterable[dict[str, Any]]) -> list[str]: if row["disposition"] != "KEEP": continue reason = str(row["reason"] or "") - if _category_only(reason) and not _shows_reference(reason, row.get("evidence")): + if not reason.strip(): + failures.append(f"{label}: KEEP reason is empty") + elif _category_only(reason) and not _shows_reference( + reason, row.get("evidence") + ): failures.append( - f"{label}: KEEP reason {reason!r} is " - f"{'empty' if not reason.strip() else 'only a category phrase'} " + f"{label}: KEEP reason {reason!r} is only a category phrase " "and no evidence shows the named tool still references the entry" ) return failures diff --git a/plugins/disk-hygiene/skills/clean/scripts/hygiene.py b/plugins/disk-hygiene/skills/clean/scripts/hygiene.py index 9692a1c34b..9016acc042 100755 --- a/plugins/disk-hygiene/skills/clean/scripts/hygiene.py +++ b/plugins/disk-hygiene/skills/clean/scripts/hygiene.py @@ -35,10 +35,6 @@ import engine_grammar # noqa: E402 (path set above; plugin-bundled module) import investigated_catalog # noqa: E402 (sibling module; a record is a hint only) - -if str(Path(__file__).resolve().parent) not in sys.path: - sys.path.insert(0, str(Path(__file__).resolve().parent)) - import deep_inventory # noqa: E402 (path set above; sibling module) MIN_PYTHON = (3, 11) diff --git a/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py b/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py old mode 100644 new mode 100755 index 555d1d76bb..bdba2e2fea --- a/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py +++ b/plugins/disk-hygiene/skills/clean/scripts/test_deep_inventory.py @@ -124,16 +124,33 @@ def test_category_phrase_with_more_detail_passes(self) -> None: with self.subTest(reason=reason): self.assertEqual(self.failures(reason), []) - def test_evidence_that_the_tool_references_the_entry_rescues_a_category_reason( + def test_evidence_that_the_named_tool_references_the_entry_rescues_the_reason( self, ) -> None: evidence = {"tool": "mise", "references": "trusted-configs/abc"} self.assertEqual(self.failures("managed by mise", evidence), []) - self.assertEqual(self.failures("tool-managed", evidence), []) + self.assertEqual(self.failures("tool-managed; managed by mise", evidence), []) + + def test_a_reason_that_names_no_tool_is_not_rescued_by_evidence(self) -> None: + evidence = {"tool": "mise", "references": "trusted-configs/abc"} + for reason in ("tool-managed", "OS-owned", "app owned and vendor-managed"): + with self.subTest(reason=reason): + self.assertEqual(len(self.failures(reason, evidence)), 1) + + def test_an_empty_reason_is_not_rescued_by_evidence(self) -> None: + evidence = {"tool": "npm", "references": "/y"} + for reason in ("", " ", None): + with self.subTest(reason=reason): + failures = self.failures(reason, evidence) + self.assertEqual(len(failures), 1) + self.assertIn("empty", failures[0]) def test_evidence_for_a_different_tool_does_not_rescue(self) -> None: evidence = {"tool": "npm", "references": "package.json"} self.assertEqual(len(self.failures("managed by mise", evidence)), 1) + self.assertEqual( + len(self.failures("managed by promise", {**evidence, "tool": "mise"})), 1 + ) def test_incomplete_evidence_does_not_rescue(self) -> None: for evidence in ( @@ -143,7 +160,7 @@ def test_incomplete_evidence_does_not_rescue(self) -> None: "mise", ): with self.subTest(evidence=evidence): - self.assertEqual(len(self.failures("tool-managed", evidence)), 1) + self.assertEqual(len(self.failures("managed by mise", evidence)), 1) def test_only_keep_rows_need_a_reason(self) -> None: for disposition in ("CANDIDATE", "UNKNOWN"): @@ -168,7 +185,7 @@ class RunningPathsTest(TempTree): def test_reads_exe_and_cwd_and_drops_the_deleted_suffix(self) -> None: proc = self.mkdir("proc") for pid, exe, cwd in ( - ("10", "/opt/tool/1.0/bin (deleted)", "/home/u"), + ("10", "/opt/tool/1.0/bin (deleted)", "/var/app"), ("11", "/opt/tool/2.0/bin", "/srv"), ): (proc / pid).mkdir() @@ -177,7 +194,7 @@ def test_reads_exe_and_cwd_and_drops_the_deleted_suffix(self) -> None: (proc / "self-note").mkdir() self.assertEqual( di.running_paths(proc), - {"/opt/tool/1.0/bin", "/home/u", "/opt/tool/2.0/bin", "/srv"}, + {"/opt/tool/1.0/bin", "/var/app", "/opt/tool/2.0/bin", "/srv"}, ) def test_unreadable_proc_root_is_none_not_empty(self) -> None: @@ -469,8 +486,8 @@ class ProjectTranscriptsTest(TempTree): def setUp(self) -> None: super().setUp() self.fs = self.root / "fs" - self.mkdir("fs/home/kyle/my.repo/sub") - self.mkdir("fs/home/kyle/.config") + self.mkdir("fs/srv/work/my.repo/sub") + self.mkdir("fs/srv/work/.config") self.projects = self.mkdir("claude/projects") def rows(self, *names: str) -> dict[str, dict]: @@ -480,28 +497,26 @@ def rows(self, *names: str) -> dict[str, dict]: def test_decodes_by_the_directory_tree_not_by_guessing_separators(self) -> None: rows = self.rows( - "-home-kyle-my-repo-sub", "-home-kyle--config", "-home-kyle-my-repo" + "-srv-work-my-repo-sub", "-srv-work--config", "-srv-work-my-repo" ) for name, source in ( - ("-home-kyle-my-repo-sub", "/home/kyle/my.repo/sub"), - ("-home-kyle--config", "/home/kyle/.config"), - ("-home-kyle-my-repo", "/home/kyle/my.repo"), + ("-srv-work-my-repo-sub", "/srv/work/my.repo/sub"), + ("-srv-work--config", "/srv/work/.config"), + ("-srv-work-my-repo", "/srv/work/my.repo"), ): self.assertEqual(rows[name]["disposition"], "KEEP", name) self.assertEqual(rows[name]["evidence"]["references"], source) self.assertEqual(di.validate_report(rows.values()), []) def test_a_gone_source_path_is_a_candidate(self) -> None: - rows = self.rows("-tmp-harness-run-7", "-home-kyle-deleted-repo") - for name in ("-tmp-harness-run-7", "-home-kyle-deleted-repo"): + rows = self.rows("-tmp-harness-run-7", "-srv-work-deleted-repo") + for name in ("-tmp-harness-run-7", "-srv-work-deleted-repo"): self.assertEqual(rows[name]["disposition"], "CANDIDATE", name) self.assertEqual(rows[name]["producer"], "claude-code") def test_a_prefix_only_match_is_not_a_source(self) -> None: - rows = self.rows("-home-kyle-my-repo-sub-gone") - self.assertEqual( - rows["-home-kyle-my-repo-sub-gone"]["disposition"], "CANDIDATE" - ) + rows = self.rows("-srv-work-my-repo-sub-gone") + self.assertEqual(rows["-srv-work-my-repo-sub-gone"]["disposition"], "CANDIDATE") def test_undecodable_names_are_unknown(self) -> None: rows = self.rows("C--Users-kyle", "-" + "a" * di.PROJECT_NAME_CAP) @@ -510,7 +525,7 @@ def test_undecodable_names_are_unknown(self) -> None: def test_decode_project_returns_none_for_missing(self) -> None: self.assertIsNone(di.decode_project("-nope", self.fs)) self.assertEqual( - di.decode_project("-home-kyle-my-repo", self.fs), "/home/kyle/my.repo" + di.decode_project("-srv-work-my-repo", self.fs), "/srv/work/my.repo" ) From 19eb0aa7df8505019acc232f48d16571dd06a43f Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:10:17 -0400 Subject: [PATCH 9/9] fix(disk-hygiene): report an unreadable mountinfo in the inventory summary The inventory summary carries mount_state_error when /proc/self/mountinfo cannot be read, so a report that walked without bind-mount detection does not read as a complete one. Refs: #5221 Co-Authored-By: Claude Opus 5.5 --- .../skills/clean/reference/scan-flags.md | 6 ++++-- .../skills/clean/scripts/hygiene.py | 4 +++- .../skills/clean/scripts/test_hygiene.py | 17 +++++++++++++++++ 3 files changed, 24 insertions(+), 3 deletions(-) diff --git a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md index a24dba7878..ffdce85c1e 100644 --- a/plugins/disk-hygiene/skills/clean/reference/scan-flags.md +++ b/plugins/disk-hygiene/skills/clean/reference/scan-flags.md @@ -58,8 +58,10 @@ files beneath it. Named categories: `superseded-version` (sibling entries, direc under a parent that holds two or more dotted version names), `plugin-cache-version` (cache versions no installed plugin references), `tmp-producer` (`/tmp` entries by producer prefix), `transcript-dir` (project transcript directories whose source path is gone), `dangling-symlink`, -and `not-walked` (an unreadable subtree, or a mount point including a bind mount, one `UNKNOWN` row). Every other entry is -`unclassified`. +and `not-walked` (an unreadable subtree, or a mount point including a bind mount, one `UNKNOWN` +row). Every other entry is `unclassified`. Bind mounts come from `/proc/self/mountinfo`; when that +cannot be read the summary carries the reason in `mount_state_error` (otherwise null) and only a +mount on another device is skipped. `superseded-version` keeps the newest version (a release outranks its own prerelease), a version a running process executes, and a version a symlink points at, read from the symlinks beside the diff --git a/plugins/disk-hygiene/skills/clean/scripts/hygiene.py b/plugins/disk-hygiene/skills/clean/scripts/hygiene.py index ac2a3955ec..4059c5f67a 100755 --- a/plugins/disk-hygiene/skills/clean/scripts/hygiene.py +++ b/plugins/disk-hygiene/skills/clean/scripts/hygiene.py @@ -4948,7 +4948,8 @@ def run_inventory(target_arg: str, deep_flag: bool) -> int: dispositions: dict[str, int] = {} failures: list[str] = [] partial = rows_path.with_name(f"{rows_path.name}.{secrets.token_hex(4)}.tmp") - mounts = frozenset(str(p) for p in linux_mount_points()[0]) + mount_points, mount_error = linux_mount_points() + mounts = frozenset(str(p) for p in mount_points) try: with partial.open("w", encoding="utf-8") as out: for row in deep_inventory.inventory_rows( @@ -4977,6 +4978,7 @@ def run_inventory(target_arg: str, deep_flag: bool) -> int: "row_count": sum(dispositions.values()), "dispositions": dispositions, "validation_failures": failures, + "mount_state_error": mount_error, "note": ( "Report only: rows are findings, never a deletion plan, and preview " "and apply do not accept this report. Removing a candidate goes " diff --git a/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py b/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py index ac61708348..5e54e26fad 100755 --- a/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py +++ b/plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py @@ -15495,6 +15495,23 @@ def test_mount_point_on_the_same_device_is_not_entered(self) -> None: self.assertNotIn(str(self.target / "bind" / "inside.txt"), rows) self.assertIn(str(self.target / "plain" / "file.txt"), rows) + def test_unreadable_mountinfo_is_reported_not_hidden(self) -> None: + self.write("a.txt") + with mock.patch.object( + hygiene, "linux_mount_points", return_value=(set(), "cannot read mountinfo") + ): + code, summary = self.run_inventory("--deep") + self.assertEqual(0, code, summary) + self.assertEqual("cannot read mountinfo", summary["mount_state_error"]) + + def test_readable_mountinfo_reports_no_error(self) -> None: + self.write("a.txt") + with mock.patch.object( + hygiene, "linux_mount_points", return_value=({self.target / "x"}, None) + ): + _, summary = self.run_inventory("--deep") + self.assertIsNone(summary["mount_state_error"]) + def test_an_interrupted_walk_leaves_no_partial_report(self) -> None: self.write("a.txt") with mock.patch.object(