diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..dfe0770 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,2 @@ +# Auto detect text files and perform LF normalization +* text=auto diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 0000000..b501d8a --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,13 @@ +version: 2 +updates: + - package-ecosystem: "pip" + directory: "/" + schedule: + interval: "weekly" + open-pull-requests-limit: 10 + + - package-ecosystem: "github-actions" + directory: "/" + schedule: + interval: "weekly" + open-pull-requests-limit: 5 diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml new file mode 100644 index 0000000..1e1bd56 --- /dev/null +++ b/.github/workflows/codeql.yml @@ -0,0 +1,32 @@ +name: CodeQL + +on: + push: + branches: [main] + pull_request: + branches: [main] + schedule: + - cron: '23 8 * * 1' + +permissions: + actions: read + contents: read + security-events: write + +jobs: + analyze: + name: CodeQL (Python) + runs-on: ubuntu-latest + timeout-minutes: 15 + + steps: + - name: Checkout + uses: actions/checkout@v6 + + - name: Initialize CodeQL + uses: github/codeql-action/init@v4 + with: + languages: python + + - name: Perform CodeQL Analysis + uses: github/codeql-action/analyze@v4 diff --git a/.github/workflows/shellcheck.yml b/.github/workflows/shellcheck.yml new file mode 100644 index 0000000..6a524a0 --- /dev/null +++ b/.github/workflows/shellcheck.yml @@ -0,0 +1,56 @@ +name: Shellcheck + +# Catches shell-script bugs in the hook templates before they hit users. +# Especially critical for the preToolUse hook: Copilot CLI >= 1.0.57 treats a +# non-zero exit as a tool-call DENY, so any regression that re-introduces +# strict mode or removes the trap pyramid would block every ShadowFrog user. + +on: + pull_request: + branches: [main] + paths: + - 'hook-templates/**' + - 'install.sh' + - 'skills/**/*.sh' + - '.github/workflows/shellcheck.yml' + +permissions: + contents: read + +jobs: + shellcheck: + name: shellcheck + runs-on: ubuntu-latest + timeout-minutes: 5 + + steps: + - name: Checkout + uses: actions/checkout@v6 + + - name: Install shellcheck + run: sudo apt-get update && sudo apt-get install -y shellcheck + + - name: Lint hook templates (strict, info-level) + run: | + fail=0 + for f in hook-templates/scripts/*.sh; do + echo "::group::shellcheck (info) $f" + shellcheck --severity=info "$f" || fail=1 + echo "::endgroup::" + done + exit $fail + + - name: Lint installer and bundled scripts (warning-level) + run: | + fail=0 + for f in install.sh $(find skills -name '*.sh'); do + [ -f "$f" ] || continue + echo "::group::shellcheck (warning) $f" + shellcheck --severity=warning "$f" || fail=1 + echo "::endgroup::" + done + exit $fail + + - name: Guard against fail-closed regressions in hook templates + run: | + python3 hook-templates/check-hook-failopen.py hook-templates/scripts/*.sh diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml new file mode 100644 index 0000000..85fc90c --- /dev/null +++ b/.github/workflows/tests.yml @@ -0,0 +1,46 @@ +name: Tests + +on: + pull_request: + branches: [main] + +permissions: + contents: read + +jobs: + pytest: + name: pytest (Python 3.12) + runs-on: ubuntu-latest + timeout-minutes: 10 + + steps: + - name: Checkout + uses: actions/checkout@v6 + + - name: Set up Python + uses: actions/setup-python@v6 + with: + python-version: '3.12' + cache: pip + cache-dependency-path: requirements-dev.txt + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -r requirements-dev.txt + + - name: Run pytest with coverage + run: | + python -m pytest \ + --cov=skills \ + --cov-report=term \ + --cov-report=xml \ + -q + + - name: Upload coverage report + if: always() + uses: actions/upload-artifact@v7 + with: + name: coverage-report + path: coverage.xml + if-no-files-found: ignore diff --git a/.gitignore b/.gitignore index d5a18de..86db3e0 100644 --- a/.gitignore +++ b/.gitignore @@ -1,429 +1,37 @@ -## Ignore Visual Studio temporary files, build results, and -## files generated by popular Visual Studio add-ons. -## -## Get latest from https://github.com/github/gitignore/blob/main/VisualStudio.gitignore - -# User-specific files -*.rsuser -*.suo -*.user -*.userosscache -*.sln.docstates -*.env - -# User-specific files (MonoDevelop/Xamarin Studio) -*.userprefs - -# Mono auto generated files -mono_crash.* - -# Build results -[Dd]ebug/ -[Dd]ebugPublic/ -[Rr]elease/ -[Rr]eleases/ - -[Dd]ebug/x64/ -[Dd]ebugPublic/x64/ -[Rr]elease/x64/ -[Rr]eleases/x64/ -bin/x64/ -obj/x64/ - -[Dd]ebug/x86/ -[Dd]ebugPublic/x86/ -[Rr]elease/x86/ -[Rr]eleases/x86/ -bin/x86/ -obj/x86/ - -[Ww][Ii][Nn]32/ -[Aa][Rr][Mm]/ -[Aa][Rr][Mm]64/ -[Aa][Rr][Mm]64[Ee][Cc]/ -bld/ -[Oo]bj/ -[Oo]ut/ -[Ll]og/ -[Ll]ogs/ - -# Build results on 'Bin' directories -**/[Bb]in/* -# Uncomment if you have tasks that rely on *.refresh files to move binaries -# (https://github.com/github/gitignore/pull/3736) -#!**/[Bb]in/*.refresh - -# Visual Studio 2015/2017 cache/options directory -.vs/ -# Uncomment if you have tasks that create the project's static files in wwwroot -#wwwroot/ - -# Visual Studio 2017 auto generated files -Generated\ Files/ - -# MSTest test Results -[Tt]est[Rr]esult*/ -[Bb]uild[Ll]og.* -*.trx - -# NUnit -*.VisualState.xml -TestResult.xml -nunit-*.xml - -# Approval Tests result files -*.received.* - -# Build Results of an ATL Project -[Dd]ebugPS/ -[Rr]eleasePS/ -dlldata.c - -# Benchmark Results -BenchmarkDotNet.Artifacts/ - -# .NET Core -project.lock.json -project.fragment.lock.json -artifacts/ -.artifacts/ - -# ASP.NET Scaffolding -ScaffoldingReadMe.txt - -# StyleCop -StyleCopReport.xml - -# Files built by Visual Studio -*_i.c -*_p.c -*_h.h -*.ilk -*.meta -*.obj -*.idb -*.iobj -*.pch -*.pdb -*.ipdb -*.pgc -*.pgd -*.rsp -# but not Directory.Build.rsp, as it configures directory-level build defaults -!Directory.Build.rsp -*.sbr -*.tlb -*.tli -*.tlh -*.tmp -*.tmp_proj -*_wpftmp.csproj -*.log -*.tlog -*.vspscc -*.vssscc -.builds -*.pidb -*.svclog -*.scc - -# Chutzpah Test files -_Chutzpah* - -# Visual C++ cache files -ipch/ -*.aps -*.ncb -*.opendb -*.opensdf -*.sdf -*.cachefile -*.VC.db -*.VC.VC.opendb - -# Visual Studio profiler -*.psess -*.vsp -*.vspx -*.sap - -# Visual Studio Trace Files -*.e2e - -# TFS 2012 Local Workspace -$tf/ - -# Guidance Automation Toolkit -*.gpState - -# ReSharper is a .NET coding add-in -_ReSharper*/ -*.[Rr]e[Ss]harper -*.DotSettings.user - -# TeamCity is a build add-in -_TeamCity* - -# DotCover is a Code Coverage Tool -*.dotCover - -# AxoCover is a Code Coverage Tool -.axoCover/* -!.axoCover/settings.json - -# Coverlet is a free, cross platform Code Coverage Tool -coverage*.json -coverage*.xml -coverage*.info - -# Visual Studio code coverage results -*.coverage -*.coveragexml - -# NCrunch -_NCrunch_* -.NCrunch_* -.*crunch*.local.xml -nCrunchTemp_* - -# MightyMoose -*.mm.* -AutoTest.Net/ - -# Web workbench (sass) -.sass-cache/ - -# Installshield output folder -[Ee]xpress/ - -# DocProject is a documentation generator add-in -DocProject/buildhelp/ -DocProject/Help/*.HxT -DocProject/Help/*.HxC -DocProject/Help/*.hhc -DocProject/Help/*.hhk -DocProject/Help/*.hhp -DocProject/Help/Html2 -DocProject/Help/html - -# Click-Once directory -publish/ - -# Publish Web Output -*.[Pp]ublish.xml -*.azurePubxml -# Note: Comment the next line if you want to checkin your web deploy settings, -# but database connection strings (with potential passwords) will be unencrypted -*.pubxml -*.publishproj - -# Microsoft Azure Web App publish settings. Comment the next line if you want to -# checkin your Azure Web App publish settings, but sensitive information contained -# in these scripts will be unencrypted -PublishScripts/ - -# NuGet Packages -*.nupkg -# NuGet Symbol Packages -*.snupkg -# The packages folder can be ignored because of Package Restore -**/[Pp]ackages/* -# except build/, which is used as an MSBuild target. -!**/[Pp]ackages/build/ -# Uncomment if necessary however generally it will be regenerated when needed -#!**/[Pp]ackages/repositories.config -# NuGet v3's project.json files produces more ignorable files -*.nuget.props -*.nuget.targets - -# Microsoft Azure Build Output -csx/ -*.build.csdef - -# Microsoft Azure Emulator -ecf/ -rcf/ - -# Windows Store app package directories and files -AppPackages/ -BundleArtifacts/ -Package.StoreAssociation.xml -_pkginfo.txt -*.appx -*.appxbundle -*.appxupload - -# Visual Studio cache files -# files ending in .cache can be ignored -*.[Cc]ache -# but keep track of directories ending in .cache -!?*.[Cc]ache/ - -# Others -ClientBin/ -~$* +# OS +.DS_Store +Thumbs.db + +# IDE +.idea/ +.vscode/ +*.swp +*.swo *~ -*.dbmdl -*.dbproj.schemaview -*.jfm -*.pfx -*.publishsettings -orleans.codegen.cs - -# Including strong name files can present a security risk -# (https://github.com/github/gitignore/pull/2483#issue-259490424) -#*.snk - -# Since there are multiple workflows, uncomment next line to ignore bower_components -# (https://github.com/github/gitignore/pull/1529#issuecomment-104372622) -#bower_components/ - -# RIA/Silverlight projects -Generated_Code/ - -# Backup & report files from converting an old project file -# to a newer Visual Studio version. Backup files are not needed, -# because we have git ;-) -_UpgradeReport_Files/ -Backup*/ -UpgradeLog*.XML -UpgradeLog*.htm -ServiceFabricBackup/ -*.rptproj.bak - -# SQL Server files -*.mdf -*.ldf -*.ndf - -# Business Intelligence projects -*.rdl.data -*.bim.layout -*.bim_*.settings -*.rptproj.rsuser -*- [Bb]ackup.rdl -*- [Bb]ackup ([0-9]).rdl -*- [Bb]ackup ([0-9][0-9]).rdl -# Microsoft Fakes -FakesAssemblies/ +# Python (for any helper scripts) +__pycache__/ +*.py[cod] +.mypy_cache/ +.pytest_cache/ -# GhostDoc plugin setting file -*.GhostDoc.xml - -# Node.js Tools for Visual Studio -.ntvs_analysis.dat +# Node node_modules/ -# Visual Studio 6 build log -*.plg - -# Visual Studio 6 workspace options file -*.opt - -# Visual Studio 6 auto-generated workspace file (contains which files were open etc.) -*.vbw - -# Visual Studio 6 workspace and project file (working project files containing files to include in project) -*.dsw -*.dsp - -# Visual Studio 6 technical files -*.ncb -*.aps - -# Visual Studio LightSwitch build output -**/*.HTMLClient/GeneratedArtifacts -**/*.DesktopClient/GeneratedArtifacts -**/*.DesktopClient/ModelManifest.xml -**/*.Server/GeneratedArtifacts -**/*.Server/ModelManifest.xml -_Pvt_Extensions - -# Paket dependency manager -**/.paket/paket.exe -paket-files/ - -# FAKE - F# Make -**/.fake/ - -# CodeRush personal settings -**/.cr/personal - -# Python Tools for Visual Studio (PTVS) -**/__pycache__/ -*.pyc - -# Cake - Uncomment if you are using it -#tools/** -#!tools/packages.config - -# Tabs Studio -*.tss - -# Telerik's JustMock configuration file -*.jmconfig - -# BizTalk build output -*.btp.cs -*.btm.cs -*.odx.cs -*.xsd.cs - -# OpenCover UI analysis results -OpenCover/ - -# Azure Stream Analytics local run output -ASALocalRun/ - -# MSBuild Binary and Structured Log -*.binlog -MSBuild_Logs/ - -# AWS SAM Build and Temporary Artifacts folder -.aws-sam - -# NVidia Nsight GPU debugger configuration file -*.nvuser - -# MFractors (Xamarin productivity tool) working folder -**/.mfractor/ - -# Local History for Visual Studio -**/.localhistory/ - -# Visual Studio History (VSHistory) files -.vshistory/ - -# BeatPulse healthcheck temp database -healthchecksdb - -# Backup folder for Package Reference Convert tool in Visual Studio 2017 -MigrationBackup/ - -# Ionide (cross platform F# VS Code tools) working folder -**/.ionide/ - -# Fody - auto-generated XML schema -FodyWeavers.xsd - -# VS Code files for those working on multiple tools -.vscode/* -!.vscode/settings.json -!.vscode/tasks.json -!.vscode/launch.json -!.vscode/extensions.json -!.vscode/*.code-snippets - -# Local History for Visual Studio Code -.history/ - -# Built Visual Studio Code Extensions -*.vsix - -# Windows Installer files from build outputs -*.cab -*.msi -*.msix -*.msm -*.msp +# Generated hooks (installed per-project by skenv) +.github/hooks/ + +# Transient flag file (created by sessionStart hook) +.shadow-frog-needs-init +.shadow/ +!examples/**/.shadow/ +eval-results/ +eval/results_dashboard.png +__pycache__/ +.agent_workdirs/ +eval/swebench-fix/reports/legacy/ +eval/feature-ideation/logs/ + +# Coverage +.coverage +.coverage.* diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..008773d --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,977 @@ +# Changelog + +All notable changes to ShadowFrog are documented here. + +ShadowFrog is a suite of AI coding agent skills that build and maintain +shadow knowledge bases for any codebase. + +--- + +## 2026-06-03 + +### Changed +- **`preToolUse.matcher` now filters at the CLI layer.** The hook used to + fire on every tool call (Read, Bash, glob, …) and filter mutations + internally in bash. With the matcher fix in Copilot CLI 1.0.36, the CLI + itself can now skip the hook for non-mutating tools, eliminating the + per-call subprocess overhead for the majority of tool invocations. + Matcher pattern covers both Copilot CLI lowercase + (`edit`/`create`/`str_replace`/`write`/`multiedit`/`notebookedit`) and + Claude Code PascalCase (`Edit`/`Write`/`MultiEdit`/`NotebookEdit`). + Backwards-compatible: on Copilot CLI < 1.0.36 the matcher field is + ignored and the hook fires on every call (same as before). + +--- + +## 2026-06-02 + +### Added (additional hook hardening) +Follow-up coverage pass before release surfaced sixteen further fixes +across the hook scripts, static guard, and test matrix. Categorized as 1 +BLOCKING, 6 HIGH, 9 MEDIUM: + +- **BLOCKING — Per-test dedup isolation.** The 70-cell fault matrix shared + a single PPID+lstart-keyed dedup directory across all parametrized cells. + The first cell to touch `a.py` marked it `.injected`; every subsequent + cell skipped the entire viewer subprocess (pre-tool.sh line 95 guard). + Result: the viewer-branch `git rev-parse` + viewer invocation were + exercised exactly ONCE per pytest run — the very code paths the matrix + was added to defend. Reproduced empirically: replacing the bounded + `_git(['rev-parse',...], 0.5)` with an unbounded call passed `70 passed` + even though production would hang for 31s. Fix: `_run()` injects a + unique `SHADOWFROG_TMP_DIR=/_sf_dedup` per test. +- **HIGH — Behavioral SIGTERM coverage.** `subprocess.run(timeout=)` sends + SIGKILL, not SIGTERM, so the matrix had zero behavioral coverage of the + TERM trap. Removing `trap 'exit 0' TERM` was invisible to the entire + test suite. Added `TestPreToolSigterm` using `Popen` + `os.kill(SIGTERM)` + at varied delays, including a "SIGTERM during hung subprocess" case + proving bounded `subprocess.run(timeout=)` ensures the queued signal is + delivered within the 5s deny threshold. +- **HIGH — PascalCase tool name coverage.** Claude Code emits + `Edit`/`Write`/`MultiEdit`/`NotebookEdit`; Copilot CLI emits lowercase. + Removing `.lower()` from TOOL_NAME normalization silently downgraded all + Claude Code mutations to the base reminder — undetected. Added + `TestPreToolPascalCaseToolNames` with 11 spellings asserting the + file-specific actionable branch fires for each. +- **HIGH — Static guard evasion patterns.** The `FORBIDDEN_SET_RE` anchor + `^\s*set\s+` was bypassed by `[[ X ]] && set -e`, `eval 'set -e'`, + `; set -e`, and `set \\\n -e` (line continuation). Substring scanning + + pre-joining line continuations now catch all four. Signal aliases + `SIGTERM` and numeric `15` are normalized to TERM-equivalent. Added 8 + evasion regression tests and 3 alias acceptance tests. +- **HIGH — Strict production wall-clock budget.** The matrix's 7s + wall-clock cap tolerates CI cold-start variance, but happy-path runs + must clear Copilot's strict 5s deny threshold. Added + `test_happy_path_meets_strict_production_budget` (best-of-3 < 5s). +- **HIGH — PATH-empty / python-crash stderr leaks.** With `PATH=""`, + bash itself emits `cat: No such file or directory` to stderr before any + hook code runs. Defensive PATH append + `cat 2>/dev/null` + stderr + redirects on final emitters close the leaks. +- **HIGH — Raw bash-level `python3` script invocations.** Static guard now + flags `python3 path.py` / `python3 -m module` at bash level (same bug + class as raw `git`). Three separate `python3 -c` calls in both hooks + consolidated into single invocations to reduce blast radius. + +Plus 9 MEDIUM cleanups: dedup tmp-dir TTL GC at sessionStart; file_path +precedence over path (was inverted); oversized path cap (1KB); +SIGPIPE in trap pyramid; final emitters silenced; state.json edge-case +coverage (symlink-to-/dev/null, directory-shaped, huge file, future +schema, NUL bytes, deep nesting); readonly-tmpdir test now uses chmod +0500 instead of `/proc` so macOS exercises it; matrix-setup regression +guard that asserts the viewer-branch is actually reachable. + +Test count: **908 passed** (+46 from 862 baseline). New tests include 11 +PascalCase tool variants, 2 SIGTERM behavioral, 6 state.json edge cases, +17 static guard evasion/alias/raw-python detection, file_path precedence, +strict 5s budget, regression-guard, and cross-platform readonly-tmpdir. + +### Added +- **Multi-layer fail-open defense for advisory hooks** — the original + `trap 'exit 0' EXIT` fix was necessary but insufficient. Three + additional layers are now in place: + 1. **Trap pyramid**: separate `trap 'exit 0' EXIT` and `trap 'exit 0' TERM + HUP INT` traps. EXIT alone returns 143/-15 under SIGTERM (empirically + verified on bash 3.2 macOS and bash 5+ Linux), which the runner's + timeout-kill triggers; the TERM trap converts that to exit 0. + 2. **Every external call bounded**: the previously-unbounded + `git rev-parse --show-toplevel` in the pre-tool viewer-discovery branch + was hanging for 30+ seconds on locked/NFS/fsmonitor-corrupted repos + (reproduced as a 31s hang in Opus-4.8's review), causing runner + SIGTERM-kill → tool denial. All git and viewer subprocesses now run + inside a single consolidated Python block with per-call + `subprocess.run(timeout=...)` wrappers. Total bounded work budget is + ~3.5s, leaving >=1.5s headroom under the hook's 5s `timeoutSec`. + 3. **Static structural enforcement**: `hook-templates/check-hook-failopen.py` + blocks PRs that re-introduce ANY of the regression vectors the panel + identified — short-form `set -e`/`-u`, long-form `set -o + errexit|nounset|pipefail|errtrace`, `source`/`.` of external files, + missing EXIT or TERM trap, comment-masquerading-as-trap, or unbounded + `git`/external calls at the bash level. +- **Fault-injection test matrix** (`tests/hooks/test_hook_fault_injection.py`) + — parametrized cells that systematically perturb the hooks across four + axes (stubbed binary failures, malformed/malicious JSON payloads, + filesystem & git state, environment). Now also seeds `.shadow/.md` + in setup so the pre-tool viewer-discovery branch is actually exercised + under fault injection — closing the matrix blind spot that let the + unbounded `git rev-parse` bug ship in the first place. Every cell now + asserts wall-clock < 4.5s (matching the production 5s `timeoutSec`). +- **Shellcheck / release automation** — runs shellcheck (info-level on hook + scripts, warning-level on installer/skill scripts) and invokes the + fail-open contract checker described above. +- **Unit tests for the fail-open checker** + (`tests/hooks/test_check_hook_failopen.py`) — 17 adversarial cases + asserting the checker correctly flags every regression vector and accepts + every legitimate pattern (including `git` inside Python heredocs and the + combined `trap '...' EXIT TERM HUP INT` form). + +### Fixed +- **`preToolUse` hook denied tool calls under Copilot CLI ≥ 1.0.57.** As of + v1.0.57, a `preToolUse` command hook that exits non-zero now **denies** the + tool call (previously such errors were silently ignored). Both hook scripts + ran under `set -euo pipefail`, so any failing sub-step — most commonly the + staleness check's `git diff … | wc -l | tr` pipeline when the shadow was + behind HEAD and `git diff` returned non-zero — exited the script non-zero and + surfaced as *"Denied by preToolUse hook (hook errored)"*. The advisory hooks + are now **fail-open** via the multi-layer defense described above. Affects + `hook-templates/scripts/shadow-frog-pre-tool.sh` and + `hook-templates/scripts/shadow-frog-check-init.sh`. +- **Unbounded `git rev-parse --show-toplevel` in viewer-discovery branch** + (pre-tool.sh:94, pre-fix). On any system where this git call hangs (NFS, + `.git/index.lock` held by `gitk`/`vscode`/`gh`, fsmonitor/Watchman, large + monorepos), the hook exceeded the 5s `timeoutSec` and the runner SIGTERM- + killed it → tool deny. Now consolidated into one Python subprocess with + `timeout=0.5`, covering the same fallback resolution paths. +- **stderr leak on missing `state.json`.** `state.json` was read via a shell + `< redirect`, which printed `No such file or directory` to stderr (before + `2>/dev/null` applied) when the file was absent. The file is now opened + inside Python so a missing file is a caught exception — no stderr noise. + +### Changed +- Tightened all bounded subprocess timeouts to leave ≥1.5s of headroom under + the 5s `timeoutSec` budget: viewer 1.5s → 1.0s; staleness rev-parse 1.0s → + 0.5s × 2; staleness diff 1.5s → 1.0s. Sum was exactly 5.0s with zero + headroom; now 3.5s. +- Removed redundant `signal.alarm(2)` from the viewer subprocess wrapper — + `subprocess.run(timeout=1.0)` already handles cancellation, and the layered + alarm could leave orphaned grandchildren under load (per Opus-4.7-xhigh + review). +- `_init_minimal_shadow` test helper now seeds `.shadow/.md` by + default so the viewer branch is reachable in tests. Pass `seed_target=None` + to opt out. +- Replaced hardcoded `/tmp/PWNED` injection-test sentinels with `tmp_path`- + scoped paths to eliminate flakiness from stale files across runs. +- Repurposed the `shadow-read-only` cwd-fault cell to make the dedup tmpdir + read-only instead of `state.json` (the latter is read-only by design and + the hook never writes to it, so the original cell was a no-op). +- Fixed `SC2295` quoting bugs (`${VAR#"$PREFIX"}`) in + `hook-templates/scripts/shadow-frog-pre-tool.sh` and `install.sh` — the unquoted + form treats `$PREFIX` as a glob pattern, which could mangle paths + containing `*`, `?`, or `[` characters. Surfaced by shellcheck. + +### Notes +- **`additionalContext` on `preToolUse` is undocumented but functional on + Copilot CLI.** The 2026 hooks reference documents only `permissionDecision`/ + `permissionDecisionReason`/`modifiedArgs` for `preToolUse` output, but the + copilot-cli v1.0.24 changelog explicitly notes that `preToolUse` hooks + "respect modifiedArgs/updatedInput, and additionalContext fields." The + current dual-shape output (top-level `additionalContext` for Copilot, + nested `hookSpecificOutput.additionalContext` for Claude Code) works on + both platforms. If Copilot ever removes this undocumented support, the + shadow-aware reminder would silently no-op on Copilot — the `sessionStart` + reminder would remain. + +--- + +## 2026-05-30 + +### Removed +- **Personal (global) install mode** — `install.sh` no longer symlinks skills + into `~/.copilot/skills/` or `~/.claude/skills/`. ShadowFrog now installs + **only into a specific repository** via the now-required `--project ` + flag. A global install would auto-engage the shadow-edit hooks across every + repository the developer touches, firing unexpected shadow writes (and a + potential information-leakage risk) in projects that never opted in. Per-repo + install also keeps skills committed to the fork, which is what fork-based + dreaming needs. Updated `install.sh` (required `--project`, dropped the + personal block + help text), `README.md` (single "Install into your repo" + section), `claude.md`, the `pre-tool` hook's viewer-resolution loop, every + SKILL.md path-resolution loop and usage example (project paths only), and the + install test suite. + +--- + +Pre-release hardening for the public MIT release: a five-model independent +audit (Opus 4.6 / 4.7 / 4.8, GPT-5.5) drove correctness fixes across the dream +pipeline, the reconciler, and the viewer, plus licensing/transparency docs and +a round of helper-script slimming. + +### Added +- **LICENSE (MIT)** and **`RESPONSIBLE_AI.md`** — standard Microsoft MIT + license plus a transparency note covering intended uses, out-of-scope uses, + evaluation results, limitations, and best practices. README links to both. +- **gitignored-`.shadow/` guard for dream** — when `shadow-frog-init`'s + "local only" option gitignores `.shadow/`, the dream workflow's + `git add -A` silently skipped the shadow content, so nothing reached the + remote and every discovery was lost without warning. `dream-setup.sh` now + fails fast via `git check-ignore .shadow` with a clear fix message; + `shadow-frog-init` step 9 documents the committed-vs-gitignored trade-off; + the dream SKILL adds a "`.shadow/` must be git-tracked" prerequisite; and the + README's Dream section carries a one-line callout. +- **dirty-tree guard on branch cleanup (B16)** — `dream-reconcile.py + --cleanup-branches` merges discoveries into the working tree without + committing, so the old ancestor check passed against a stale HEAD and could + delete dream branches (the only durable copy) before the merge was + persisted. Cleanup now also refuses when `git status --porcelain -- .shadow/` + is non-empty, enforcing reconcile → commit → push → cleanup. + +### Fixed +- **Pre-release audit bug sweep** — a broad set of correctness fixes surfaced + by the 5-model audit: `dream-validate` normalizes bare-string/non-dict + discoveries and tolerates BOM/CRLF in frontmatter; `dream-reconcile` emits + canonical per-file shadow headers, unions refs into existing `_cross` slugs + instead of dropping them, stops truncating the tip SHA, and mirrors + manifest/patch even when a report is corrupt; `dream-coverage` filters to the + shadowed source set; `shadow-viewer` preserves dotfile paths and scopes + cross-ref parsing to the `**Refs**` block; `meditate-repair` recovers + category from frontmatter; `shadow-init` prunes only `EXCLUDE_DIRS` + `.git`; + `install.sh` / `dream-setup.sh` validate args and pass values via quoted + heredocs to prevent shell injection. README drops nonexistent viewer flags; + several SKILL docs corrected. Added 28 regression tests. +- **Discovery metadata loss on duplicate merge (B15)** — on an exact-text + duplicate the reconciler returned early and discarded the new discovery's + metadata, so a later dream recording the same claim with stronger evidence, + higher-trust source, or extra labels lost the upgrade. Exact matches now + merge metadata (union labels, upgrade source by trust, promote + uncertain → verified, never touch `refuted`). Fuzzy near-duplicates still + skip. +- **`_`-prefixed source dirs dropped from totals (B19)** — `update_state` and + `rebuild_top_index` pruned every `_*` directory at every depth, which also + excluded mirrored shadows under `_`-prefixed *source* dirs (e.g. + `src/_internal/foo.py.md`). Internal dirs only live at the top level, so the + `_*` prune now applies only at `.shadow/`'s root. +- **`_index.md` parent column convention (D9)** — meditate's `repair_parent` + wrote a resolved `dream_id` while the reconciler and the branch-keyed lineage + reader used branch names, orphaning nodes after a meditate pass. Standardized + on branch names everywhere. +- **manifest-vs-shadow source-of-truth contradiction** — the dream SKILL + contradicted itself and misdescribed reconcile (the reconciler merges + discoveries from `manifest.json`, not by replaying the branch shadow). + Reworded to match the code: the manifest is authoritative for propagation, + per-file shadows are the human-readable copy and a required validate gate, + and every discovery must be written into both. + +### Removed +- **dream-lineage Graph tab** — the interactive radial force-graph tab in + `dream-lineage.py`'s HTML output (~326 lines: graph data prep, JS + `CAT_COLORS`, constellation CSS, the tab button/container, and `initGraph()`). + The documented Chains / Fresh / Full Tree tabs are unchanged. (`dream-lineage.py` + 1076 → 750 lines.) +- **`_dreams/_coverage.json`** — the exploration coverage map written by + `dream-reconcile.py`'s `rebuild_coverage`. Nothing read it: `dream-coverage.py` + recomputes coverage live on every invocation. Removed the function and its + reconcile call (reconcile steps renumbered 1–9). The shared discovery-counting + helper it used is retained for `rebuild_top_index`. +- **Dead constants and legacy fossils** — unused `VALID_CATEGORIES` / + `VALID_VERDICTS` in `dream-reconcile.py`; the obsolete `_None yet._` + placeholder heal (only the canonical `_No cross-cutting discoveries yet._` is + emitted); and the undocumented `**Parent**:` / `**Chain**:` report-body + scrapers in `dream-lineage.py` (the `builds_on` frontmatter parse is the + supported lineage source). + +--- + +## 2026-05-19 + +Round-2 multi-reviewer audit (Opus 4.7 xhigh + high, Opus 4.6, GPT-5.5): +correctness, security, and documentation fixes across the dream pipeline, +the preToolUse hook, and the meditate index repair. A follow-on Tier 3 +bug sweep landed seven additional correctness fixes in the same area. + +### Fixed (Tier 3 bug sweep) +- **preToolUse hook dedup collision** — `DEDUP_DIR` was keyed on `$PPID` + alone, so distinct Claude Code sessions sharing `PPID=1` under a + process manager, transient shells reusing a PPID, or PID-wrap on + long-running systems would share a dedup bucket and silently suppress + each other's discovery injection. Now keys on `$PPID` plus + `ps -p $PPID -o lstart=` (parent process start time), falling back to + PPID-only if `ps` is unavailable inside sandboxed containers. +- **shadow-init last_commit empty-string sentinel** — `build_state_json` + and the main init path defaulted `last_commit` to `""` when + `git rev-parse` failed. The preToolUse hook reads + `state.get('last_commit', 'none')`, but `.get()`'s default only fires + on missing keys, not empty values — so `git rev-parse --verify ""` + failed and every subsequent staleness comparison misfired with a + false warning. Both call sites now default to `"none"`, which the + hook already handles as the non-git sentinel. +- **shadow-viewer parse_discovery "Dream report:" leakage** — the + continuation-line catch-all branch in `parse_discovery` swept + `Dream report: \`_dreams//\`` markers into the discovery body, + so `--top` and `--summary` output rendered them concatenated to the + behavioral statement. Added an explicit `Dream report:` branch that + extracts the slug into `meta["dream_report"]` instead. +- **dream-reconcile back-pointer idempotency false positive** — the + "is the back-pointer already present?" check used substring + `f'_cross/{slug}.md' in line`, which false-matched any discovery + body mentioning the same slug (e.g. `Also involves: \`_cross/foo.md\``). + That silently swallowed legitimate back-pointer adds on subsequent + dreams. Now matches the actual markdown link target + `]({prefix}_cross/{slug}.md)` using the same depth-aware prefix as + the write path. +- **dream-reconcile case-insensitive heading reads** — most cross-ref + read sites already lowercased the comparison, but + `find_cross_references_heading` used exact match. A meditate or user + rewrite that lowercased `## cross-references` would slip past the + finder, causing `_ensure_cross_references_section` to append a + duplicate section and back-pointer dedup to miss entirely. All + read sites are now consistently case-insensitive; writes still + emit the canonical `## Cross-References`. +- **dream-reconcile top-level `_index.md` not refreshed** — the + reconciler updated `state.json`, per-file shadows, and cross-cutting + files, but `.shadow/_index.md` (the human-readable manifest with + symbol lists and discovery counts) was never regenerated after + reconciliation. It stayed frozen at init values until a manual + meditate or re-init. Added `rebuild_top_index` as Step 8 (runs after + `update_state` so totals reflect the new dream cycle), with + extracted `_count_discoveries` and `_shadow_symbol_names` helpers + shared between coverage and index rebuilds. +- **shadow-frog-dream SKILL anchor mismatch** — the per-validation + error message pointed reviewers to "see Critical Invariants" for + the artifact-format requirement, but the Critical Invariants section + covers paths, branches, RUN_PREFIX, and reconciliation rules — + artifact format is a sub-section. Updated the cross-reference to + "see Critical Invariants → Artifact Format" so reviewers land at + the correct sub-section. + +### Fixed +- **dream-reconcile / dream-coverage count inflation** — `rebuild_coverage` + and `update_state` counted `- ` bullets inside `## Cross-References` + (which are back-pointer links, not discoveries) and counted + `## Cross-References` / `## File-Level` as symbols. Every reconcile + silently corrupted `state.json` totals and `_coverage.json`. Both + loops (plus `dream-coverage.py`'s `check_coverage`) now use an + `in_xref` state machine that matches `--check-invariants` semantics. +- **dream-reconcile back-pointer paths** — subdir shadows (e.g. + `.shadow/src/foo.py.md`) wrote relative links as `_cross/slug.md` + instead of `../_cross/slug.md`, breaking markdown rendering and + `--check-invariants` on any non-flat repo. Now computes depth and + prepends `../` per level. +- **dream-reconcile canonical layout** — bootstrap for never-before-seen + shadow files used `## ` + filename as the heading (treating a filename + as a symbol) and `_None yet._` for the cross-ref placeholder. Now + uses `## File-Level` and `_No cross-cutting discoveries yet._` to + match `shadow-init.py`. Placeholder-detection accepts both forms so + meditate runs heal older shadows. +- **dream-validate first-dream mirror gate** — `git status --porcelain` + rolled untracked subtrees up to a single `?? .shadow/` line, so + first-dream cases where the entire `.shadow/src/` subtree is untracked + false-failed the discovery-mirror gate. Added `--untracked-files=all`. +- **dream-validate mirror-gate error message** — claimed manifest entries + were "LOST at merge time", but `dream-reconcile.py` reads + `manifest.discoveries` directly. Gate kept (PR reviewers still need + shadows in sync with the branch), but the message now accurately + describes the workflow contract being enforced. +- **preToolUse shell→Python injection** — the discovery-inlining heredoc + used an unquoted `<` marker. Also: Phase 7 said "seven invariants" but + listed five — clarified as "five core (full 7-invariant set in + `/shadow-frog`)". +- **`dream-reconcile.py --all` references** — three places documented + an `--all` flag that was never implemented. Rewrote + parallel-batch instructions to use positional branch arguments. +- **`--check-invariants` surfacing** — added missing row to the + shadow-frog-viewer/SKILL.md Available Views table and an example + invocation. Previously only documented in shadow-frog/SKILL.md. +- **Duplicate `# 9.` numbering** in `dream-validate.py` — two checks + were both labelled step 9 in source comments and docstring. + Renumbered consistently (op validator → 9, mirror gate → 10, + label triage → 11). + +### Changed +- **Discovery format spec dedup** — claude.md and + shadow-frog-update/SKILL.md trimmed their duplicated discovery-format + sections (~31% reduction across both) and now point at + shadow-frog/SKILL.md as the canonical source. shadow-frog-init was + audited and left alone (no actual discovery-writing spec there). + +### Fixed (code-quality audit follow-on) + +Systematic 5-script code-quality audit (Opus 4.7 xhigh × 2, Opus 4.7 +high, Opus 4.6 × 2) of `shadow-init.py`, `shadow-viewer.py`, +`dream-reconcile.py`, `dream-lineage.py`, and the seven smaller scripts +(`meditate-repair`, `dream-validate`, `dream-coverage`, `dream-setup.sh`, +both hooks, `install.sh`). Two real bugs surfaced; rest was inline +tidying. Risky refactors flagged for future review. + +- **dream-reconcile prefix-substring data loss in cleanup_branches** — + Safety check 2 used `if dream_id not in index_content:` raw substring + against `.shadow/_dreams/_index.md`. When our dream_id is a prefix of + any indexed ID (e.g. branch `…1400-foo` plus indexed + `…1400-foo-extended`), the check falsely passed, allowing cleanup + to DELETE the unreconciled branch — irreversible loss of whatever + discoveries were only on it. Same false-pass shape in the + descendant-detection check at line 1097. Both now use + `_read_indexed_dream_ids` for parsed-ID set membership. +- **dream-reconcile prefix-substring false-pass in verify_reconciliation** — + Same `if dream_id not in f.read():` substring against the index. + Less severe (verify only reports failures; no data mutation), but + still silently swallowed real index-mismatch bugs. Same fix. +- **dream-lineage md_to_html emitted invalid HTML** — `- item` lines + became bare `
  • ` tags with no `
      ` wrapper, producing + structurally invalid HTML in the dream-lineage panel-body. Switched + to a `render_ul` callback that wraps consecutive list runs in a + single `
        `; indented items keep the `class='nested'` hook so + the existing `.panel-body li.nested` CSS still drives visual + nesting (no rendering regression). + +### Changed (code-quality audit follow-on) + +- **shadow-viewer discovery-metadata regex consolidation** — the + `_(status, source: type, labels: […])_` pattern was inline-duplicated + in `parse_discovery`, `parse_cross_cutting`, and + `view_check_invariants`. Now one module-level `_DISCOVERY_META_RE`; + future drift between the three sites is now impossible. +- **Dead-code removal** (zero behavior change, all CLI snapshots + byte-identical before/after): unused `defaultdict` import + dead + `created_cross` variable in `dream-reconcile`; dead `CAT_ICONS` + dict in `dream-lineage`; dead `HELP_TEXT = __doc__` alias in + `dream-validate`; no-op `try/except SystemExit: raise` in + `shadow-init`'s argparse path (SystemExit derives from + BaseException, not Exception); dead `all_disc = []` initialization + in `shadow-viewer`'s `view_summary`; unused `other_ts` tuple-unpack + target in `meditate-repair`; dead `TIMESTAMP=$(date …)` in + `shadow-frog-check-init.sh`. + +### Added (test suite) + +- **First test suite** — 408 pytest cases across 11 test files + covering every script in the package (5 Python scripts + 4 shell + hook template scripts + `install.sh` + `dream-setup.sh`). Full suite + runs in ~22s; no per-test fixtures rely on network, user gitconfig, + or `~/.copilot`. Counts by file: + - `shadow-init.py` — 164 (10 language extractor families, brace + nesting, decorator skip, B2 sentinel regression). + - `shadow-viewer.py` — 55 (parse, summary, top, search, dream + parser; B3 "Dream report:" leakage regression). + - `dream-lineage.py` — 28 (load_index, generate_html, md_to_html + inline + nested-list HTML regression). + - `dream-reconcile.py` — 57 (B4 idempotency, B5 case-insensitive, + B6 rebuild_top_index, two prefix-substring data-loss regressions + in `verify_reconciliation` + `cleanup_branches`). + - `dream-validate.py` — 19 (all 11 validation checks). + - `dream-coverage.py` — 18 (coverage, fan-in, scope). + - `meditate-repair.py` — 27 (dedup, state repair, I4 parent_branch + regression). + - `shadow-frog-pre-tool.sh` — 8 (B1 PPID dedup + K4 injection). + - `shadow-frog-check-init.sh` — 5 (B2 sentinel). + - `install.sh` — 6 (project install, --no-hooks, --no-context). + - `dream-setup.sh` — 15 (worktree, branch naming, namespace, + RUN_PREFIX, slug validation, --dry-run). + - Plus 6 smoke tests in `tests/test_smoke.py` covering the + shared importlib loader + git-config isolation fixtures. +- **Pytest scaffolding** — `pytest.ini` (testpaths + custom + `slow`/`integration` markers) and `tests/conftest.py` (shared + `_load_script` importlib helper for hyphenated script filenames, + per-script module fixtures, mutable `coupon_demo` copy, isolated + `tmp_git_repo` with `GIT_CONFIG_GLOBAL=/dev/null` so tests are + independent of the developer's `~/.gitconfig`). + +### Added (test suite — Phase 3 coverage push) + +- **+281 tests across 4 files** lifting overall Python coverage from + 53% to **87%** (689 tests total, all passing in 56s). Each script's + primitive layer was already covered by Phase-2; Phase-3 closed the + render/output/orchestrator gap: + - `shadow-viewer.py` 27% → **82%**: 91 cases for `view_summary`, + `view_search`, `view_prefs`, `view_labels`, `view_recent`, + `view_check_invariants` (synthetic-violation injection covering + all 7 shadow invariants), and `main()` CLI dispatch. + - `dream-lineage.py` 49% → **92%**: 46 cases for `generate_html` + (happy path on coupon-demo + empty/missing/malformed edge cases + + dream-chain ancestry rendering), `node_html`/`compact_node`/ + `tree_depth` helpers, and `--output` CLI. + - `shadow-init.py` 61% → **81%**: 63 cases for `main()` CLI (in- + process via argparse + subprocess for `__main__` smoke), + `init_shadow` integration (mixed-language project, `.shadowignore` + end-to-end, totals reconciliation), and the remaining extractor + edge branches not exercised by the existing 164 cases. + - `dream-reconcile.py` 50% → **93%**: 74 cases for the integration + paths (`merge_discoveries`, `mirror_reports`, `update_index`, + `update_state`, `rebuild_coverage`, `_resolve_tip_commit`, + `cleanup_branches` real-delete path, full `main()` orchestrator + end-to-end with multi-branch fixtures). +- **Mock-audit passed**: 0 `unittest.mock` / `Mock` / `MagicMock` / + `@patch` / `mocker` / custom Fake/Stub classes across all 689 + tests. Real `subprocess` (63 invocations), real `git` (via + isolated `tmp_git_repo`), real filesystem (via `tmp_path` and + per-test `coupon_demo` copies). `monkeypatch.chdir` / `setenv` / + `setattr(sys, "argv", ...)` are used only for genuine env/cwd/ + argparse plumbing — never to stub source code. This means a + regression in any covered code path is guaranteed to fail tests + on regression (e.g., the Phase-1 prefix-substring + nested-list- + HTML bugs would have surfaced immediately under these tests). + +### Fixed (Phase-3 coverage-audit follow-on) + +The Phase-3 coverage push surfaced two real bugs that the new tests +now also cover as regressions: + +- **dream-lineage `tree_depth` unbounded recursion on cyclic + `_index.md`** — `tree_depth(node, children)` was single-line + recursive with no cycle guard. A malformed `_dreams/_index.md` + with a self-parent (`A → A`) or any cycle (`A → B → A`, + `A → B → C → A`) crashed `generate_html` with `RecursionError`. + Cycles aren't expected in well-formed input, but the parser does + not reject them and a human editing the index manually can + produce them. Now passes a `_seen` set through the recursion and + treats a re-visited node as terminal — strictly safer than + crashing on one malformed row. +- **dream-reconcile `update_index --dry-run` writing to disk** — + `update_index` bootstrapped `.shadow/_dreams/_index.md` (with + `os.makedirs` + initial header write) BEFORE the `if dry_run: + return` check. The user-facing contract of `--dry-run` is to + preview without touching disk, but the bootstrap silently + materialized the `_dreams/` directory + an empty index file on + every dry-run, even when the user just wanted to see what would + happen. Fix: dry-run check moved above the bootstrap; dry-run + now prints the preview lines and returns without any disk write. + +--- + +## 2026-05-18 + +Branch-based dream persistence, systematic eval framework with results +dashboard, hook upgrade to inline shadow knowledge, plus the format/ +validator hardening from a multi-agent review of the whole repo. + + +### Added +- **Branch-based dream persistence** — each experiment becomes a + `dream//` branch pushed to the fork. Dreams compound + by branching from prior dream branches, inheriting code + shadow from + the ancestor. The `` segment isolates dreams across + concurrent projects sharing a fork. +- **Dream pipeline scripts** — `dream-setup.sh`, `dream-validate.py`, + `dream-reconcile.py`, `dream-coverage.py`. Self-documenting, with + `--help` and clear error messages. Listed in SKILL.md `scripts:` + frontmatter and resolved at runtime via the `SKILL_DIR` lookup. +- **`dream-coverage.py --scope`** — repeatable flag for focused + exploration (e.g. `--scope src/ --scope tests/`). Coverage map only + considers files under the given prefixes; planning prompt narrows + accordingly. +- **`dream-lineage.py`** in viewer — visualizes dream ancestry chains + and branch relationships across reconciled dreams. +- **`shadow-viewer.py --top FILE`** — derived view returning the top-N + actionable discoveries for a single file, ranked by label/status/source. + Powers the preToolUse hook (no file mutation, no invariant violations). +- **Hook upgrade**: `preToolUse` now inlines actionable shadow + discoveries before mutation tools (`edit`, `create`, `str_replace`, + `write`) instead of just reminding the agent to read the shadow. + Includes per-session dedup, hard latency cap, output cap, and + `_cross/` discoveries surfaced alongside per-file ones. +- **`meditate-repair.py`** — extracted from a 100-line inline Python + block in the SKILL into a real script. Backs up `_dreams/_index.md`, + detects corrupted reports (frontmatter `dream_id` != folder name), + and rebuilds `category` / `verdict` / `title` cells from each dream's + report + manifest. Idempotent. +- **`_prefs.md`** — project-wide preferences file (not tied to any + file/symbol). Captured from user conventions, separate from per-file + shadows. +- **`_dreams/_coverage.json`** — exploration coverage map rebuilt by + the reconciler so future dream planners can see what's saturated. +- **7-column `_dreams/_index.md` schema** — `dream_id | category | + verdict | title | branch | parent | tip_commit`. Replaces the old + 4-column form; init now emits the canonical header. +- **`dream_cycles_completed`** field in `_meta/state.json` for tracking + dream activity over a project's lifetime. +- **Label triage at discovery write time** — dream prompts agents to + apply `bug` / `security` / `performance` / `feature-gap` / + `tech-debt` labels as discoveries are written, not retroactively. + `dream-validate.py` emits non-blocking warnings when actionable text + is missing a label. +- **Eval framework** (`eval/`) — systematic SWE-Smith evaluation + (`eval/swesmith/`) with stacked-bug methodology, Docker-based + anti-cheat, three-level scoring (L1 file / L2 function / L3 + LLM-as-judge), and per-repo dream lineage figures. +- **Results dashboard** (`eval/results_dashboard.html`) — interactive + Chart.js dashboard structured around three findings, embedded + navigation eval, sunburst lineage figures, and full scoring + methodology. + +### Changed +- **Dream SKILL.md** rewritten end-to-end around the branch-based + workflow: pipeline phases call out script entry points, mandatory + reconciliation timing (parallel batch mode vs sequential mode) is + explicit, and the "Curating Dream Experiments for Upstream PRs" + cheat sheet (maintainer test, devil's-advocate framing, "so what?" + test, "no credit for effort" rule) is embedded at the end so users + can apply curation heuristics in any agent session. +- **Backtick-in-heading** promoted from a passing mention to a hard + rule in `shadow-frog/SKILL.md`. The viewer parser only matches + `` ## `SYMBOL` ``; headings without backticks have their discoveries + silently dropped from search and `--top` output. +- **`shadow-frog-update/SKILL.md`** description and trigger list + rewritten to match actual behavior: hooks remind (sessionStart and + preToolUse), they don't run update, and dream calls reconcile + directly rather than going through `/shadow-frog-update`. +- **`shadow-frog-meditate/SKILL.md`** index-repair section now calls + `meditate-repair.py` via the `SKILL_DIR` lookup instead of inlining + 100 lines of Python. +- **`examples/coupon-demo/.shadow/`** regenerated to match the current + spec: ghost shadows for non-existent source files deleted, ghost + dream folders deleted, surviving dreams given required `manifest.json` + + `patch.diff` + `branch`/`parent_branch`/`remote` frontmatter, + `_cross/` category enum corrected (`bug` → `edge-case`), + `_index.md` and `state.json` recounted to match reality. +- **`examples/coupon-demo`** repurposed as a pure example of "what a + real `.shadow/` looks like." Eval scaffolding (`task.json`, + `instructions.md`, `EVAL_RESULTS.md`) removed; README rewritten + around browsing the shadow with the viewer. Systematic eval moves to + `eval/`. + +### Removed +- **`eval.sh`** — coupon-demo-specific A/B eval runner, superseded by + the systematic SWE-Smith framework under `eval/swesmith/`. +- **`examples/coupon-demo/task.json`**, + **`examples/coupon-demo/instructions.md`**, + **`examples/coupon-demo/EVAL_RESULTS.md`** — eval scaffolding for + the now-deleted `eval.sh` runner. + +### Fixed +- `_dreams/_index.md` header in `shadow-init.py` regenerated as the + canonical 7-column schema (was emitting the obsolete 4-column form, + which broke `dream-reconcile.py` and `dream-lineage.py` against + freshly-initialized repos). +- `state.json` template in `shadow-frog-init` includes + `dream_cycles_completed: 0` from initialization (was missing, so + dream's first increment failed against a freshly-initialized repo). +- Reconciliation-timing language in dream SKILL.md was self- + contradictory ("MUST reconcile before starting another dream" vs + "After all agents complete, merge discoveries"). Rewritten as two + explicit modes — parallel batch (one `dream-reconcile.py` call with + all branch names after the batch) vs sequential (reconcile after each + dream) — with the invariant that batches are never queued un-reconciled. + +--- + +## 2026-04-17 + +Major dream overhaul: experiment-only mode, 5-model review fixes, dreams +archive, hook refinements, dead hook removal. + +### Added +- **Dream is now experiment-only** — removed observe mode entirely. Every + task must produce code, run it, and capture a patch. Reading code is + preparation, not a deliverable. +- **`_dreams/` experiment archive** — persistent storage for dream reports + (`report.md`) and patches (`patch.diff`), enabling compounding across + dream sessions. +- **`_dreams/_index.md`** — tabular index of all experiments with dream_id, + category, verdict, status, and title. Bootstrapped with header on first + dream. +- **Phase 7: Experiment Review** — interactive walk-through of each + experiment with Keep/Delete/Follow-up actions. +- **Experiment completion criteria** — tasks require non-empty `patch.diff`, + at least one executed command with recorded output, and a saved `report.md`. +- **Experiment-Only Files rule** — shadows are never created for files that + only exist in the worktree. Discoveries are anchored to the existing code + the experiment relates to. +- **Continuing a Prior Dream** subsection — apply prior `patch.diff`, + handle conflicts, set `builds_on` lineage. +- **Delegate mode artifact contract** — PR must contain `report.md`, + `patch.diff`, updated `_dreams/_index.md`, and per-file shadows. +- **Applying Experiment Patches** section — `git apply` instructions with + conflict guidance. +- **Standalone harness guidance** — for repos with no test infrastructure, + create scripts that exercise the code directly. +- **Parallel safety** — task counter in slugs; only orchestrator writes + `_dreams/_index.md`. +- **Hallucination guard** — mandatory `echo` + verify step for `DREAM_ID` + timestamps; agents must never fabricate dates. +- **Dream hygiene in meditate** — index consistency checks, report + completeness validation, stale patch warnings. + +### Changed +- **Diff capture** — commit-then-diff (`git diff $BASE HEAD`) replaces the + unreliable staging-based approach. +- **Verdict vs status taxonomy** — simplified to just `verdict` (agent: + useful/dead_end). No status column — everything in the index is kept; + deleted experiments are removed entirely. +- **`REPO_ROOT`** — uses `git rev-parse --show-toplevel` instead of `$(pwd)` + so it works across tool calls. +- **Worktree reuse** — adds `git clean -fdx && git checkout .` before reuse. +- **Investigation category** — requires assertion-based tests with falsifiable + claims, not just print statements. +- **Phase 7 in delegate mode** — skipped entirely; the PR is the review surface. +- **Security experiments** — constrained to local/test only, never external systems. +- **preToolUse message** — now includes mirroring convention example + (`src/auth.py -> .shadow/src/auth.py.md`) and `/shadow-frog` skill pointer. +- **preToolUse reminder** — prompts agent to capture user-shared knowledge + as `source: user` discoveries and preferences to `_prefs.md`. + +### Removed +- **Observe mode** in dream — every task is now an experiment. +- **sessionEnd hook** — output was silently ignored by Copilot CLI runtime. + Removed script, `hooks.json` entry, and all doc references. Two active + hooks remain: `sessionStart` + `preToolUse`. +- `approach: experiment` metadata field (vestigial — only one legal value). +- `/tmp` fallback for failed worktrees (broke git context; now marks task + blocked and replans). + +### Fixed +- `_dreams/` directory not created before temp patch write (first-dream bug). +- Orphaned markdown in init and viewer SKILL.md. +- 5-model parallel review (GPT-5.3-Codex, GPT-5.4, Opus 4.6, Opus 4.7, + Goldeneye) identified 15 issues — all addressed: + - `shadow-init.py`: store full SHA (was `--short`), fixing false staleness. + - Dream: save `REPO_ROOT` before `cd` worktree; capture `BASE_COMMIT` + before changes. + - Hook counts read `state.json` instead of broken regex on markdown. + - `total_discoveries` defined as per-file only (excludes `_cross/`, `_dreams/`). + - Script paths use explicit installed paths, not broken `$0` trick. + - Cross-cutting threshold consistently 3+ files across all skills. + - Viewer shell fallback excludes `_dreams/` from find commands. + +--- + +## 2026-04-15 + +Delegate mode, project install, pure hooks. + +### Added +- **Delegate mode for dream** — run dream via `/delegate` on cloud agent + infrastructure. Creates branch, runs experiments, opens draft PR. +- **Project install** (`install.sh --project `) — copies skills and + hooks into `.github/skills/` and `.github/hooks/` for cloud agent access. +- **`agent-context.md`** — minimal always-on context injected into + `copilot-instructions.md` for project-level agent awareness. + +### Changed +- **Hooks are now pure** — removed all file-writing side effects + (`_meta/stale`, `_meta/hooks.log`). Hooks read state, output + `additionalContext`, done. No more dirty git status from hooks. +- **install.sh** — detects and removes symlinks from prior personal install + before copying (prevents silent writes into symlink target). + +### Fixed +- Removed `.shadow-frog-needs-init` flag file mechanism (replaced by + `sessionStart` hook check). + +--- + +## 2026-04-14 + +Hook system overhaul: `additionalContext` injection, `preToolUse` as +primary hook. + +### Changed +- **Hooks use `additionalContext`** (Copilot CLI v1.0.11+) — context is + injected directly into the agent conversation instead of being printed + to ignored stdout. +- **`preToolUse` replaces `postToolUse`** as the primary hook — fires + before every tool call, giving the agent shadow context at the moment + it needs it most. +- Renamed `shadow-frog-post-tool.sh` → `shadow-frog-pre-tool.sh`. + +### Fixed +- `grep -c '|' || echo '0'` double-output bug (was printing `0\n0`). +- `_index.md` count handles both table (`|`) and list (`- .md`) formats. +- Safe JSON serialization via env vars instead of shell interpolation. + +--- + +## 2026-03-23 + +Agent context, broader trigger, install improvements. + +### Added +- **`agent-context.md`** — single source of truth for project instructions, + read by `install.sh` instead of hardcoded snippet. +- **Broader shadow-frog trigger** — skill loads for more query patterns. +- **Custom instructions guidance** in README for manual setup. + +### Fixed +- Circular symlinks in hook template scripts (skenv artifact). + +--- + +## 2026-03-22 + +Dream enforcement, init script. + +### Changed +- **2-per-category minimum enforced** in dream (was 3, consistently ignored). + MANDATORY language repeated 3 times. 12 tasks minimum (6 categories × 2). +- **User-focus override** — `dream focus on security` allocates all tasks + to one category. + +### Added +- **`shadow-init.py`** (1375 lines) — language-agnostic symbol extraction + for 15 languages, git-based file discovery, `.shadowignore` filtering, + full `.shadow/` scaffolding. +- **Meditate structured output** — JSON-per-line for auto-apply of + merge/conflict resolutions. +- **Script discoverability** — `scripts` field in SKILL.md frontmatter, + explicit installation paths. +- **Dream: direct-write** instead of JSON handoff for parallel agents. +- **Dream: reconciliation phase** for post-agent metadata consistency. + +--- + +## 2026-03-20 + +Labels, AFK-safe patterns, viewer hardening. + +### Added +- **Discovery labels** — optional `labels: [bug, performance, security, + feature-gap, tech-debt]` for actionable discoveries. +- **AFK-safe patterns** for dream — worktrees outside repo, temp scripts in + `/tmp`, deferred cleanup, no working-branch modifications. +- **Category-first dream planning** — plan organized by category, not by file. + +### Changed +- Viewer hardened with better error handling and edge cases. +- Orchestrator feedback addressed: write rules, dedup, output schema, + incremental meditate. + +--- + +## 2026-03-18 + +Meditate skill, dream categories, README refresh. + +### Added +- **`shadow-frog-meditate`** — shadow hygiene skill that scans for + duplicates, near-duplicates, and conflicts. Merges automatically where + possible, asks user for ambiguous cases. +- **6 investigation categories** for dream — investigation, bug hunting, + feature design, refactoring, optimization, security audit. +- **Format compliance** section in meditate for strict discovery format. +- **Harmonized discovery format** across all 6 skills. + +### Changed +- README: added meditate and viewer to Usage section, shadow repo diagram, + froggy team logo. + +--- + +## 2026-03-16 + +Viewer skill, audit fixes. + +### Added +- **`shadow-frog-viewer`** — browse and query the shadow knowledge base + with 4 commands: overview, search, view preferences, recent discoveries. + Includes Python helper script with shell fallbacks. + +### Fixed +- 14 audit issues across the skill suite. + +--- + +## 2026-03-11 + +Shadow filtering and preferences. + +### Added +- **`.shadowignore`** — gitignore-syntax file for filtering which files + get shadow coverage (skip vendor, generated, etc.). +- **`_prefs.md`** — project-wide user preferences not tied to any specific + file or symbol. Captured from conversations with `source: user`. + +--- + +## 2026-03-10 + +Hooks, dream experiments, README rewrite, EMU migration. + +### Added +- **Dream experiment mode** — alongside observe mode, dream can now set up + git worktrees, implement changes, run tests, and produce patches. +- **Hook system** — `sessionStart` and `postToolUse` shell hooks with + shadow status and staleness detection. + +### Changed +- **README rewritten** — skenv and direct install presented as equal + options, stale repo URL fixed. +- Hooks use side effects instead of ignored stdout. + +### Fixed +- Hook config: `timeoutSec` field name, stale discovery grep pattern. +- Added `.github/hooks/` to `.gitignore` (skenv-generated). +- Discovery format: removed surprise scores, stale line range mention. +- Heading format: fixed rendering, use code blocks for examples. + +--- + +## 2026-03-09 + +ShadowFrog pivots from Python codebase to distributable skill suite. + +### Changed +- **Repurposed ShadowFrog** from a Python shadow-generation tool to a + distributable suite of 6 AI coding agent skills. +- All SKILL.md files rewritten for LLM agent comprehension. + +### Added +- **6-skill suite**: `shadow-frog` (main), `shadow-frog-init`, + `shadow-frog-update`, `shadow-frog-dream`, `shadow-frog-meditate` + (placeholder), `shadow-frog-viewer` (placeholder). +- **Symbol-level granularity** — shadows mirror at the symbol level, not + just file level. Every class, function, method gets a `##` section. +- **Bidirectional reference system** — `file::symbol` notation with 7 + invariants ensuring code↔shadow integrity. +- **User-agent conversational knowledge capture** — `source: user` and + `source: interaction` trust levels alongside `source: exploration`. +- **Auto-placement logic** for user-shared knowledge. +- **Verification and dedup procedures** for discoveries. +- **Cross-cutting discoveries** with descriptive slug filenames in `_cross/`. +- **`claude.md`** with comprehensive project guidelines. + +### Removed +- Confidence scores from discovery format. +- Per-file discovery IDs (anchored by `file::symbol` heading instead). +- Line numbers from shadow headings and refs. +- Numeric IDs from cross-cutting filenames. + +### Fixed +- 20 audit findings across the skill suite. +- 11 issues from systematic audit. + +--- + +## 2026-02-18 + +Initial implementation (Python codebase phase). + +### Added +- Initial ShadowFrog implementation as a Python shadow-generation tool. +- Common utilities ported from gray_treefrog (config, caching, logging). +- Thread-safe caching for QueryEngine, LLMDescriber, PromptRenderer. +- LLMDescriber tests. +- Docker image enforcement for CLI commands, pre-commit hooks. +- README with architecture diagrams. + +### Fixed +- Config wiring: frozen model override, dead config fields, missing CLI args. +- Upward dependency, split pipeline, mirrored test directories. diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..9e841e7 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ + MIT License + + Copyright (c) Microsoft Corporation. + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE diff --git a/README.md b/README.md new file mode 100644 index 0000000..25e689c --- /dev/null +++ b/README.md @@ -0,0 +1,437 @@ +# ShadowFrog + +ShadowFrog gives coding agents a **shadow knowledge base** for any codebase: +a file-backed memory of tacit codebase knowledge learned from code reading, +experiments, and conversations with you. + +Most agent memory preserves what happened in past chats. ShadowFrog is built +for **tacit knowledge** that is hard to recover from chat history or source +alone: which refactor breaks downstream callers, which invariant the tests +never exercise, which "obvious" cleanup removes a production workaround, or +which cross-file edge case is easy to miss. The code tells you *what runs*. The +shadow tells future agents *what has been learned about how it behaves*. + +Read the launch blog post: [Shadow-Frog: Coding Agents that Dream and +Discover](https://microsoft.github.io/debug-gym/blog/2026/06/shadow-frog/). + +

        + Shadow repository structure +

        + +--- + +## Quick Start + +```bash +# From your ShadowFrog checkout, install into the repo you want agents to remember. +cd /path/to/ShadowFrog +# Choose one: +./install.sh --project /path/to/your-repo # Copilot CLI (default) +# ./install.sh --agent claude --project /path/to/your-repo # Claude Code +``` + +Commit the installed files in the target repo: + +```bash +cd /path/to/your-repo +# Stage the files for the agent you installed: +git add .github/skills/ .github/hooks/ .github/copilot-instructions.md +# git add .claude/skills/ .claude/hooks/ .claude/settings.json CLAUDE.md +git commit -m "Add ShadowFrog skills, hooks, and context" +git push +``` + +Then open that repo in your AI agent session: + +| Command | Use | +|---------|-----| +| `/shadow-frog-init` | Create the shadow | +| `/shadow-frog-update` | Refresh after code changes | +| `/shadow-frog-dream` | Explore and experiment while you're AFK | +| `/shadow-frog-meditate` | Deduplicate and resolve conflicts | +| `/shadow-frog-viewer` | Browse what's in the shadow | + +> See [Installation](#installation) for full details. + +--- + +## What is a Shadow? + +A shadow is a `.shadow/` directory that mirrors your source tree with markdown +files. It is not generated API documentation and it is not a transcript store. +It stores **discoveries**: behavioral facts, edge cases, implicit contracts, +warnings, and cross-file interactions that are useful to future agents. + +The retrieval design is **index-free** in the same sense that source-code +navigation is index-free: the codebase itself tells the agent where to look. +If an agent is editing `src/auth.py`, the corresponding knowledge lives at +`.shadow/src/auth.py.md`; if it is reasoning about `src/auth.py::login`, the +same shadow file contains the symbol-level section. Cross-file discoveries use +`file::symbol` back-pointers in `.shadow/_cross/`. No vector store, embedding +database, or separate retrieval service is required for this lookup path. + +``` +source code + user context + dream experiments + | + v + shadow-frog skills (init, update, dream) + | + v + .shadow/ (per-file discoveries, cross-cutting notes, prefs, dreams) + | + v + future agent sessions (hooks, viewer, ordinary cat/grep) +``` + +The `.shadow/` layout mirrors the repo: + +``` +your-repo/ + src/ + auth.py + db/models.py + .shadow/ + .shadowignore gitignore-syntax excludes + _index.md file list with discovery counts + _prefs.md project-wide user preferences + _cross/ cross-cutting discoveries (span multiple files) + token-expiry-config-split.md + _meta/ + state.json tracking state + _dreams/ experiment archive (reports + branch metadata) + _index.md table of all experiments (branch, parent, tip) + 20250115-143012Z-retry-logic/ one folder per experiment (mirrored from branch) + report.md structured report with YAML frontmatter + patch.diff code-only diff (excludes .shadow/) + manifest.json machine-readable discovery manifest + src/ + auth.py.md discoveries about auth.py + db/ + models.py.md discoveries about models.py +``` + +Discoveries come from three sources: + +**Agent exploration**: the agent reads code and runs experiments: +```markdown +- authenticate_user() silently returns None on expired tokens + instead of raising. 3 of 7 callers don't check the return value. + _(verified, source: exploration, labels: [bug])_ +``` + +**User knowledge**: things you tell the agent during conversation: +```markdown +- The retry logic here took 3 iterations to get right -- it handles + a subtle race condition during rolling deployments. Do not simplify. + _(verified, source: user)_ +``` + +**Collaborative work**: insights from debugging, refactoring, etc.: +```markdown +- While debugging issue #42, discovered that process_batch() silently + drops items exceeding 1MB -- logged at DEBUG level only. + _(verified, source: interaction)_ +``` + +Good discoveries are claims a future agent can act on, not summaries of what a +function is named. Prefer "silently returns `None` on expired tokens" over +"handles token expiration." + +--- + +## Skills + +| Skill | What it does | When to use | +|-------|-------------|-------------| +| **shadow-frog** | Reference docs for the shadow format and conventions | Use when working in a repo with `.shadow/` | +| **shadow-frog-init** | Creates `.shadow/` with structural templates for every file | Once per repo | +| **shadow-frog-update** | Refreshes shadows after code changes; captures conversational knowledge | After commits, or when you share context | +| **shadow-frog-dream** | Autonomous exploration AND experimentation while you're AFK | Before lunch, overnight, weekends | +| **shadow-frog-meditate** | Deduplicates, merges, and resolves conflicting discoveries | Periodically, to keep the shadow clean | +| **shadow-frog-viewer** | Browse and query the shadow: overview, search, top-discoveries-per-file, recent, preferences, labels, dream lineage, structural invariant check | When you want to see what's in the shadow, or audit its integrity | + +**Design note:** skills are readable instructions backed by small helper +scripts for deterministic work: initializing shadows, managing dream +worktrees, validating artifacts, reconciling branches, repairing structure, +and rendering viewer outputs. The installer copies both into the target repo. + +### The Dream Skill + +Dream is ShadowFrog's active-discovery mode. Each dream is an **experiment**: +the agent implements real code, runs it, and persists the result as a **named +git branch** pushed to a remote, typically your fork. The experiment branch is +live, runnable code, but the primary output is the knowledge distilled into +`.shadow/`. + +- **Branch-based persistence**: each experiment becomes a `dream//` branch +- **Natural compounding**: future dreams branch from prior dream branches, + inheriting code + shadow from the ancestor chain +- **Shadow follows lineage**: each branch has its ancestor's shadow, not + sibling branches. The default branch accumulates all discoveries during + reconciliation. + +The experiment mode can surface tacit knowledge that was not already recorded +in the shadow. While you're away, the agent might try adding retry logic, +parallelizing a pipeline, or refactoring auth into middleware, then distill +what worked, what broke, and why into the shadow. + +**Compounding dreams**: Every experiment saves a report and branch to the +remote. Future dreams read past reports and can branch from prior experiment +branches, continuing partially useful work, avoiding dead ends, and chaining +discoveries across sessions. Dream #3 can branch from dream #1's code and pick +up where it left off. + +> Selecting which dream experiments to submit upstream is a manual +> curation step. The dream skill embeds a brief "Curating Dream +> Experiments for Upstream PRs" cheat sheet covering the maintainer +> test, devil's-advocate framing, and the 70% rejection heuristic. + +--- + +## Installation + +ShadowFrog supports both **GitHub Copilot CLI** (the default) and **Claude +Code**. The installer targets one agent's conventions at a time via +`--agent`: + +| Agent | Skills | Hooks | Context | +|-------|--------|-------|---------| +| `copilot` (default) | `.github/skills/` | `.github/hooks/hooks.json` | `.github/copilot-instructions.md` | +| `claude` | `.claude/skills/` | `.claude/settings.json` | `CLAUDE.md` | + +### Install into your repo + +ShadowFrog installs **into a specific repository**. It is never installed +globally. This is deliberate: its shadow-edit hooks should only fire inside +projects you have opted in, and a per-repo install is what enables both local +agent use and fork-based [dream experiments](#4-dream). + +Prerequisites: + +- `git` and `python3` +- GitHub Copilot CLI or Claude Code +- A git repository as the target project +- For `/shadow-frog-dream`: a pushable fork/remote and a git-tracked `.shadow/` + +```bash +cd /path/to/ShadowFrog +# Choose one: +./install.sh --project /path/to/your-repo # Copilot CLI (default) +# ./install.sh --agent claude --project /path/to/your-repo # Claude Code +``` + +This installs skills, hooks, and agent-context all at once. +Use `--no-hooks` or `--no-context` to skip individual components. + +**After installing**, commit and push so future agent sessions find the skills. +The installer prints the exact `git add` paths for your chosen agent. For +Copilot CLI: + +```bash +cd your-repo +git add .github/skills/ .github/hooks/ .github/copilot-instructions.md +git commit -m "Add ShadowFrog skills, hooks, and context" +git push +``` + +For Claude Code, stage `.claude/skills/`, `.claude/hooks/`, +`.claude/settings.json`, and `CLAUDE.md` instead. + +--- + +## Usage + +### 1. Initialize + +Open your project in Copilot CLI or Claude Code and run: + +``` +/shadow-frog-init +``` + +This scans the codebase, extracts symbols (functions, classes, constants), and +creates `.shadow/` with a template for every file. + +After init, choose how `.shadow/` should live: + +| Mode | Use when | Tradeoff | +|------|----------|----------| +| **Committed** | You want team-shared memory and `/shadow-frog-dream` | Best for compounding knowledge; `.shadow/` travels through git | +| **Gitignored** | You want local-only notes | Update, meditate, and viewer still work; dream is disabled | + +If you plan to run `/shadow-frog-dream`, commit `.shadow/` after init. + +### 2. Work Normally + +As you code and talk to the agent, ShadowFrog gives the agent places to record +what would otherwise be lost: + +- **You share context** ("don't touch the retry logic, it's subtle") → agent writes it as a `source: user` discovery +- **You debug together** → agent captures insights as `source: interaction` +- **You commit** → before the next mutating agent action, the `preToolUse` hook notices the shadow is behind HEAD (compares `state.json::last_commit` to current HEAD) and reminds the agent to run `/shadow-frog-update` + +### 3. Update + +After significant changes: + +``` +/shadow-frog-update +``` + +Detects what changed via git diff, updates symbol headings, checks if existing +discoveries still hold, and captures any unrecorded session knowledge. + +### 4. Dream + +Going AFK? Let the agent work while you're gone: + +``` +/shadow-frog-dream +``` + +The agent explores uncovered code areas, runs experiments in isolated +worktrees, pushes results as persistent dream branches, and writes what it +learns into the shadow. When you return, the shadow is richer and experiment +code is accessible on named branches. + +> **Requires a git-tracked `.shadow/`.** Dream pushes shadow updates through +> git, so if you chose "local only" (gitignored `.shadow/`) during init, dream +> is disabled. `dream-setup.sh` will tell you. The other skills (update, +> meditate, viewer) work either way. + +#### Fork-Based Workflow + +For dreaming on external repos, fork the target repo first: + +1. Fork the repo on GitHub +2. Clone your fork locally +3. Install ShadowFrog: `./install.sh --project /path/to/fork` +4. Init shadow: invoke `/shadow-frog-init` +5. Dream: invoke `/shadow-frog-dream` + +Dream branches are pushed to the fork, keeping the original repo clean. + +### 5. Meditate + +Shadow getting noisy? Clean it up: + +``` +/shadow-frog-meditate +``` + +Scans for duplicate discoveries, merges near-duplicates, and resolves +conflicting claims. Escalates hard conflicts to you. + +### 6. Browse + +Want to see what's in the shadow? + +``` +/shadow-frog-viewer --summary # counts + per-file breakdown +/shadow-frog-viewer --search "auth" # keyword search across all discoveries +/shadow-frog-viewer --recent 5 # most recent discoveries +/shadow-frog-viewer --top src/auth.py # top actionable discoveries for one file +/shadow-frog-viewer --labels bug,security # filter actionable discoveries by label +/shadow-frog-viewer --prefs # all `_prefs.md` entries (project-wide) +/shadow-frog-viewer --check-invariants # audit structural integrity +``` + +Dream experiments are listed in `.shadow/_dreams/_index.md` (dream_id, +category, verdict, title, branch, parent, tip_commit). To render an +interactive view of the dream-branch tree (chains, fresh, and full-tree +tabs), run the bundled script directly: + +``` +python3 .github/skills/shadow-frog-viewer/dream-lineage.py -o lineage.html +# (Claude Code: .claude/skills/shadow-frog-viewer/dream-lineage.py) +``` + +--- + +## How Discoveries Work + +Each discovery is anchored to a file or symbol and has these properties: + +| Property | Values | Meaning | +|----------|--------|---------| +| **Status** | `verified` / `uncertain` / `refuted` | Has the claim been confirmed? | +| **Source** | `exploration` / `user` / `interaction` | Where did this knowledge come from? | +| **Labels** | `bug`, `performance`, `security`, `feature-gap`, `tech-debt` | Optional; marks actionable discoveries | + +### Trust Hierarchy + +| Rank | Source | Trust | +|------|--------|-------| +| 1 | `source: user` | Highest; human stated it. Always verified. | +| 2 | `source: interaction` | Emerged from collaborative work. Always verified. | +| 3 | `verified, source: exploration` | Agent confirmed via code analysis or tests. | +| 4 | `uncertain` | Plausible but unconfirmed. | +| 5 | `refuted` | Known wrong; skip. | + +### Searching the Shadow + +```bash +cat .shadow/src/auth.py.md # read a file's shadow +cat .shadow/_prefs.md # project-wide preferences +grep -rl "src/auth.py::" .shadow/_cross/ # cross-cutting discoveries +grep -rl "error.handling\|exception" .shadow/ --include="*.md" # search by topic +grep -r "source: user" .shadow/ --include="*.md" # all user knowledge +``` + +--- + +## Repository Structure + +For contributors, the main directories are: + +| Path | Purpose | +|------|---------| +| `skills/` | The six ShadowFrog skills and their helper scripts | +| `hook-templates/` | Copilot CLI and Claude Code hook configs plus shared hook scripts | +| `examples/coupon-demo/` | Tiny worked example with a real `.shadow/` | +| `eval/` | Evaluation methodology and results dashboard | +| `tests/` | Pytest suite for installer behavior, hooks, and skill helpers | + +--- + +## Tests + +ShadowFrog ships with a comprehensive test suite: **1,063 tests, 76% line +coverage with the declared dev dependencies, and no mocked helper layers**. +Tests exercise the real Python scripts and shell hooks against temporary +shadow trees and git repositories. + +Run the suite locally: + +```bash +pip install -r requirements-dev.txt # pytest, pytest-cov, pathspec +python3 -m pytest # all 1,063 tests +python3 -m pytest tests/skills/ # just the skill-script tests +python3 -m pytest -k viewer # everything matching "viewer" +python3 -m pytest --cov=skills # coverage report +``` + +Test layout mirrors the source layout: `tests/skills/shadow_frog_viewer/` +tests `skills/shadow-frog-viewer/`, etc. + +--- + +## Responsible AI + +ShadowFrog is a research project. Before using it, please review our +[Responsible AI transparency note](RESPONSIBLE_AI.md), which covers intended +uses, out-of-scope uses, evaluation, limitations, and best practices. + +--- + +## License + +This project is licensed under the MIT License. See the [LICENSE](LICENSE) file for details. + +--- + +

        + Froggy team logo +
        + Built by the Froggy team +

        diff --git a/RESPONSIBLE_AI.md b/RESPONSIBLE_AI.md new file mode 100644 index 0000000..07eb8a0 --- /dev/null +++ b/RESPONSIBLE_AI.md @@ -0,0 +1,103 @@ +# ShadowFrog + +## Overview + +ShadowFrog is a suite of AI coding agent skills that builds and maintains a shadow knowledge base for any software codebase. It turns idle coding-agent time into autonomous discovery loops: the agent explores source code, runs experiments in isolated branches, and records behavioral insights (edge cases, implicit contracts, cross-file interactions) in a structured .shadow/ directory that mirrors the repository. Knowledge compounds across sessions, so an agent returning to the same codebase can recall what it previously learned rather than rediscovering it from scratch. + +ShadowFrog is implemented entirely as prompt instructions and lightweight helper scripts (Python, Bash) that plug into existing AI agent harnesses such as GitHub Copilot CLI and Claude Code. It does not bundle or fine-tune any machine learning models; it relies on the host agent's LLM for reasoning. The system also captures knowledge shared by human developers during conversations, treating user-provided insights as the highest-trust source. All data is stored locally in plain-text Markdown and JSON files within the repository, with no external service dependencies beyond the configured git remote. + +### What Can ShadowFrog Do + +ShadowFrog was developed to give AI coding agents a persistent, compounding memory of the codebases they work in. Without it, every agent session starts from zero; with it, prior discoveries about how the code actually behaves carry forward. Specifically, ShadowFrog enables an agent to: (1) initialize a shadow knowledge base by scanning a repository's source files and extracting its symbol structure; (2) autonomously explore and experiment with the codebase during idle time ("dreaming"), recording behavioral findings such as hidden edge cases, implicit contracts between modules, and latent bugs; (3) capture knowledge shared by human developers during normal coding conversations; (4) navigate and query the accumulated knowledge base at the symbol level when working on future tasks; and (5) maintain the shadow over time through incremental updates, deduplication, and conflict resolution. + +The system is designed for a research audience studying how AI agents can build and leverage long-term understanding of software. It operates entirely within the user's local repository and git workflow, producing human-readable Markdown artifacts. During autonomous exploration ShadowFrog may write and run throwaway experiments in isolated, disposable git branches, but it does not merge or ship that experimental code into your production branches on its own — only the resulting behavioral discoveries (not the experiment code) are integrated into the knowledge base; it builds a knowledge layer that the host agent can consult when performing downstream tasks such as bug fixing, code review, or feature planning. + +A detailed discussion of ShadowFrog, including how it was developed and tested, can be found in our [blog post](https://microsoft.github.io/debug-gym/blog/2026/06/shadow-frog/). + +### Intended Uses + +ShadowFrog is best suited for software developers and researchers who want to give their AI coding agents persistent, compounding knowledge about a codebase. Typical use cases include running autonomous exploration to surface latent bugs or architectural patterns, equipping an agent with pre-built context before bug-fixing or feature-planning tasks, and capturing institutional knowledge from developer conversations so it survives beyond a single session. + +ShadowFrog is being shared with the research community to facilitate reproduction of our results and foster further research in this area. + +ShadowFrog is intended to be used by domain experts who are independently capable of evaluating the quality of outputs before acting on them. Shadow discoveries are agent-generated behavioral claims that may be incorrect or stale; developers should treat them as hypotheses to verify, not ground truth. + +### Out-of-Scope Uses + +ShadowFrog is not well suited for fully autonomous code deployment without human review, safety-critical systems where unverified behavioral claims could mask defects, or as a substitute for formal verification, static analysis, or security auditing tools. + +We do not recommend using ShadowFrog in commercial or real-world applications without further testing and development. It is being released for research purposes. + +ShadowFrog was not designed or evaluated for all possible downstream purposes. Developers should consider its inherent limitations as they select use cases, and evaluate and mitigate for accuracy, safety, and fairness concerns specific to each intended downstream use. + +Without further testing and development, ShadowFrog should not be used in sensitive domains where inaccurate outputs could suggest actions that lead to injury or negatively impact an individual's legal, financial, or life opportunities. + +We do not recommend using ShadowFrog in the context of high-risk decision making (e.g. in law enforcement, legal, finance, or healthcare). + +## How to Get Started + +To begin using ShadowFrog, follow the installation and usage instructions in the repository [README.md](https://github.com/microsoft/ShadowFrog/blob/main/README.md). + +## Evaluation + +ShadowFrog was evaluated on its ability to: (1) navigate and retrieve relevant shadow knowledge given a file path (read-path recall); (2) independently discover known real-world bugs through autonomous exploration without being given a problem statement (blind bug hunting on SWE-Bench Verified and at scale on SWE-Smith); (3) improve bug-fix success rates by providing pre-built shadow context to a coding agent (bug fixing on SWE-Bench Verified); and (4) generate higher-quality, more architecturally grounded feature ideas compared to a no-shadow baseline (feature ideation across 8 open-source repositories, blind-judged by an ensemble of three LLMs). + +A detailed discussion of our evaluation methods and results can be found in our [blog post](https://microsoft.github.io/debug-gym/blog/2026/06/shadow-frog/). + +### Evaluation Methods + +We used recall (file/function level), LLM-judge verdict rates, alignment to shipped features, and blind-judged quality scores (Groundedness, Insight, User Impact, Spec Clarity) to measure ShadowFrog's performance. + +We compared the performance of ShadowFrog against a matched no-shadow baseline using SWE-Bench Verified, SWE-Smith, and a feature ideation benchmark across 8 open-source repositories. + +The model used for evaluation was Claude Opus 4.6, running inside the GitHub Copilot CLI agent harness. Cross-LLM robustness was verified with Claude Opus 4.7 and GPT-5.5 as independent judges. + +Results may vary if ShadowFrog is used with a different model based on its unique design, configuration, and training. + +### Evaluation Results + +At a high level, we found that ShadowFrog performed strongly on knowledge retrieval and blind bug discovery, modestly on bug fixing, and with a distinctive quality profile on feature ideation. Specifically: the read path achieves ~98% recall at realistic tool-call budgets. On blind bug hunting (no problem statement given), the agent independently locates 88% of real-world bugs to the correct subsystem and 22% exactly, purely from idle-time exploration; at scale (20 repos × 100 stacked bugs), it leads the no-shadow baseline by +25.4 percentage points at peak. On bug fixing (50 SWE-Bench Verified tasks), ShadowFrog resolves 82.0% vs the baseline's 77.3% (+4.7 pp), though most of the lift traces to the structured workflow rather than shadow content itself, highlighting a consumption bottleneck we plan to address. On feature ideation (3,310 ideas blind-judged), ShadowFrog generates ideas rated higher on insight (+0.40) and user impact (+0.24), while trading off slightly on specification clarity, a gap that largely dissolves when controlling for problem size. We refer readers to our [blog post](https://microsoft.github.io/debug-gym/blog/2026/06/shadow-frog/) for detailed evaluation results. + +## Limitations + +ShadowFrog was developed for research and experimental purposes. Further testing and validation are needed before considering its application in commercial or real-world scenarios. + +ShadowFrog was designed and tested using the English language. Performance in other languages may vary and should be assessed by someone who is both an expert in the expected outputs and a native speaker of that language. + +Outputs generated by AI may include factual errors, fabrication, or speculation. Users are responsible for assessing the accuracy of generated content. All decisions leveraging outputs of the system should be made with human oversight and not be based solely on system outputs. + +ShadowFrog inherits any biases, errors, or omissions produced by its base model. Developers are advised to choose an appropriate base LLM/MLLM carefully, depending on the intended use case. + +There has not been a systematic effort to ensure that systems using ShadowFrog are protected from security vulnerabilities such as indirect prompt injection attacks. Any systems using it should take proactive measures to harden their systems as appropriate. + +Shadow staleness. Discoveries are anchored to specific code symbols and file paths. As the codebase evolves, shadows can become stale or reference code that no longer exists. The system includes staleness detection (comparing the last-update commit to HEAD), but users should not assume older discoveries remain accurate without re-verification. + +Plain-text persistence. All shadow content (including user-shared knowledge) is stored as unencrypted Markdown files within the repository. If the .shadow/ directory is committed and pushed, its contents become visible to anyone with repository access. Users should avoid sharing sensitive information (credentials, PII, proprietary business logic) through the shadow capture mechanism. + +## Best Practices + +Better performance can be achieved by running multiple compounding dream sessions rather than a single long one, keeping the shadow up to date after significant code changes, and capturing domain knowledge from developers during conversations. Directing exploration toward under-explored, high-fan-in areas of the codebase (via the coverage map) also improves discovery yield. + +We strongly encourage users to use LLMs/MLLMs that support robust Responsible AI mitigations, such as Azure Open AI (AOAI) services. Such services continually update their safety and RAI mitigations with the latest industry standards for responsible use. For more on AOAI's best practices when employing foundations models for scripts and applications: + +- [What is Azure AI Content Safety?](https://learn.microsoft.com/en-us/azure/ai-services/content-safety/overview) +- [Overview of Responsible AI practices for Azure OpenAI models](https://learn.microsoft.com/en-us/legal/cognitive-services/openai/overview) +- [Azure OpenAI Transparency Note](https://learn.microsoft.com/en-us/legal/cognitive-services/openai/transparency-note) +- [OpenAI's Usage policies](https://openai.com/policies/usage-policies) +- [Azure OpenAI's Code of Conduct](https://learn.microsoft.com/en-us/legal/cognitive-services/openai/code-of-conduct) + +## License + +MIT License + +Nothing disclosed here, including the Out of Scope Uses section, should be interpreted as or deemed a restriction or modification to the license the code is released under. + +## Trademarks + +This project may contain trademarks or logos for projects, products, or services. Authorized use of Microsoft trademarks or logos is subject to and must follow Microsoft's Trademark & Brand Guidelines. Use of Microsoft trademarks or logos in modified versions of this project must not cause confusion or imply Microsoft sponsorship. Any use of third-party trademarks or logos are subject to those third-party's policies. + +## Contact + +This research was conducted by members of [Microsoft Research](https://www.microsoft.com/en-us/research/). We welcome feedback and collaboration from our audience. If you have suggestions, questions, or observe unexpected/offensive behavior in our technology, please contact us at [debug-gym@microsoft.com](mailto:debug-gym@microsoft.com) + +If the team receives reports of undesired behavior or identifies issues independently, we will update this repository with appropriate mitigations. diff --git a/agent-context.md b/agent-context.md new file mode 100644 index 0000000..4118155 --- /dev/null +++ b/agent-context.md @@ -0,0 +1,22 @@ +## ShadowFrog Knowledge Base + +This project uses a `.shadow/` knowledge base with verified discoveries about non-obvious code behavior. **You MUST consult the shadow before making any code change.** + +1. **Check the shadow first** — before editing any file, read its shadow: + ``` + cat .shadow/.md + ``` +2. **Check preferences** — `cat .shadow/_prefs.md` for project conventions +3. **Check cross-cutting** — `cat .shadow/_cross/*.md` for multi-file patterns +4. **Act on what you find** — apply what you learn from the shadow to your work. +5. **After making changes** — run `/shadow-frog-update` to capture learnings + +The shadow contains discoveries from code analysis and user conversations. Always consult it before making assumptions about code behavior. + +### Key directories + +- `.shadow/.md` — per-file shadows with symbol-level discoveries +- `.shadow/_cross/` — cross-cutting discoveries spanning multiple files +- `.shadow/_prefs.md` — project-wide user preferences and conventions +- `.shadow/_dreams/` — experiment archive from dream runs (reports + implementation diffs) +- `.shadow/_meta/state.json` — tracking state (last commit, counts) diff --git a/claude.md b/claude.md new file mode 100644 index 0000000..b5a8fdc --- /dev/null +++ b/claude.md @@ -0,0 +1,272 @@ +# ShadowFrog — AI Agent Guidelines + +## Project Overview + +ShadowFrog is a suite of AI coding agent skills that build and maintain shadow knowledge bases for any codebase. It consists of 6 skills (`shadow-frog`, `shadow-frog-init`, `shadow-frog-update`, `shadow-frog-dream`, `shadow-frog-meditate`, `shadow-frog-viewer`) and associated hooks, all sharing a common `.shadow/` filesystem. This is a **distributable skills package** — users install it into their own projects via `install.sh`. + +## Repository Structure + +``` +ShadowFrog/ + skills/ + shadow-frog/SKILL.md Main entrypoint (docs, reference system, search) + shadow-frog-init/ First-time setup (create .shadow/) + SKILL.md Init instructions + fallback steps + shadow-init.py Python helper script + shadow-frog-update/SKILL.md Incremental update (after changes) + shadow-frog-dream/ Autonomous exploration + experimentation (AFK mode) + SKILL.md Dream instructions + pipeline phases + dream-setup.sh Worktree + branch creation + dream-validate.py Pre-push artifact validation + dream-reconcile.py Merge dream branches into main's shadow + dream-coverage.py Exploration coverage map + dream-cleanup.sh Safe per-worktree cleanup (replaces inline snippet) + dream-gc.sh Orphan-worktree sweep (defense-in-depth) + _worktree_safety.py Shared safety gate for rm-rf paths + shadow-frog-meditate/SKILL.md Dedup, merge, and resolve conflicting discoveries + shadow-frog-viewer/ Browse and query the shadow knowledge base + SKILL.md Query instructions + shell fallbacks + shadow-viewer.py Python helper script + dream-lineage.py Dream lineage visualization + hook-templates/ + shadow-frog-hooks.json Copilot CLI hook config (sessionStart, preToolUse) + claude-settings.json Claude Code hook config (.claude/settings.json: SessionStart, PreToolUse) + scripts/ + shadow-frog-check-init.sh Session-start: check .shadow/ exists + shadow-frog-pre-tool.sh Pre-tool: shadow awareness + knowledge capture reminder + examples/ + coupon-demo/ Example of what `.shadow/` looks like (3 source files + .shadow/) + cart.py Cart logic (coupon lookup + total calculation) + inventory.py Coupon validation (cross-file case mismatch) + test_cart.py Passing tests for existing coupons + README.md Tour of the .shadow/ for this demo + .shadow/ Agent-discovered knowledge base (init + dream) + eval/ Systematic eval — see eval/README.md + README.md Methodology + results + results_dashboard.html Interactive results dashboard + swesmith/manifests_canonical/ SWE-Smith stacked-bug task manifests + agent-context.md Always-on context for project instructions + install.sh Install skills + hooks into a project repo + README.md User-facing documentation + claude.md This file +``` + +## Key Principles + +1. **General-purpose** — ShadowFrog works with any codebase, any language. Skills and examples must be language-agnostic. Never assume Python, JS, or any specific stack. + +2. **Discoveries, not descriptions** — Shadows contain behavioral insights (edge cases, implicit contracts, non-obvious interactions), NOT code summaries or descriptions. Write "silently returns None on expired tokens" not "handles token expiration". + +3. **Two sources of knowledge** — The shadow captures knowledge from autonomous code analysis (`source: exploration`) AND from user-agent conversations (`source: user`, `source: interaction`). User-shared knowledge is the highest-trust source — capture it immediately, anchored to the exact file and symbol. + +4. **Symbol-level granularity** — Shadows mirror the codebase at the symbol level, not just file level. Every class, function, and method has a `##` section in its shadow file. This enables precise bidirectional lookup: code→shadow and shadow→code. + +5. **Bidirectional references are the core mechanism** — The entire system rests on robust, accurate references between code and shadow. The canonical format is `file::symbol` (e.g., `src/auth.py::UserAuth.validate`). Seven invariants must hold (see `shadow-frog/SKILL.md`). When editing skills, never break reference integrity. + +6. **Trust hierarchy** — `source: user` (always verified) > `source: interaction` (always verified) > `verified` exploration > `uncertain` > `refuted`. + +7. **Cross-cutting is critical** — `_cross/` discoveries span multiple files and are stored once. Per-file shadows have `## Cross-References` back-pointers. Always maintain both directions. + +8. **No backward compatibility** — When refactoring, only keep the latest code. No re-exports, deprecation wrappers, or compatibility shims. + +## Important Conventions + +### Discovery Format + +Canonical formal spec: `/shadow-frog`. The shapes below are the minimum an agent needs to write a valid discovery from claude.md alone. + +Per-file discovery (anchored by `file::symbol` heading; labels and `Also involves:` are optional): +``` +- + _(, source: [, labels: [bug, security]])_ + Also involves: `file::symbol`, `file::symbol` +``` + +Cross-cutting (`_cross/.md`, slug = kebab-case from title, e.g. "DB connection lifecycle" → `db-connection-lifecycle.md`): +``` +# + +**Category**: <pattern|behavior|edge-case|contract|performance|intent|warning|history|convention> +**Refs**: +- `file::symbol` + +**Discovery**: <behavioral statement> + +_(<verified|uncertain|refuted>, source: <exploration|user|interaction>)_ +``` + +Preference (`_prefs.md` — project-wide, no file/symbol anchor): +``` +- <preference or convention> + _(source: <user|interaction>)_ +``` + +- Labels (lowercase, comma-separated): `bug`, `performance`, `security`, `feature-gap`, `tech-debt`. Only for actionable discoveries. +- `Also involves:` always uses `file::symbol`, never bare file paths. +- `Dream report: _dreams/<dream-id>/` is optional — only for experiment-derived discoveries. + +### Verification +- Observe-based: read source at `file::symbol`, trace logic, confirm claim. +- Do-based: write and run a short test/script to confirm or refute. +- `source: user` and `source: interaction` → always `verified`. + +### Dedup +- Before writing, read existing discoveries at the target symbol. +- Same claim → update existing. Extends existing → merge. Contradicts → keep both, mark weaker `refuted`. +- If `_No discoveries yet._` placeholder → replace it. If discoveries already exist → append after them. + +### Shadow File Headings +- Top-level symbols: `##` heading with symbol in backticks +- Nested symbols: `###` heading with symbol in backticks + +Examples: +``` +## `authenticate_user` +## `class UserAuth` +### `UserAuth.validate` +``` +- `## Cross-References` at the bottom of every per-file shadow + +### Cross-Cutting Files (`_cross/<slug>.md`) +- Use `**Refs**:` with `file::symbol` entries +- Category field values: pattern, behavior, edge-case, contract, performance, intent, warning, history, convention + +### state.json Schema (canonical) +```json +{ + "version": 1, + "initialized_at": "<ISO timestamp>", + "last_update_at": "<ISO timestamp>", + "last_commit": "<full 40-char HEAD SHA>", + "last_update_type": "init|auto|manual|dream|meditate", + "total_files": 0, + "total_symbols": 0, + "total_discoveries": 0, + "dream_cycles_completed": 0 +} +``` + +`total_discoveries` counts **per-file discoveries only** (excludes `_cross/` +and `_dreams/`). Cross-cutting discoveries are tracked separately via +`ls .shadow/_cross/*.md | wc -l`. + +### Dream Reports (`_dreams/`) + +Dream experiment reports are archived in `_dreams/` for compounding knowledge +across dream sessions. Each experiment gets a folder named `YYYYMMDD-HHMMSSZ-slug`. + +Report frontmatter (YAML): +```yaml +--- +dream_id: "20250417-183012Z-retry-logic" +category: feature design +verdict: useful | dead_end +base_commit: abc1234def5678 +branch: "dream/myproject/20250417-183012Z-retry-logic" +parent_branch: "main" +remote: "origin" +related_symbols: + - "src/http.py::HttpClient.send" +builds_on: [] +--- +``` + +Note: `tip_commit` is NOT stored in the report (chicken-and-egg problem). +The reconciler derives it via `git rev-parse origin/$BRANCH` and records +it in `_dreams/_index.md`. + +- `_dreams/_index.md` — table of all experiments (dream_id, category, verdict, title, branch, parent, tip_commit). The `parent` column is a **branch name** (the parent dream's branch, or `main` if rooted at the base branch) — never a dream_id. Both the reconciler (writer) and `dream-lineage.py` (reader) treat it as a branch name; meditate's index repair resolves to and writes the parent row's branch. +- `_dreams/<dream-id>/report.md` — structured report with frontmatter (mirrored from dream branch) +- `_dreams/<dream-id>/patch.diff` — code-only diff against `base_commit` (excludes `.shadow/`) +- `_dreams/<dream-id>/manifest.json` — machine-readable discovery manifest (on dream branch) +- Per-file discoveries cross-reference with `Dream report: _dreams/<dream-id>/` +- `_dreams/` is excluded from discovery counts and viewer file listings + +## SKILL.md Format + +Each skill has a `SKILL.md` with YAML frontmatter: + +```yaml +--- +name: skill-name +description: >- + One-paragraph description. This is what the agent matches + against to decide when to load the skill. +scripts: # optional — list script filenames in this directory + - my-script.py +--- + +# Skill Title + +Markdown instructions for the agent. +``` + +The `description` field is critical — it determines when the agent auto-loads the skill. Make it specific and action-oriented. + +The `scripts` field (optional) lists executable scripts bundled with the skill. Scripts live in the same directory as SKILL.md. With a project install, agents can find them via: +```bash +python3 .github/skills/<skill-name>/<script>.py +# or, for Claude Code: +python3 .claude/skills/<skill-name>/<script>.py +``` + +## Hook Format + +Two agent platforms, two hook-config shapes, **one set of shared scripts**: + +**Copilot CLI** — `hook-templates/shadow-frog-hooks.json` (installed to +`.github/hooks/hooks.json`): +- `sessionStart` / `preToolUse` events; handler uses `bash:` + `timeoutSec` +- Reads context from the top-level `additionalContext` output key. For + `preToolUse`, support for `additionalContext` is undocumented in the 2026 + hooks reference but explicitly confirmed in the copilot-cli v1.0.24 + changelog. If Copilot ever removes this, the `sessionStart` reminder + remains; only the pre-edit injection silently no-ops. + +**Claude Code** — `hook-templates/claude-settings.json` (merged into +`.claude/settings.json`): +- `SessionStart` / `PreToolUse` events (PascalCase), matcher-group nesting, + `command:` + `timeout`, scripts referenced via `${CLAUDE_PROJECT_DIR}` +- Reads context from the nested `hookSpecificOutput.additionalContext` key + +**Shared script contract** (both `check-init.sh` and `pre-tool.sh`): +- Receive JSON on stdin; parse both camelCase (Copilot `toolName`/`toolInput`) + and snake_case (Claude `tool_name`/`tool_input`) field names +- Emit JSON carrying BOTH output shapes so one payload drives both agents +- Use `python3 -c "import json,sys; ..."` for JSON parsing (not grep/cut) +- Keep hooks fast (< 5 second timeout) +- **Fail-open — the hooks are advisory and MUST always exit 0.** Copilot CLI + ≥ 1.0.57 denies the tool call when a `preToolUse` command hook exits + non-zero. The scripts therefore use a **multi-layer defense** (interactive + scripts like `install.sh` and `dream-setup.sh` are the opposite — they + fail-fast): + + 1. **No `set -e`/`-u`/`pipefail`** — failing sub-steps don't abort the script. + 2. **Trap pyramid** — separate `trap 'exit 0' EXIT` AND + `trap 'exit 0' TERM HUP INT`. EXIT alone returns 143/-15 under SIGTERM + (empirically verified on bash 3.2 macOS / bash 5+ Linux), which the + runner's `timeoutSec` enforcement triggers; the TERM trap converts it to 0. + 3. **Every external call bounded** — `git`, `python3`, and viewer + subprocesses MUST run inside Python `subprocess.run(timeout=...)` + wrappers. Bash queues signals while waiting for a foreground child, so + the trap pyramid cannot save us from an unbounded hang. Total bounded + work budget is ~3.5s, leaving ≥1.5s headroom under the hook's 5s + `timeoutSec`. The previously-unbounded `git rev-parse --show-toplevel` + in pre-tool.sh was reproduced as a 31s hang in production. + 4. **`state.json` read inside Python** (`json.load(open(...))`) rather than + a shell `< redirect`, so a missing file is a caught exception instead of + an stderr leak. + 5. **Static enforcement** — `hook-templates/check-hook-failopen.py` blocks changes + that re-introduce any of: short/long-form strict-mode flags, `source`/`.` + of external files, missing EXIT or TERM trap, + comment-masquerading-as-trap, or unbounded `git` calls at bash level. + +## Development + +- SKILL.md files ARE the product — edit them directly +- Test by running skills in Copilot CLI / Claude Code +- Hook scripts are bash with python3 for JSON — keep them simple and fast +- Use `install.sh --project <repo>` (with `--agent copilot|claude`) to copy + skills, hooks, and context into a project repo +- The `examples/coupon-demo/.shadow/` must stay consistent with skill docs (same formats, same field names, matching counts) +- After any format change, audit ALL files for consistency (skills, examples, hooks, README, claude.md) diff --git a/eval/README.md b/eval/README.md new file mode 100644 index 0000000..926dfc4 --- /dev/null +++ b/eval/README.md @@ -0,0 +1,845 @@ +# ShadowFrog Evaluation Suite + +This folder is the view-time entry point for the ShadowFrog evaluation. +It contains exactly two things: + +- **[`results_dashboard.html`](results_dashboard.html)** — a single + self-contained HTML report rolling up every experiment (the xarray + Lens 1 dream-lineage sunburst is inlined via `<iframe srcdoc>`). +- **This README** — a near-reproducibility-grade description of *what we + did* across all five sub-experiments: corpus, environment, pipeline, + prompts, scoring rubrics, and key design decisions. Numbers are quoted + when they describe scope (corpus sizes, run counts, axis sweeps) but + no results are reported here — for those, open the dashboard. + +The README is intended to be **self-contained**: every step, prompt, +and scoring decision is described in enough detail that a reader could +re-derive the pipeline from public sources (the corpora are all public +GitHub repos; SWE-Bench Verified and SWE-Smith are public datasets). +The raw outputs (patches, dreams, judge verdicts, logs, manifests, +shadow KBs, per-task analyses, per-arm operator runbooks, Dockerfiles) +amount to ~42 GB and are not open-sourced. + +``` +1. swebench — blind bug hunting on SWE-Bench Verified +2. swebench-fix — bug fixing with vs. without shadow knowledge +3. swesmith — stacked synthetic-bug hunting on SWE-Smith +4. feature-ideation — shadow vs. baseline feature ideation +5. navigation — H8/H10/H11 navigation hypotheses + wrong-needle probe +``` + +Common conventions across all five: +- **Model**: `claude-opus-4.6` for the agent, the judges, and (where + applicable) needle authoring / perturbation / question generation. +- **Agent runtime**: GitHub Copilot CLI invoked non-interactively + (`copilot -p "$PROMPT"`) with `--output-format json + --no-custom-instructions` and the appropriate `--allow-all*` flag(s) + to permit unattended tool use. The exact flag combination is repeated + per experiment below since it varied across the five. +- **Reproducibility seed**: `seed=42` for task sampling; `random_seed: + 1729` for the navigation eval. +- **Host/Docker split**: the agent runs on the host so it has full + Copilot CLI + ShadowFrog skills infrastructure; `python` / `pytest` / + `pip` are routed into a per-task Docker container via wrappers in + `.docker-bin/`. File operations (read/edit/grep/git) run on the host + against bind-mounted files so Docker sees them too. +- **Dream branches isolated**: every experiment uses bare local clones + as `origin` so dream branches never push to upstream. + +--- + +## 1. Blind Bug Hunting on SWE-Bench Verified (`swebench`) + +**Question.** Can ShadowFrog's autonomous dream exploration discover +real-world bugs *without being told they exist*? No problem statement, +no failing test — just the codebase. + +### Corpus + +50 SWE-Bench Verified tasks (out of 500), stratified-sampled across 12 +Python repos (seeded random sample with `seed=42`): + +| Repo | Tasks | +|---|---:| +| astropy/astropy | 5 | +| django/django | 5 | +| matplotlib/matplotlib | 5 | +| mwaskom/seaborn | 2 | +| pallets/flask | 1 | +| psf/requests | 4 | +| pydata/xarray | 5 | +| pylint-dev/pylint | 4 | +| pytest-dev/pytest | 4 | +| scikit-learn/scikit-learn | 5 | +| sphinx-doc/sphinx | 5 | +| sympy/sympy | 5 | + +Allocation algorithm: floor-divide tasks by repo count, then redistribute +the shortfall to larger repos round-robin until total = N. Within each +repo, uniform random sampling from the SWE-Bench Verified `test` split. + +### Pipeline + +1. **Stratified sampling** (`seed=42`) selects 50 instances from the + SWE-Bench Verified `test` split. The sampler records, per task, + `instance_id` / `repo` / `base_commit` / `version` / `difficulty`, + alongside the gold patch, the test patch, and the `fail_to_pass` + test list — the last three are kept private from the agent and used + only at scoring time. +2. **Per-task setup**. For each unique repo we `git clone --bare` + upstream and, for each task on that repo, `git worktree add` at the + task's `base_commit`. `origin` in the worktree is repointed to the + local bare clone so dream branches stay private. A small metadata + file (containing only the namespace string, no bug info) is written + so the dream skill can name branches by `instance_id` when multiple + tasks share a repo. ShadowFrog is installed into the worktree with + `install.sh --project <worktree> --no-hooks`, then the hooks JSON + and accompanying scripts are copied into `.github/hooks/`. Finally + the per-task SWE-Bench Docker image + (`swebench/sweb.eval.x86_64.<owner>_1776_<repo>-<n>`) is pulled. +3. **Per-task docker env**. A sourced wrapper starts the per-task + container, bind-mounts the worktree, writes wrapper executables for + `python` / `python3` / `pytest` / `pip` / `conda` into a + `.docker-bin/` directory, and prepends that directory to PATH. File + operations stay on the host; Python execution runs inside Docker. + The wrappers detect the agent's current working directory and + translate worktree-relative paths into the container, so dreams that + create sub-worktrees under `/tmp/shadowfrog-dreams/dream-<slug>/` + for compounding experiments still see `python` / `pytest` working + at arbitrary depth. +4. **Dreams**. In each worktree the operator runs `/shadow-frog-init` + then prompts for 5 iterations of compounding dreams focusing on bug + hunting. +5. **Patch + report collection**. Each worktree (plus its dream + sub-worktrees) is walked to gather dream artifacts in two formats: + the canonical subdirectory layout + (`_dreams/<id>/{report.md, manifest.json, patch.diff}`) and a + flat single-file report-only variant (`_dreams/<id>.md`) used by + later compounding sessions. Sub-worktree dreams are attributed to + their parent task via a 4-level fallback (a `.dream_parent` marker + file → the metadata file → exact `base_commit` match → closest + ancestor by commit distance). Aggregate: **6,701 dreams across 50 + tasks, ~134 dreams per task on average**. +6. **Localization scoring**. L1 (file hit) and L2 (function hit) are + automated (described in the rubric below); L3 emits per-task LLM-judge + prompts and then merges the returned verdicts back into the summary. + +### Scoring rubric + +Three levels, "best match over all dream reports for a task": + +| Level | Name | How | +|---|---|---| +| L1 | File hit | string match: did any dream mention/touch a file in the gold patch? matches both path suffixes and Python module notation (e.g. `sklearn.metrics._ranking` ↔ `sklearn/metrics/_ranking.py`) | +| L2 | Function hit | word-boundary regex match against function names extracted from the gold diff. Extracts from (a) `@@` hunk headers, (b) `+`/`-` def/class lines, (c) **context** lines inside hunks showing `def`/`class` (so we catch the actual method when the hunk header only shows the enclosing class) | +| L3-4 | IDENTIFIED | LLM judge: bug describable from the report alone — root cause or trigger matches the gold patch | +| L3-3 | PARTIAL | LLM judge: symptom of the bug, right code path with mis-attributed root cause, or proposes a fix that would partially address the issue | +| L3-2 | ADJACENT | LLM judge: real, verified bugs in the **same function** as the gold patch but distinct from it | +| L3-1 | AREA | LLM judge: right file/module with demonstrated subsystem understanding, no in-function bug | +| L3-0 | MISSED | LLM judge: unrelated to the bug's file/module/subsystem | + +The L3 judge sees the problem statement, the gold patch, and **all** +non-trivial dream reports (reports shorter than 50 characters are +treated as empty placeholders and skipped; L1-hitting reports are +listed first). The "best-match" semantic means a single high-level +report among 100+ wins +the verdict. ADJACENT exists because "5 verified real bugs in the exact +gold function" is meaningful work even if the agent missed the specific +SWE-Bench bug. + +### Pipeline summary + +End-to-end the experiment is: stratified sampler → per-task setup +(bare clone + worktree + ShadowFrog install + Docker pull) → sourced +docker-env wrapper → 5 compounding dream iterations per task → +collect dream reports + patches → score L1/L2 automatically and L3 via +LLM judge. + +--- + +## 2. Bug Fixing on SWE-Bench Verified (`swebench-fix`) + +**Question.** Does pre-collected shadow knowledge help an agent *fix +known bugs* better than a baseline agent without it? Same corpus, same +Docker, but here the agent **is told** about the bug. + +### Corpus + +The same 50-task SWE-Bench Verified subset as `swebench`, locked in a +task manifest (12 repos: astropy 5, django 5, matplotlib 5, xarray 5, +scikit-learn 5, sphinx 5, sympy 5, requests 4, pylint 4, pytest 4, +seaborn 2, flask 1). **3 seeds per task per arm** for variance +(50 tasks × 3 seeds × 2 arms = 300 runs). + +### Pre-built shadow knowledge bases + +The SF arm consumes a per-repo shadow KB built once before the fix +experiment (~88 MB across all 12 repos). The KBs were built by an +independent dream campaign with no access to the SWE-Bench problem +statements, so the shadow's coverage of any particular bug is +incidental, not engineered. The pre-filter (next step) consumes these. + +### Pipeline + +1. **Setup**. For each of the 50 tasks, create a per-task worktree at + `base_commit`, pull the SWE-Bench Docker image, bind-mount the + worktree, write a `TASK.md` containing the official problem + statement. +2. **Pre-filter (SF arm only)**. Given the task directory (which + contains `TASK.md` and a copy of `.shadow/`), extract file paths and + symbols from the problem statement using: + - File-path regex (e.g. `path/to/file.py`) + - Module dotted refs (`django.db.models.query`) → `django/db/models/query.py` + - Backtick-quoted paths + - **Distinctive tokens**: CamelCase, snake_case, ALL_CAPS only — common + words are filtered against a 124-word stopword list + - **Dream-index search**: matches dream experiment titles against + problem keywords; the matched dreams' files become candidates + - **BUG-tag proximity boost**: shadow files with `BUG:` markers + within the matched windows score higher + - **Cross-reference scoring**: require ≥2 keyword matches to promote + a cross-reference (reduces false positives) + + The pre-filter writes a `SHADOW_HINTS.md` capped at ~4 KB into the + task dir. Tasks with no extracted keywords or no matching shadow + content produce an empty hint file, in which case the agent proceeds + with zero shadow context. +3. **Run agents** (one runner per arm), 3 seeds each, model + `claude-opus-4.6`, agent timeout 1800 s, Docker memory 4 g, batch + size 25 in parallel. SF prompt enforces an **investigate-first** + workflow (see "Prompts" below). BL prompt is a minimal 55-word + "fix the bug" instruction. +4. **Score**. Apply each candidate patch to a clean worktree in Docker, + run repo-specific test commands (a curated lookup table maps each + `(repo, version)` pair to its install / test / coverage commands — + the table is derived from SWE-Bench's `MAP_REPO_VERSION_TO_SPECS` + plus a few env-drift overrides), parse with SWE-Bench's log parsers, + compute the `fail_to_pass` set, and declare RESOLVED iff all + `fail_to_pass` tests pass. Aggregates are emitted per arm × seed. + +### Prompts + +SF prompt structure (abridged; the full text is the literal block +below plus the four numbered workflow steps that follow): + +``` +RULES: +- Do not fix pre-existing environment issues … +- Do not run pip install … +- Do not edit setup.py, setup.cfg, pyproject.toml, tox.ini unless the + bug originates there … +- When using grep or find, exclude the .shadow/ directory … + +WORKFLOW — follow this order: +Step 1: INVESTIGATE THE BUG (do this first, before anything else) + Read the bug report. Reproduce it. grep / read / trace to identify + buggy file(s) and function(s). Form your own hypothesis about + root cause. +Step 2: CHECK PRIOR ANALYSIS (optional) + If SHADOW_HINTS.md exists, read it after Step 1. Use any relevant + insights as hints, but verify everything against the actual source. + If no SHADOW_HINTS.md exists, skip this step entirely. +Step 3: IMPLEMENT THE FIX + Write the fix based on your investigation. Aim for correctness + over minimality. +Step 4: VERIFY COMPLETENESS + - If the fix touches a shared helper, check sibling methods + - Run git status — if you created new files, make sure they're tracked + - Run relevant tests to verify + - An empty patch means the bug is not fixed +``` + +BL prompt: minimal (~55-word fix-the-bug instruction) — same task +environment, no shadow, no workflow scaffolding. + +Patch-comparison excludes: `.shadow`, `.docker-bin`, `TASK.md`, +`TASK_INFO.json`, `SHADOW_HINTS.md`, `.github`, `.copilot`, `.claude`, +`test-output`, `*.pyc`, `__pycache__`, `*.so`, `*.o`, +`copilot-instructions.md`. + +### Key design decisions + +- **Pre-filter, not browsing.** Free browsing of the shadow distracts the + agent — too much off-target content drowns out the few relevant hints. + The investigate-first + pre-filter design lets the agent see only + shadow content already judged relevant to the bug at hand. +- **Source-over-shadow on conflict.** If shadow content disagrees with + source, trust the source — discoveries can become stale. +- **No index browsing.** The agent never sees `.shadow/_index.md` or + `_dreams/` directories — index browsing causes "attention theft": the + agent spends turns navigating the index rather than acting on the + hints already extracted. + +### Pipeline summary + +End-to-end the experiment is: per-task setup (worktree + Docker + +problem statement) → SF-only pre-filter (extract problem-statement +keywords → match against the pre-built shadow KB → write a +≤4 KB `SHADOW_HINTS.md` into the task dir) → run the SF and BL agents +3× each → score by re-applying each candidate patch in a clean Docker +worktree and checking that all `fail_to_pass` tests pass. + +--- + +## 3. Stacked Synthetic Bug Hunting on SWE-Smith (`swesmith`) + +**Question.** How does blind bug hunting scale when we don't need to +treat each bug as a separate task? Stack many synthetic bugs into one +container and measure coverage from longer dream sessions. + +### Corpus + +20 Python repos drawn from SWE-Smith (which catalogs 50,000+ synthetic +bugs across 128 repos), 100 non-conflicting synthetic bugs each → 2,000 +bugs total. Bug selection is greedy non-conflicting with `seed=42` +(SWE-Smith is ~90% single-file, so conflicts are rare). Unique bugged +files per repo: 11 (pypika) to 121 (conan). + +``` +cantools, conan, pypika, oauthlib, jinja (pallets), pandas, +paramiko, pyasn1, patsy (pydata), pydicom, pygments, astroid, +python-docx, trio, safety, python-pptx, deepdiff, sqlfluff, +sunpy, sqlglot +``` + +### Pipeline + +1. **Manifest generation**. We pick 100 non-conflicting bugs per repo + from the SWE-Smith HuggingFace dataset (seed=42, greedy hunk-overlap + detection with a 4-line margin). The manifests are reproducible: + re-running the selection produces a byte-identical result for 19/20 + repos (pandas-dev differs due to unrecoverable original generation + code but is equally valid). One manifest per repo is committed so + scoring is always reproducible. +2. **Bug stacking**. For each repo we pull the SWE-Smith Docker image, + start a container with `--network none --memory=8g --memory-swap=8g`, + then inside the container: + - apply each of 100 bug patches forward (clean → buggy); + - remove **all** test files — pattern-based (test directories, + `test_*.py`, `*_test.py`, `conftest.py`) plus manifest-based (any + `fail_to_pass` paths); + - `rm -rf .git && git init && git add -A && git commit -m 'initial state'` + so there's no recoverable history; + - record the apply outcomes for each bug. +3. **Lock-down**: every run uses `--network none` to block GitHub/PyPI + fetches of upstream source or tests. +4. **Run dreams** (SF arm): 10 structured dream sessions × 12 sub-agents + each = 120 agents per repo. +5. **Run baseline** (BL arm): 120 independent parallel agents per repo. + Same Docker image, same bug stack, but no ShadowFrog skills installed + and a minimal prompt (verbatim): + + ``` + You are given a codebase to explore. Your task is to find bugs, edge + cases, and issues in the code. Examine the source files, trace logic + flows, and identify problems. + Output your findings as a JSON array. Each finding must have: + "file" / "symbol" / "line" / "description" (1-2 sentences). + Write ONLY valid JSON to: {output_path} + ``` + + The 120-vs-120 design lets us match SF's total compute exactly while + varying only the SF-vs-BL axis. +6. **Budget sweep**. Re-score the same transcripts at 10 budgets + `b12, b24, …, b120` controlling how many sessions / agents are + included. For SF, `b<N>` = N sub-agents from `N/12` sessions; for + BL, `b<N>` = N agents. +7. **Score**: + - **Stacked scorer**: L1/L2/L3 (exact intersection of changed-line + sets, per file) against the canonical manifest. + - **Budget judge**: an LLM judge re-grades the top candidates at + each budget on the IDENTIFIED rubric and emits a per-budget + verdict file. + - These are rolled up into a single per-repo results blob that + drives the final figure. +8. **Render** the headline figure: a 3-panel plot of L1 / L2 / Judge + IDENTIFIED% as a function of budget. + +### Scoring rubric + +| Level | Name | Definition | +|---|---|---| +| L1 | File hit | agent explored/modified a file containing a bug | +| L2 | Function hit | agent's patch touches the bugged function | +| L3 | Line overlap | exact intersection of changed-line sets per file (no tolerance) | +| Judge | IDENTIFIED% | per-budget Opus-4.6 judge verdict, % of 100 bugs | + +L3 uses actual changed lines (not full hunk ranges) to avoid context-line +false positives. + +### Anti-cheat hardening (locked-in container) + +| Threat | Mitigation | +|---|---| +| Run pytest to see failures | All test files removed (pattern + manifest) | +| `git reset` to recover clean | `rm -rf .git && git init` (no history) | +| Fetch from GitHub/PyPI | `--network none` on container | +| Inspect git reflog/fsck | Single commit, no dangling objects | + +### Pipeline summary + +End-to-end the experiment is: deterministic bug-stack manifests +(100 non-conflicting bugs per repo, seed=42) → stack into a per-repo +container with tests removed, git history wiped, and +`--network none` → run 120 agents per repo per arm → budget-sweep +re-scoring at b12…b120 → aggregate into a single results blob and +plot the L1 / L2 / Judge-IDENTIFIED% headline panels. + +--- + +## 4. Feature Ideation (`feature-ideation`) + +**Question.** Does an agent with a `.shadow/` knowledge base propose +*better feature ideas* for a codebase than a baseline agent without one? +"Better" = ideas that overlap with real GitHub feature requests the +community/maintainers actually engaged with **after** the shadow's +build-time snapshot. + +### Corpus (Phase 0) + +10 SWE-Bench Verified repos pinned at January 2023 commits. 8 are +"dream-eligible" — they carry enough community-validated silver-standard +feature requests to support per-repo recall metrics. `django` (uses +Trac, not GitHub Issues) and `pallets/flask` (only ~33% of recent `main` +commits are PR-driven, violating the "green via PR" assumption) were +dropped from the candidate set entirely. + +| Repo | Jan 2023 SHA | Branch | Silver pool LB | Shipped silver | Dream-eligible | +|---|---|---|---:|---:|:---:| +| astropy/astropy | `36ba03588a` | main | 521 | 129 | ✓ | +| matplotlib/matplotlib | `d247979a1e` | main | 719 | 150 | ✓ | +| mwaskom/seaborn | `0ebc82e858` | master | 60 | 3 | ✗ (too few shipped silver) | +| psf/requests | `15585909c3` | main | 61 | 3 | ✗ (too few shipped silver) | +| pydata/xarray | `67d0ee20f6` | main | 453 | 156 | ✓ | +| pylint-dev/pylint | `c70285dfdf` | main | 333 | 64 | ✓ | +| pytest-dev/pytest | `4a46ee8bc9` | main | 341 | 81 | ✓ | +| scikit-learn/scikit-learn | `7b13a8f120` | main | 734 | 154 | ✓ | +| sphinx-doc/sphinx | `ecfd08d325` | master | 327 | 46 | ✓ | +| sympy/sympy | `9f492aa538` | master | 549 | 130 | ✓ | +| | | | | **916 total** | **8** | + +**Silver pool LB**: max of three engagement signals (linked-PR closes, +≥5 comments, ≥3 👍) over GitHub issues created in window +`2023-01-01 → 2025-12-31`. Window pinned for reproducibility and +post-shadow-build to prevent temporal leakage. + +**Shipped silver**: LLM-confirmed feature request that was actually +built (closed by a merged PR or has a linked merged PR). + +**Engagement filter** (the permissive OR): feature-request AND +(merged-PR-closed OR ≥5 comments OR ≥3 👍). + +### Pipeline (6 phases) + +| # | Phase | What we did | +|---|---|---| +| 0 | Repo selection | 10 repos locked at Jan 2023 SHAs; 8 dream-eligible | +| 1 | Docker construction | per-repo hermetic images named `shadowfrog-feat/<short>:jan2023`. 10 images, ~14 GB total. | +| 1.5 | Full-test green-up | every image runs its full pytest suite (with documented `--deselect` flags for env-drift issues unrelated to the repo). Result was a green baseline of 90,816 passing tests across the 10 images so dreams don't waste time chasing pre-existing failures. | +| 2 | Issue collection | 13,151 issues across 10 repos pulled via GitHub API for the window; then LLM-classified into feature-request vs. other; filtered to silver (1,603) and shipped silver (916 in the raw pool; 910 made it into per-silver alignment scoring in Phase 5) | +| 3 | Shadow KB build | 957 dream experiments (120 per repo, 117 for xarray) on the 8 eligible repos, yielding 7,731 discoveries across 8,412 shadow files. Per-file coverage 1–20% (intentional — dreams optimize depth, not breadth). | +| 4 | Idea generation | 48 idea-generation runs × 50 ideas = 2,400 ideas (8 repos × 2 arms × 3 seeds). Plus a **third "Human" arm** for Phase 6: each repo's silver-shipped issues are rewritten into the same agent-style schema (title / description / subsystem / files_referenced), yielding **910 Human ideas** — used only in Phase 6 (Lens 2), not in Phase 5 (Lens 1) | +| 5 | Alignment scoring (Lens 1) | a batched silver-anchor judge scores every (idea × shipped-silver) pair on the 1–5 rubric below — agent arms only | +| 6 | Intrinsic eval (Lens 2) | a separate multi-judge per-idea quality eval, arm-blind, across the 3-arm pool (1,200 BL + 1,200 SF + 910 Human = **3,310 ideas**). Re-scored by a **3-judge ensemble** (Opus 4.6 + Opus 4.7 + GPT-5.5) → **9,930 verdicts** | + +### Phase 1.5 — Full-test green-up + +Every image's full pytest suite was run with JUnit XML parsed for +structured counts. **0 failures, 0 errors** across all 10 images after +applying documented `--deselect` flags for env-drift failures unrelated +to repo behavior. This gives us confidence that dream sessions running +inside these containers won't waste time chasing pre-existing failures +— the agent always sees a "green tree" baseline. Total: **90,816 +passing tests** across the 10 images. + +### Phase 3 — Dreams + +Run on a remote machine (the same one for Phase 4 later). A setup +script extracts the Jan-2023 source from each per-repo docker image +into a per-repo "host dir", installs ShadowFrog skills + hooks, sets +up a bare-clone `origin` so dream branches stay local, and writes the +docker image name into a sentinel file so the start-docker wrapper +knows which image to bind-mount. For each repo, in its host dir, the +knows which image to bind-mount. For each repo, in its host dir, the +operator sources a per-repo wrapper script (starts the +container, writes `.docker-bin/{python,pip,pytest}` wrappers, prepends +them to PATH) and launches the agent non-interactively with +`copilot --allow-all-tools`. + +The dream-specific eval context is auto-loaded as `AGENTS.md` and +contains the prompt template: + +``` +Run 5 /shadow-frog-dream sessions one after another, focus on feature +design (forward-looking ideation: extension points, gaps, friction, +"what would users want next"). In each dream session run at least 12 +experiments. Run parallel subagents with batch size of 3, always use +opus 4.6 or stronger models. Between every two dream sessions, run a +/shadow-frog-meditate session if needed to fix lineage. After every 3 +dreams, reconcile back to main from this host dir with +`./.github/skills/shadow-frog-dream/dream-reconcile.py --push`. +``` + +Tune the 5/12 numbers per repo. Network is disabled +(`--network=none`); `python` / `pip` / `pytest` are routed into the +container via `.docker-bin/` wrappers. + +### Phase 4 — Idea generation + +A one-time, idempotent prep step creates **two sibling host dirs per +repo** — one per arm — so each `(repo, arm, seed)` run gets a clean +container with the correct content: + +- `<short>-shadow/` — rsync of the dream host dir from Phase 3, plus + ShadowFrog skills + hooks installed (so the agent knows how to query + `.shadow/`). +- `<short>-baseline/` — fresh extract from the original docker image, + with **no** `.shadow/`, **no** ShadowFrog skills, **no** hooks. + +**Hard isolation by image** (locked design decision). The baseline image +has *no* `/repo/.shadow/` at all — not merely an instruction not to read +it. The cleanest way is to extract `/repo` from the image, which never +had `.shadow/`. Eval infrastructure (`AGENTS.md`, `TASK_INFO.json`) is +stripped from both arms so the agent can't tell which arm it's in. + +Then for each `(repo, arm, seed)` (8 × 2 × 3 = 48): + +1. Spin up a per-run docker container `feat-ideation-<short>-<arm>-s<seed>`, + bind-mount the host dir at `/repo`. +2. Write `.docker-bin/{python,pip,pytest,bash}` wrappers. +3. Call: + + ``` + copilot -p "$PROMPT" --model claude-opus-4.6 \ + --output-format json --no-custom-instructions --allow-all-tools + ``` + + The prompt is the arm-specific ideation prompt (shadow vs. baseline, + differing only in whether the agent is told about `.shadow/` and the + `shadow_anchor` schema field) with `{repo}` and `{repo_short}` + substituted. Both arms ask for **exactly 50 distinct feature ideas** + in a strict JSON schema (id, title, description, subsystem, + files_referenced). The shadow-arm schema additionally includes + `shadow_anchor`, which the baseline schema omits. +4. Parse the final assistant message: extract the fenced JSON block, + validate (50 items, distinct ids, fields present). +5. Emit one result file per `(repo, arm, seed)`. + +Output schema example: + +```json +{ + "repo": "pydata/xarray", + "arm": "shadow", + "seed": 0, + "model": "claude-opus-4.6", + "n_ideas_requested": 50, + "n_ideas_returned": 50, + "wall_seconds": 2389.0, + "ideas": [ + {"id": 1, "title": "…", "description": "…", "subsystem": "…", + "files_referenced": [...], "shadow_anchor": "…|null"}, + ... + ] +} +``` + +### Phase 5 — Alignment scoring (the headline judge, Lens 1) + +For each `(repo, shipped-silver-issue)` we score every candidate idea +across all (arm, seed) runs against that one silver issue using a +**silver-anchor alignment judge**. For every candidate the judge sees +only `{title, description, subsystem}`; `shadow_anchor` and +`files_referenced` are **stripped** before scoring (strict arm-blind). +Only the two agent arms (BL + SF) are scored in Lens 1 — the Human arm +is reserved for Phase 6. + +``` +Score per candidate: integer 1–5 + 5 = exact match (duplicate request) + 4 = strong overlap (same need, slightly different scope) + 3 = adjacent / related (same module, related need) + 2 = weak / tangential (same broad subsystem only) + 1 = unrelated + — = no Phase 4 idea matched this dream (sentinel for "no judge score") +``` + +Of the 916 raw shipped silvers, **910** entered alignment scoring. +The 6-silver gap is the seaborn (3) and requests (3) silvers, both +of which were excluded from the dream-eligible set ahead of time +(neither repo received its own dream campaign in the final design). +Batched: every (silver, idea) pair is scored, **5,460 batched judge +calls** total (910 silvers × 6 arm-seed combos), ~16,400 premium +requests, ~6h45m wall on a laptop with `--jobs 8`, yielding +**~272,500 individual alignment scores**. +**The dashboard's Lens 1 sunburst is built from these scores**, colored +by judge rubric level; the "—" sentinel applies to dreams that never +surfaced as an idea in Phase 4 (so the judge never had a chance to +score them). + +### Phase 6 — Intrinsic eval (Lens 2) + +A separate per-idea quality scoring pass on the full 3-arm pool of +**3,310 ideas (1,200 BL + 1,200 SF + 910 Human)** using an +**intrinsic-quality judge**. The judge is arm-blind: it sees only +`{title, description, subsystem, files_referenced}` of each idea (the +arm and repo identity are stripped before scoring). Lens 2 evaluates +ideas on their own merit rather than against any specific silver +target. + +**Rubric — five scored dimensions plus one categorical tag**: + +| Dimension | Prompt label | What it scores | Type | +|---|---|---|---| +| **Groundedness** | GROUNDEDNESS | Does the proposal demonstrate project-specific knowledge (real APIs, modules, conventions) vs. plausible-sounding generalities? | 1–5 | +| **Insight** | NON-OBVIOUSNESS | How unlikely is this idea to emerge from a 5-minute brainstorm by a regular contributor? | 1–5 | +| **User Impact** | USER VALUE | How many real users would benefit and how meaningfully? | 1–5 | +| **Spec Clarity** | ACTIONABILITY | Could a maintainer turn this into a PR scope without back-and-forth? | 1–5 | +| **Change Size** | SCOPE | Blast radius of the implied change — how many subsystems / modules / files would it touch? | 1–5 (descriptive, not evaluative) | +| **Change Locality** | Locus of Change | Categorical tag, one of: `api_surface`, `new_feature`, `internal_machinery`, `cross_cutting`, `ecosystem_bridge`, `dx_and_tooling`, `operational_quality` | enum | + +The display names (Insight, User Impact, Spec Clarity, Change Size) +are used throughout the dashboard and reports; the bracketed +prompt-label column is the raw token shown to the judge. + +The rubric is designed so the four evaluative dimensions +(Groundedness, Insight, User Impact, Spec Clarity) are mutually +orthogonal: a small local idea can be deeply grounded; a sweeping idea +can be ungrounded; insight is independent of size; spec clarity is +independent of impact. The judge prompt's per-dimension anchors +explicitly instruct independent scoring per dimension. + +**3-judge ensemble for robustness**: every idea is scored by **Opus +4.6 + Opus 4.7 + GPT-5.5** → **3,310 ideas × 3 judges = 9,930 +verdicts**. The Human arm comes in two style variants: +- The original GitHub-issue title + body, lightly normalized into the + JSON schema (the "Human" arm proper). +- The same issue **rewritten** by an LLM into the agent register + (title ≤120 chars, 1–3 sentence description, subsystem, 1–5 file + refs) — the "Style Control" variant. This lets the dashboard + separate the "register effect" (humans write differently from the + agent prompt format) from the "content effect" (whether the + underlying idea is better). + +### Pipeline summary + +End-to-end the experiment is six phases: (0) pick 10 repos, lock at +Jan-2023 SHAs, identify 8 dream-eligible; (1) build per-repo Docker +images and (1.5) run their full pytest suites until green; (2) pull +13,151 issues from GitHub for the window and LLM-classify into +silver / shipped silver; (3) run 957 dream experiments on the 8 +eligible repos to build a shadow KB per repo; (4) generate 50 ideas +per (repo, arm, seed) for both SF and BL arms (48 runs, 2,400 ideas), +plus a 910-idea Human arm (raw + style-controlled rewrites) for Lens +2; (5) score every (idea × shipped-silver) pair with the alignment +judge (Lens 1, ~272K scores); (6) score every idea on the 5-dimension +intrinsic-quality rubric with a 3-judge ensemble (Lens 2, 9,930 +verdicts). + +--- + +## 5. Navigation (`navigation`) + +**Question.** Does shadow knowledge help an agent *navigate* a large +codebase (find the right symbol fast under a tool-call budget)? This +underpins all four bug/ideation experiments — if shadow doesn't help +navigation, it can't help anything downstream. + +### Corpora (pinned) + +``` +fastapi pinned_tag: 0.136.1 pinned_sha: e54e5a89 source_root: fastapi/ (~360 callable symbols) +django pinned_tag: 5.1.4 pinned_sha: 2d4add11 source_root: django/ (10,546 callable symbols) +``` + +### Conditions + +Three conditions, each lives in its own worktree with its own shadow tree: + +| External name | Internal name | Layout | +|---|---|---| +| `shadow-frog` | `mirror-nested` | per-file shadow mirroring the source tree (canonical ShadowFrog) | +| `flat-shadow` | `symbol-keyed-flat-giant` | one big `.shadow/SHADOW.md` with `## file::symbol` (ablation: same content, no spatial structure) | +| `no-shadow` | `no-KB` | empty: agent runs on the raw source tree with no shadow installed | + +Internal names are baked into 60K+ data records and worktree paths; +display names appear in the dashboard only. The `flat-shadow` condition +ships a single concatenated `.shadow/SHADOW.md` (built at the +build-shadows stage) instead of the per-file mirror, paired with +prompt-level guidance that points the agent at the flat file rather +than per-file shadow paths. + +### Axes (full sweep) + +| Axis | Values | +|---|---| +| Corpus | `django`, `fastapi` | +| Condition | `shadow-frog`, `flat-shadow`, `no-shadow` | +| Scale (needles per cell) | small=50, medium=400, large=2,500, xlarge=10,000 (django only), xxlarge=50,000 configured / ~35,000 delivered (django only) | +| Wrong-needle rate | 0.00, 0.15, 0.30, 0.50 | +| Tool-call budget | 1, 2, 4, 8, 16, 32, `inf` | +| Task type | `path_known` (60 base + cross-scale extension: 23 for django → 83 total; 22 for fastapi → 82 total), `path_unknown` (same shape — 83 django / 82 fastapi) | + +xlarge/xxlarge are django-only because fastapi's source has only ~360 +callable symbols — it cannot produce 10K+ needles. xxlarge targets +50,000 needles in the config but actually delivers ~35,000 (capped by +the seedable-symbol pool × wrong-rate fan-out). Total **69,342 agent +runs judged** across the full matrix. + +### Hypotheses (pre-registered, locked before any runs) + +| ID | Statement | Threshold | +|---|---|---| +| **H8 — Budget collapse** | At `large × path_unknown × wrong=0`, dropping budget from `inf` to `1` reduces recall by ≥30 pp on both corpora | `recall(inf) − recall(1) ≥ 30 pp` per corpus | +| **H10 — Corpus generality** | At `large × path_unknown × wrong=0 × budget=inf`, `shadow-frog` recall exceeds `no-shadow` recall by ≥30 pp on both corpora | `recall(shadow-frog) − recall(no-shadow) ≥ 30 pp` per corpus¹ | +| **H11 — Multi-MB shadow scaling** | At `xxlarge × path_unknown × wrong=0 × budget=inf` (shadow ~6 MB), `shadow-frog` recall exceeds `flat-shadow` recall by ≥10 pp on django | `recall(shadow-frog) − recall(flat-shadow) ≥ 10 pp`¹ | + +¹ Thresholds shown are the pre-registered values. The +operationalized scorer applies a common +20 pp cutoff for both H10 and H11 when stamping PASS/FAIL into +the hypothesis-verdicts output; reproducers running the scorer should +expect to see the 20 pp formulation in the output. + +xlarge serves as a secondary observation (no hard threshold). fastapi is +excluded from H11 because its corpus is too small. Pass/fail is +determined automatically by the aggregator. + +**Wrong-needle probe** (methodology, not a hypothesis). A small fraction +of needles is replaced by plausible-but-wrong twins (`wrong_rate` ∈ {0.15, +0.30, 0.50}). The agent doesn't know which; the judge does. If the +agent reports a wrong-but-shadow-planted fact, that's positive evidence +the agent is **actually reading the shadow** rather than recalling +training data of these open-source codebases. Without this probe, high +recall could in principle reflect memorized knowledge of +django/fastapi internals. + +### Pipeline (16 stages, idempotent, strict dataflow order) + +Stages 01–08 build the **needles** (the questions and the +shadow-content the agent should be able to find them through). Stages +09–10 build the **shadows and worktrees** (one cell per +condition × scale × wrong-rate combination). Stages 11–13 run, judge, +and aggregate. Stages 14–16 produce the dashboard. + +| # | Stage | Reads | Writes | +|---|---|---|---| +| 01 | Clone corpus | corpora config | corpus clone, records the pinned SHA | +| 02 | Enumerate symbols | corpus source tree | per-corpus symbol index (jsonl) | +| 03 | Subsample symbols | symbols | seeded subsample (stratified for large corpora) | +| 04 | Author needles | symbols + author prompt | raw needles jsonl | +| 05 | Finalize needles | raw needles + scales | final needles jsonl (tagged per scale, strict-superset invariant) | +| 06 | Perturb needles | final needles + perturbation prompt | wrong-twin needles jsonl (plausible-but-false) | +| 07 | Generate base tasks | needles + path-known / path-unknown templates + needle-to-question prompt | `path_known` (pk-001..060) and `path_unknown` (pu-001..060) task jsonl | +| 08 | Cross-scale tasks | same inputs as 07 | extended task ids (pk-101+, pu-101+) | +| 09 | Build shadows | final + wrong needles + config | per-cell `.shadow/...` tree + manifest, hashed by `build_seed` | +| 10 | Setup worktrees | corpus + shadows + `install.sh` | one worktree per (corpus × condition × scale × wrong_rate) cell | +| 11 | Run agent | tasks + worktrees | per-run result jsonl + a SQLite run manifest | +| 12 | Judge recall | results + recall-judge prompt + shadow manifests | per-cell judgment json | +| 13 | Aggregate | judgments + run manifest | summary CSVs, headline metrics, hypothesis verdicts | +| 14 | Plot interaction | summaries | diagnostic PNGs | +| 15 | Render dashboard | metrics + verdicts | dashboard fragment HTML | +| 16 | Splice dashboard | dashboard fragment | the final `eval/results_dashboard.html` (the fragment is spliced into the §1 section between marker comments) | + +Stages 09 → 16 are orchestrated as a single pipeline runner with a +`K`-way parallelism knob (default 16). + +### Cell schema + +``` +<corpus>/<condition>/<scale>/wrong_<rate>/budget_<budget>/<task_type>/<task_id> +``` + +Budget is **not** a worktree key — worktrees are keyed on the first +four dims only; budget is enforced at agent invocation via prompt +augmentation plus harness-side hard interrupt. + +### Orchestrator config (verbatim) + +```yaml +agent: + model: claude-opus-4.6 + max_turns: 50 + reasoning_effort: medium + +judge: + model: claude-opus-4.6 + ensemble: false # single-judge; per-claim verdicts add internal richness + emit_per_claim: true # judge must classify each agent claim individually + +needle_author: + model: claude-opus-4.6 + over_generation_target: 3000 # author over-generates, finalizer trims + +needle_perturber: + model: claude-opus-4.6 + plausibility_max_token_delta: 0.50 + plausibility_require_in_file_identifiers: true + +needle_to_question: + model: claude-opus-4.6 + +qa: + sample_fraction: 0.05 + random_seed: 1729 + perturbation_qa_count_per_corpus: 50 + +orchestrator: + parallelism: 8 + max_attempts_per_task: 3 + per_task_backoff_seconds: [10, 60, 300] + rate_limit_window_seconds: 60 + rate_limit_trip_threshold: 3 + pool_pause_initial_seconds: 300 + pool_pause_max_seconds: 3600 + heartbeat_interval_seconds: 60 + per_task_timeout_seconds: 1500 + copilot_extra_args: + - "--allow-all" + - "--output-format" + - "json" + - "--no-remote" + - "--no-custom-instructions" + rate_limit_stderr_patterns: + - "429" + - "rate limit" + - "rate_limit" + - "quota_exceeded" + - "RESOURCE_EXHAUSTED" + - "too many requests" +``` + +Pre-registered constants: `random_seed: 1729` plus a `build_seed` +("shadowfrog-v2") baked into the shadow-tree hash at the perturb-needles +and build-shadows stages — changing it invalidates the entire 60K-cell +dataset. `build_seed` is hardcoded in those stages rather than surfaced +in the orchestrator config, by design +(it should not be edited casually). + +### Pipeline summary + +The pipeline runs from a single config file (the YAML above) plus a +set of prompts (authoring, needle-to-question, perturbation, recall +judge, QA, and signature rewrite). Each corpus passes through: +clone → enumerate symbols → seeded subsample → author needles → +finalize (per-scale tagging, strict-superset invariant) → perturb +into plausible-but-false wrong-twin needles → generate `path_known` +and `path_unknown` task variants → build per-cell shadow KBs and +worktrees → run the agent under each cell → judge recall claim-by-claim → +aggregate per-cell summaries → render the dashboard fragment and splice +it into `eval/results_dashboard.html`. The full 60K-cell sweep is +reproducible from the YAML, the corpus SHAs, and the pre-registered +seeds. + +--- + +## What this folder contains + +| File | What it is | +|---|---| +| `results_dashboard.html` | The single self-contained results dashboard rolling up every experiment (Lens 1 sunburst inlined via `<iframe srcdoc>`). | +| `README.md` (this document) | Methodology-only summary. For every experiment: corpus, environment, pipeline description, prompts (with verbatim wording for the key ones), scoring rubric, design decisions. | + +The raw outputs of the evaluation — per-task analysis, patches, +dreams, judge verdicts, logs, manifests, shadow KBs, executable +scripts, Dockerfiles, per-arm operator runbooks, and the curated +reports — are kept outside this repo in a private archive. diff --git a/eval/results_dashboard.html b/eval/results_dashboard.html new file mode 100644 index 0000000..ceb9c57 --- /dev/null +++ b/eval/results_dashboard.html @@ -0,0 +1,4321 @@ +<!DOCTYPE html> +<html lang="en"> +<head> +<meta charset="UTF-8"> +<meta name="viewport" content="width=device-width, initial-scale=1.0"> +<title>ShadowFrog — Evaluation Dashboard + + + + + +
        + +

        🐸 ShadowFrog — Evaluation Dashboard

        +

        Shadow read-path, bug hunting, bug fixing, and feature ideation — evaluated on real-world and synthetic benchmarks

        + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
        + + + + diff --git a/examples/coupon-demo/.shadow/.shadowignore b/examples/coupon-demo/.shadow/.shadowignore new file mode 100644 index 0000000..a920eb5 --- /dev/null +++ b/examples/coupon-demo/.shadow/.shadowignore @@ -0,0 +1,35 @@ +# Directories +node_modules/ +vendor/ +venv/ +.venv/ +__pycache__/ +dist/ +build/ +target/ +out/ + +# Generated / minified +*.min.js +*.min.css +*.map +*.lock + +# Binary +*.png +*.jpg +*.gif +*.ico +*.woff +*.woff2 +*.ttf +*.eot +*.pdf +*.zip +*.tar.gz + +# The shadow itself +.shadow/ + +# Project-specific +README.md diff --git a/examples/coupon-demo/.shadow/_cross/coupon-case-normalization-mismatch.md b/examples/coupon-demo/.shadow/_cross/coupon-case-normalization-mismatch.md new file mode 100644 index 0000000..4538e31 --- /dev/null +++ b/examples/coupon-demo/.shadow/_cross/coupon-case-normalization-mismatch.md @@ -0,0 +1,11 @@ +# Coupon case normalization mismatch + +**Category**: edge-case +**Refs**: +- `inventory.py::validate_coupon` +- `cart.py::calculate_total` +- `cart.py::load_coupon` + +**Discovery**: validate_coupon normalizes coupon codes to uppercase via code.upper() before lookup, but calculate_total passes coupon_code directly to get_coupon without normalization. A user who validates "save20" (returns True) and then passes "save20" to calculate_total gets no discount — the coupon silently fails because load_coupon's keys are uppercase. This creates a validate-then-use inconsistency where validated codes don't work. + +_(verified, source: exploration, labels: [bug])_ diff --git a/examples/coupon-demo/.shadow/_cross/global-coupon-cache-side-effects.md b/examples/coupon-demo/.shadow/_cross/global-coupon-cache-side-effects.md new file mode 100644 index 0000000..3ee95a3 --- /dev/null +++ b/examples/coupon-demo/.shadow/_cross/global-coupon-cache-side-effects.md @@ -0,0 +1,12 @@ +# Global coupon cache side effects + +**Category**: behavior +**Refs**: +- `cart.py::COUPON_CACHE` +- `cart.py::get_coupon` +- `inventory.py::validate_coupon` +- `test_cart.py::test_coupon` + +**Discovery**: COUPON_CACHE is a module-level global dict shared by all importers. Any call to get_coupon (directly or via validate_coupon) permanently populates the cache, including caching None for invalid codes. In tests, cache entries from one test persist into the next — there is no reset mechanism. validate_coupon caches under the uppercased key, while calculate_total would cache under the original-case key, so a single logical coupon code can produce two separate cache entries ("SAVE20" and "save20") with different values. + +_(verified, source: exploration, labels: [bug])_ diff --git a/examples/coupon-demo/.shadow/_cross/mutation-through-discount-pipeline.md b/examples/coupon-demo/.shadow/_cross/mutation-through-discount-pipeline.md new file mode 100644 index 0000000..3dc557d --- /dev/null +++ b/examples/coupon-demo/.shadow/_cross/mutation-through-discount-pipeline.md @@ -0,0 +1,11 @@ +# Mutation through discount pipeline + +**Category**: edge-case +**Refs**: +- `inventory.py::apply_bulk_discount` +- `cart.py::calculate_total` +- `test_cart.py::test_bulk_then_coupon` + +**Discovery**: apply_bulk_discount mutates item dicts in-place (modifying "price" keys) and returns the same list object. When piped into calculate_total, the mutation is invisible — calculate_total sees already-reduced prices. But any code holding a reference to the original items list now sees the discounted prices permanently. Calling apply_bulk_discount multiple times compounds discounts (0.90^N multiplier). The test_bulk_then_coupon test avoids this by creating fresh items, but real usage with shared item references would silently corrupt prices. + +_(verified, source: exploration, labels: [bug])_ diff --git a/examples/coupon-demo/.shadow/_dreams/20260420-140000Z-cache-poison-sequence/manifest.json b/examples/coupon-demo/.shadow/_dreams/20260420-140000Z-cache-poison-sequence/manifest.json new file mode 100644 index 0000000..6ed27f3 --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/20260420-140000Z-cache-poison-sequence/manifest.json @@ -0,0 +1,21 @@ +{ + "dream_id": "20260420-140000Z-cache-poison-sequence", + "branch": "dream/coupon-demo/20260420-140000Z-cache-poison-sequence", + "parent_branch": "main", + "category": "bug hunting", + "verdict": "useful", + "title": "Cache poisoning via validate-then-calculate sequence", + "discoveries": [ + { + "op": "add", + "anchor": "cart.py::COUPON_CACHE", + "text": "Case-variant lookups create duplicate cache entries for the same logical coupon. validate_coupon(\"save20\") caches \"SAVE20\" \u2192 valid, then calculate_total(\"save20\") caches \"save20\" \u2192 None. Cache grows 2\u00d7 faster with mixed-case usage.", + "status": "verified", + "source": "exploration", + "labels": ["bug", "performance"], + "also_involves": ["cart.py::get_coupon", "inventory.py::validate_coupon"], + "dream_report": "_dreams/20260420-140000Z-cache-poison-sequence/" + } + ], + "cross_cutting": [] +} diff --git a/examples/coupon-demo/.shadow/_dreams/20260420-140000Z-cache-poison-sequence/patch.diff b/examples/coupon-demo/.shadow/_dreams/20260420-140000Z-cache-poison-sequence/patch.diff new file mode 100644 index 0000000..95f8a72 --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/20260420-140000Z-cache-poison-sequence/patch.diff @@ -0,0 +1,53 @@ +diff --git a/dream_exp1_cache_poison.py b/dream_exp1_cache_poison.py +new file mode 100644 +--- /dev/null ++++ b/dream_exp1_cache_poison.py +@@ -0,0 +1,52 @@ ++"""Dream experiment: prove validate-then-calculate poisons COUPON_CACHE. ++ ++Runs three scenarios that reset COUPON_CACHE between runs and check the ++cache state after each call sequence. Demonstrates that the case- ++normalization mismatch between validate_coupon and calculate_total ++creates duplicate cache entries for the same logical coupon code. ++""" ++ ++import cart ++from cart import COUPON_CACHE, calculate_total ++from inventory import validate_coupon ++ ++ITEMS = [{"price": 25.00, "qty": 3}] ++ ++ ++def reset(): ++ COUPON_CACHE.clear() ++ ++ ++def scenario_1_validate_then_calculate(): ++ reset() ++ assert validate_coupon("save20") is True ++ total = calculate_total(ITEMS, coupon_code="save20") ++ assert set(COUPON_CACHE.keys()) == {"SAVE20", "save20"} ++ assert COUPON_CACHE["save20"] is None ++ assert total == 81.00, f"expected 81.00 (no discount), got {total}" ++ ++ ++def scenario_2_calculate_then_validate(): ++ reset() ++ total = calculate_total(ITEMS, coupon_code="save20") ++ assert validate_coupon("save20") is True ++ assert set(COUPON_CACHE.keys()) == {"SAVE20", "save20"} ++ assert total == 81.00 ++ ++ ++def scenario_3_uppercase_only(): ++ reset() ++ total = calculate_total(ITEMS, coupon_code="SAVE20") ++ assert set(COUPON_CACHE.keys()) == {"SAVE20"} ++ assert total == 64.80, f"expected 64.80 (discount applied), got {total}" ++ ++ ++if __name__ == "__main__": ++ scenario_1_validate_then_calculate() ++ scenario_2_calculate_then_validate() ++ scenario_3_uppercase_only() ++ print("All 3 scenarios passed.") diff --git a/examples/coupon-demo/.shadow/_dreams/20260420-140000Z-cache-poison-sequence/report.md b/examples/coupon-demo/.shadow/_dreams/20260420-140000Z-cache-poison-sequence/report.md new file mode 100644 index 0000000..8ad977a --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/20260420-140000Z-cache-poison-sequence/report.md @@ -0,0 +1,67 @@ +--- +dream_id: "20260420-140000Z-cache-poison-sequence" +category: bug hunting +verdict: useful +base_commit: 69de9194d8d30d980a41fa0e1bace8fc76d82e24 +branch: "dream/coupon-demo/20260420-140000Z-cache-poison-sequence" +parent_branch: "main" +remote: "origin" +related_symbols: + - "cart.py::COUPON_CACHE" + - "cart.py::get_coupon" + - "cart.py::calculate_total" + - "inventory.py::validate_coupon" +builds_on: [] +--- + +# Cache Poisoning via Validate-Then-Calculate Sequence + +## Motivation + +The shadow documents a case-normalization mismatch between `validate_coupon` (uppercases) and `calculate_total` (does not). This experiment quantifies the exact runtime behavior: what gets cached, in what order, and how the cache ends up with contradictory entries for the same logical coupon code. + +## Hypothesis + +Calling `validate_coupon("save20")` then `calculate_total(items, coupon_code="save20")` will produce TWO separate cache entries — `"SAVE20"` → valid coupon and `"save20"` → None — and the user gets no discount despite a successful validation. + +## Implementation + +Wrote 3 scenarios testing: +1. validate → calculate (lowercase) +2. calculate → validate (reverse order) +3. Both uppercase (control) + +Each scenario resets `COUPON_CACHE` between runs and checks cache state, return values, and final totals. + +## Commands Run + +``` +$ python dream_exp1_cache_poison.py +Exit code: 0 +All 3 scenarios passed. +``` + +Key output: +- Scenario 1: Cache has 2 entries after validate+calculate. `SAVE20` → valid, `save20` → None. Total = $81.00 (no discount). +- Scenario 2: Same result in reverse order. Both paths create independent cache entries. +- Scenario 3: Uppercase only → 1 cache entry, total = $64.80 (discount applied correctly). + +## Evaluation + +All assertions passed. The cache poisoning is deterministic and order-independent: +- `validate_coupon` always caches under the uppercased key +- `calculate_total` always caches under the original-case key +- These are separate cache slots, so one doesn't prevent the other +- A user who validates "save20" (True) and passes it to calculate_total gets zero discount + +The dual-entry cache pollution also means the cache grows 2× faster for case-variant lookups. + +## Takeaways + +- The root cause is `calculate_total` not calling `.upper()` on `coupon_code`. A one-line fix (`coupon_code = coupon_code.upper()` at the top of `calculate_total`) would resolve both the normalization mismatch and the cache pollution. +- The cache poisoning is invisible — no error, no warning. The discount silently vanishes. This is the worst kind of bug: it looks like it works. +- Order of operations doesn't matter — both paths poison independently. + +## Verdict Details + +Useful: Confirmed the exact cache state across validate-first and calculate-first orderings, proving the bug is deterministic and silent. The shadow already documented this bug, but this experiment adds concrete proof with cache snapshots. diff --git a/examples/coupon-demo/.shadow/_dreams/20260420-141000Z-bulk-min-total-interaction/manifest.json b/examples/coupon-demo/.shadow/_dreams/20260420-141000Z-bulk-min-total-interaction/manifest.json new file mode 100644 index 0000000..10ac9ad --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/20260420-141000Z-bulk-min-total-interaction/manifest.json @@ -0,0 +1,20 @@ +{ + "dream_id": "20260420-141000Z-bulk-min-total-interaction", + "branch": "dream/coupon-demo/20260420-141000Z-bulk-min-total-interaction", + "parent_branch": "main", + "category": "investigation", + "verdict": "useful", + "title": "Bulk discount vs coupon min_total interaction", + "discoveries": [ + { + "op": "add", + "anchor": "cart.py::calculate_total", + "text": "min_total check evaluates against the post-bulk-discount subtotal because apply_bulk_discount mutates item prices in place before calculate_total runs. If bulk discount drops the subtotal below min_total, the coupon is correctly rejected \u2014 but only by accident of the in-place mutation contract.", + "status": "verified", + "source": "exploration", + "also_involves": ["inventory.py::apply_bulk_discount"], + "dream_report": "_dreams/20260420-141000Z-bulk-min-total-interaction/" + } + ], + "cross_cutting": [] +} diff --git a/examples/coupon-demo/.shadow/_dreams/20260420-141000Z-bulk-min-total-interaction/patch.diff b/examples/coupon-demo/.shadow/_dreams/20260420-141000Z-bulk-min-total-interaction/patch.diff new file mode 100644 index 0000000..eda0086 --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/20260420-141000Z-bulk-min-total-interaction/patch.diff @@ -0,0 +1,50 @@ +diff --git a/dream_exp2_bulk_min_total.py b/dream_exp2_bulk_min_total.py +new file mode 100644 +--- /dev/null ++++ b/dream_exp2_bulk_min_total.py +@@ -0,0 +1,48 @@ ++"""Dream experiment: clarify how bulk discount interacts with coupon min_total. ++ ++Resolves an uncertain shadow discovery by checking whether ++calculate_total's min_total check sees pre- or post-bulk-discount ++subtotals. Two scenarios cover HALF (min_total=100, drops below) and ++SAVE20 (min_total=50, boundary case). ++""" ++ ++import cart ++from cart import COUPON_CACHE, calculate_total ++from inventory import apply_bulk_discount ++ ++ ++def reset(): ++ COUPON_CACHE.clear() ++ ++ ++def scenario_drops_below_min_total(): ++ reset() ++ items = [{"price": 21.00, "qty": 5}] ++ apply_bulk_discount(items) ++ assert items[0]["price"] == 18.90 ++ with_coupon = calculate_total(items, coupon_code="HALF") ++ without_coupon = calculate_total(items) ++ assert with_coupon == without_coupon, ( ++ f"coupon should NOT apply (subtotal $94.50 < min_total $100), " ++ f"but got {with_coupon} vs {without_coupon}" ++ ) ++ ++ ++def scenario_just_above_min_total(): ++ reset() ++ items = [{"price": 11.12, "qty": 5}] ++ apply_bulk_discount(items) ++ subtotal = items[0]["price"] * items[0]["qty"] ++ assert subtotal > 50.0 ++ with_coupon = calculate_total(items, coupon_code="SAVE20") ++ without_coupon = calculate_total(items) ++ assert with_coupon < without_coupon, "coupon should apply at boundary" ++ ++ ++if __name__ == "__main__": ++ scenario_drops_below_min_total() ++ scenario_just_above_min_total() ++ print("All scenarios passed.") diff --git a/examples/coupon-demo/.shadow/_dreams/20260420-141000Z-bulk-min-total-interaction/report.md b/examples/coupon-demo/.shadow/_dreams/20260420-141000Z-bulk-min-total-interaction/report.md new file mode 100644 index 0000000..5021ede --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/20260420-141000Z-bulk-min-total-interaction/report.md @@ -0,0 +1,64 @@ +--- +dream_id: "20260420-141000Z-bulk-min-total-interaction" +category: investigation +verdict: useful +base_commit: 69de9194d8d30d980a41fa0e1bace8fc76d82e24 +branch: "dream/coupon-demo/20260420-141000Z-bulk-min-total-interaction" +parent_branch: "main" +remote: "origin" +related_symbols: + - "inventory.py::apply_bulk_discount" + - "cart.py::calculate_total" +builds_on: [] +--- + +# Bulk Discount vs Coupon min_total Interaction + +## Motivation + +The shadow contains an **uncertain** discovery on `cart.py::calculate_total`: +> "Coupon min_total check uses the pre-discount subtotal. If bulk discount reduces items below min_total, the coupon still applies because the check sees the already-reduced subtotal from apply_bulk_discount, not the original price." + +This is contradictory — it says "pre-discount subtotal" but then says it "sees the already-reduced subtotal". The experiment resolves which interpretation is correct. + +## Hypothesis + +`calculate_total` computes subtotal from `items` as-is. Since `apply_bulk_discount` mutates item prices in-place BEFORE `calculate_total` runs, the subtotal seen by `calculate_total` IS the post-bulk-discount value. Therefore, if bulk discount drops the subtotal below `min_total`, the coupon will NOT apply. + +## Implementation + +2 scenarios testing the HALF coupon (min_total=100) and SAVE20 (min_total=50): +1. **Key test**: Bulk discount drops subtotal from $105 to $94.50 (below min_total=100, HALF coupon) +2. Boundary test: SAVE20 subtotal just barely above min_total ($50.05) after bulk discount + +## Commands Run + +``` +$ python dream_exp2_bulk_min_total.py +Exit code: 0 +All scenarios passed. +``` + +Key findings: +- Scenario 1: Original=$105, post-bulk=$94.50. Total WITH coupon = total WITHOUT coupon. **Coupon did NOT apply.** +- Scenario 2: Post-bulk=$50.05 (just above $50). Coupon applied correctly. + +## Evaluation + +The uncertain discovery is **incorrect in its conclusion**. The truth is: +- `calculate_total` sees whatever prices are in the `items` dicts at call time +- Since `apply_bulk_discount` mutates prices in-place, `calculate_total` sees post-bulk prices +- The min_total check correctly evaluates against the actual (post-mutation) subtotal +- If bulk discount drops subtotal below min_total, the coupon is correctly rejected + +The boundary test (scenario 2) confirms the `>=` comparison works precisely. + +## Takeaways + +- The uncertain discovery should be **refuted** and replaced with the correct behavior +- The mutation-based pipeline actually produces correct min_total behavior — but only by accident. If someone refactored `apply_bulk_discount` to return new items instead of mutating, the behavior would change. +- The real risk is that the correctness depends on an implicit contract: "apply_bulk_discount must be called before calculate_total, and must mutate in place" + +## Verdict Details + +Useful: Resolved an uncertain discovery to refuted, and identified that the correct behavior depends on an implicit mutation contract rather than explicit design. diff --git a/examples/coupon-demo/.shadow/_dreams/20260420-142000Z-adversarial-inputs/manifest.json b/examples/coupon-demo/.shadow/_dreams/20260420-142000Z-adversarial-inputs/manifest.json new file mode 100644 index 0000000..6cc8b44 --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/20260420-142000Z-adversarial-inputs/manifest.json @@ -0,0 +1,49 @@ +{ + "dream_id": "20260420-142000Z-adversarial-inputs", + "branch": "dream/coupon-demo/20260420-142000Z-adversarial-inputs", + "parent_branch": "main", + "category": "security audit", + "verdict": "useful", + "title": "Adversarial input audit across all public functions", + "discoveries": [ + { + "op": "add", + "anchor": "inventory.py::validate_coupon", + "text": "Also crashes on any non-string type: int, list, dict all raise AttributeError on .upper(). Needs isinstance(code, str) guard.", + "status": "verified", + "source": "exploration", + "labels": ["security"], + "dream_report": "_dreams/20260420-142000Z-adversarial-inputs/" + }, + { + "op": "add", + "anchor": "inventory.py::apply_bulk_discount", + "text": "Crashes with KeyError if items lack 'qty' key, and TypeError if 'qty' is a string. No input validation.", + "status": "verified", + "source": "exploration", + "labels": ["security"], + "dream_report": "_dreams/20260420-142000Z-adversarial-inputs/" + }, + { + "op": "add", + "anchor": "cart.py::get_coupon", + "text": "Accepts any hashable type as key (True, 42, lists-as-errors). Non-string keys permanently cache None entries that can never resolve to valid coupons \u2014 silent cache pollution.", + "status": "verified", + "source": "exploration", + "labels": ["security"], + "also_involves": ["cart.py::COUPON_CACHE"], + "dream_report": "_dreams/20260420-142000Z-adversarial-inputs/" + }, + { + "op": "add", + "anchor": "cart.py::calculate_total", + "text": "Non-string coupon_code values (True, 42, etc.) pass the `if coupon_code:` truthiness check, reach get_coupon, cache None under the non-string key, and silently produce no discount. No type validation exists.", + "status": "verified", + "source": "exploration", + "labels": ["security"], + "also_involves": ["cart.py::get_coupon", "cart.py::COUPON_CACHE"], + "dream_report": "_dreams/20260420-142000Z-adversarial-inputs/" + } + ], + "cross_cutting": [] +} diff --git a/examples/coupon-demo/.shadow/_dreams/20260420-142000Z-adversarial-inputs/patch.diff b/examples/coupon-demo/.shadow/_dreams/20260420-142000Z-adversarial-inputs/patch.diff new file mode 100644 index 0000000..fd23a76 --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/20260420-142000Z-adversarial-inputs/patch.diff @@ -0,0 +1,75 @@ +diff --git a/dream_exp3_adversarial.py b/dream_exp3_adversarial.py +new file mode 100644 +--- /dev/null ++++ b/dream_exp3_adversarial.py +@@ -0,0 +1,70 @@ ++"""Dream experiment: adversarial input audit across all public functions. ++ ++Systematically tests non-string coupons, missing keys, wrong types, ++extreme values, and non-string cache pollution. Records which inputs ++crash vs are silently accepted. ++""" ++ ++import cart ++from cart import COUPON_CACHE, calculate_total, get_coupon ++from inventory import apply_bulk_discount, validate_coupon ++ ++ ++def reset(): ++ COUPON_CACHE.clear() ++ ++ ++def expect_crash(label, fn): ++ try: ++ fn() ++ except (AttributeError, KeyError, TypeError) as exc: ++ print(f" CRASH ({type(exc).__name__}): {label}") ++ return True ++ print(f" NO CRASH (unexpected): {label}") ++ return False ++ ++ ++def expect_silent(label, fn): ++ try: ++ result = fn() ++ print(f" SILENT ({result!r}): {label}") ++ return True ++ except Exception as exc: ++ print(f" CRASH (unexpected): {label} -> {exc}") ++ return False ++ ++ ++def main(): ++ reset() ++ ++ expect_crash("validate_coupon(None)", lambda: validate_coupon(None)) ++ expect_crash("validate_coupon(123)", lambda: validate_coupon(123)) ++ expect_crash("validate_coupon(['X'])", lambda: validate_coupon(["X"])) ++ ++ expect_crash( ++ "calculate_total missing 'price'", ++ lambda: calculate_total([{"qty": 1}]), ++ ) ++ expect_crash( ++ "apply_bulk_discount missing 'qty'", ++ lambda: apply_bulk_discount([{"price": 10.0}]), ++ ) ++ ++ expect_silent( ++ "calculate_total(price=-50, qty=1)", ++ lambda: calculate_total([{"price": -50.0, "qty": 1}]), ++ ) ++ expect_silent( ++ "calculate_total(price=50, qty=-1)", ++ lambda: calculate_total([{"price": 50.0, "qty": -1}]), ++ ) ++ ++ reset() ++ get_coupon(True) ++ get_coupon(42) ++ assert True in COUPON_CACHE and 42 in COUPON_CACHE ++ print(f" cache pollution: {dict(COUPON_CACHE)!r}") ++ ++ ++if __name__ == "__main__": ++ main() diff --git a/examples/coupon-demo/.shadow/_dreams/20260420-142000Z-adversarial-inputs/report.md b/examples/coupon-demo/.shadow/_dreams/20260420-142000Z-adversarial-inputs/report.md new file mode 100644 index 0000000..006a68d --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/20260420-142000Z-adversarial-inputs/report.md @@ -0,0 +1,79 @@ +--- +dream_id: "20260420-142000Z-adversarial-inputs" +category: security audit +verdict: useful +base_commit: 69de9194d8d30d980a41fa0e1bace8fc76d82e24 +branch: "dream/coupon-demo/20260420-142000Z-adversarial-inputs" +parent_branch: "main" +remote: "origin" +related_symbols: + - "cart.py::calculate_total" + - "inventory.py::validate_coupon" + - "inventory.py::apply_bulk_discount" + - "cart.py::get_coupon" +builds_on: [] +--- + +# Adversarial Input Audit + +## Motivation + +No function in the codebase validates input types or required keys. This experiment systematically tests what happens with adversarial, malformed, and edge-case inputs across all public functions. + +## Hypothesis + +Functions will crash with unhandled exceptions (AttributeError, KeyError, TypeError) on non-standard inputs, and will silently accept nonsensical values (negative prices, zero quantities) without error. + +## Implementation + +8 scripted checks sampling 5 adversarial categories: +1. **Coupon code type confusion**: None, int, list passed to validate_coupon +2. **Items with missing keys**: calculate_total missing 'price', apply_bulk_discount missing 'qty' +3. **Negative values**: calculate_total with negative price and negative quantity +4. **Non-string coupon codes and cache effects**: True, 42 as coupon codes — what gets cached? + +(The broader surface — string/None numeric types, non-dict items, extreme +values — was additionally mapped by code inspection; the script exercises a +representative subset.) + +## Commands Run + +``` +$ python dream_exp3_adversarial.py +Exit code: 0 +Results: 5 expected crashes, 3 silent acceptances +``` + +### Scripted crashes (5): +- `validate_coupon(None)` → AttributeError (no .upper() on None) +- `validate_coupon(123)` → AttributeError (no .upper() on int) +- `validate_coupon(['SAVE20'])` → AttributeError (no .upper() on list) +- `calculate_total` with missing 'price' → KeyError +- `apply_bulk_discount` with missing 'qty' → KeyError + +### Scripted silent acceptance (3): +- `calculate_total` with price=-50, qty=1 → -54.0 (negative total!) +- `calculate_total` with price=50, qty=-1 → -54.0 (negative total!) +- `get_coupon(True)` and `get_coupon(42)` cache `{True: None, 42: None}` — non-string keys pollute the cache + +## Evaluation + +The codebase has zero input validation. Every function trusts its caller completely: +- **validate_coupon**: crashes on any non-string input +- **calculate_total**: crashes on malformed items, silently accepts nonsensical numeric values +- **apply_bulk_discount**: crashes on missing/wrong-type keys +- **COUPON_CACHE**: accepts any hashable type as key, permanently caching garbage entries + +The negative-price acceptance is particularly concerning — a cart with negative items produces negative totals (money owed TO the customer). + +## Takeaways + +- `validate_coupon` needs a type guard: `if not isinstance(code, str): return False` +- `calculate_total` needs item validation or at minimum a try/except with a clear error +- Non-string coupon codes silently pass the `if coupon_code:` truthiness check, hit the cache, and cache garbage — the guard should be `if isinstance(coupon_code, str) and coupon_code:` +- `apply_bulk_discount` with negative qty (<5, so no discount applied) is "correct" by accident — negative qty 5 should probably be rejected +- Empty items list returning 0.0 is actually reasonable behavior + +## Verdict Details + +Useful: Mapped the input-validation surface across all 4 public functions — 5 scripted crash vectors and 3 scripted silent-acceptance vectors, plus additional vectors identified by inspection. New discoveries for validate_coupon (type crashes), calculate_total (non-string coupons), and get_coupon/COUPON_CACHE (non-string key pollution). diff --git a/examples/coupon-demo/.shadow/_dreams/_index.md b/examples/coupon-demo/.shadow/_dreams/_index.md new file mode 100644 index 0000000..6d916ca --- /dev/null +++ b/examples/coupon-demo/.shadow/_dreams/_index.md @@ -0,0 +1,7 @@ +# Dream Experiment Archive + +| dream_id | category | verdict | title | branch | parent | tip_commit | +|----------|----------|---------|-------|--------|--------|------------| +| 20260420-140000Z-cache-poison-sequence | bug hunting | useful | Cache poisoning via validate-then-calculate sequence | dream/coupon-demo/20260420-140000Z-cache-poison-sequence | main | 7f3a1b2 | +| 20260420-141000Z-bulk-min-total-interaction | investigation | useful | Bulk discount vs coupon min_total interaction | dream/coupon-demo/20260420-141000Z-bulk-min-total-interaction | main | 8c2d4e5 | +| 20260420-142000Z-adversarial-inputs | security audit | useful | Adversarial input audit across all public functions | dream/coupon-demo/20260420-142000Z-adversarial-inputs | main | 9e6f7a8 | diff --git a/examples/coupon-demo/.shadow/_index.md b/examples/coupon-demo/.shadow/_index.md new file mode 100644 index 0000000..4043f96 --- /dev/null +++ b/examples/coupon-demo/.shadow/_index.md @@ -0,0 +1,10 @@ +# Shadow Index + +> Generated by shadow-frog-init on 2026-04-20 +> Total files: 3 | Symbols: 9 | Discoveries: 33 | Cross-cutting: 3 | Dream cycles: 3 + +| File | Language | Symbols | Discoveries | +|------|----------|---------|-------------| +| cart.py | Python | 4 (COUPON_CACHE, load_coupon, get_coupon, calculate_total) | 14 | +| inventory.py | Python | 2 (validate_coupon, apply_bulk_discount) | 10 | +| test_cart.py | Python | 3 (test_basic_total, test_coupon, test_bulk_then_coupon) | 9 | diff --git a/examples/coupon-demo/.shadow/_meta/state.json b/examples/coupon-demo/.shadow/_meta/state.json new file mode 100644 index 0000000..422ce6b --- /dev/null +++ b/examples/coupon-demo/.shadow/_meta/state.json @@ -0,0 +1,11 @@ +{ + "version": 1, + "initialized_at": "2026-04-20T14:59:09Z", + "last_update_at": "2026-04-20T16:30:00Z", + "last_commit": "69de9194d8d30d980a41fa0e1bace8fc76d82e24", + "last_update_type": "dream", + "total_files": 3, + "total_symbols": 9, + "total_discoveries": 33, + "dream_cycles_completed": 3 +} diff --git a/examples/coupon-demo/.shadow/_prefs.md b/examples/coupon-demo/.shadow/_prefs.md new file mode 100644 index 0000000..10e1bc8 --- /dev/null +++ b/examples/coupon-demo/.shadow/_prefs.md @@ -0,0 +1,3 @@ +# Preferences + +_No preferences recorded yet._ diff --git a/examples/coupon-demo/.shadow/cart.py.md b/examples/coupon-demo/.shadow/cart.py.md new file mode 100644 index 0000000..63a62af --- /dev/null +++ b/examples/coupon-demo/.shadow/cart.py.md @@ -0,0 +1,65 @@ +# Shadow: cart.py + +**Language**: Python | **Lines**: 28 | **Last modified**: 2026-04-20 + +## File-Level + +- COUPON_CACHE is a module-level mutable global dict shared across all importers — any module that imports from cart.py shares the same cache instance, causing cross-module state pollution. + _(verified, source: exploration)_ + Also involves: `inventory.py::validate_coupon` + +## `COUPON_CACHE` + +- Never cleared or evicted — grows monotonically for the lifetime of the process. In a long-running server, every unique coupon code ever queried remains cached forever. + _(verified, source: exploration, labels: [performance])_ +- Caches None for invalid codes. Once an invalid code is looked up, the None result is permanently cached, preventing any future lookup even if the underlying data changes. + _(verified, source: exploration, labels: [bug])_ +- Case-variant lookups create duplicate cache entries for the same logical coupon. validate_coupon("save20") caches "SAVE20" → valid, then calculate_total("save20") caches "save20" → None. Cache grows 2× faster with mixed-case usage. + _(verified, source: exploration, labels: [bug, performance])_ + Dream report: `_dreams/20260420-140000Z-cache-poison-sequence/` + +## `load_coupon` + +- Case-sensitive lookup against uppercase keys ("SAVE20", "HALF"). Passing lowercase (e.g., "save20") returns None even though the coupon conceptually exists. + _(verified, source: exploration)_ +- Returns None for unknown codes (via dict.get default), not an exception. Callers must handle None. + _(verified, source: exploration)_ + +## `get_coupon` + +- The `if code not in COUPON_CACHE` guard means each code is loaded exactly once per process. But because None is a valid cached value, invalid codes are also "loaded once" and permanently considered invalid. + _(verified, source: exploration, labels: [bug])_ + Also involves: `cart.py::COUPON_CACHE`, `cart.py::load_coupon` +- Accepts any hashable type as key (True, 42, lists-as-errors). Non-string keys permanently cache None entries that can never resolve to valid coupons — silent cache pollution. + _(verified, source: exploration, labels: [security])_ + Dream report: `_dreams/20260420-142000Z-adversarial-inputs/` + Also involves: `cart.py::COUPON_CACHE` + +## `calculate_total` + +- Does NOT normalize coupon_code to uppercase before lookup. Lowercase codes silently produce no discount (coupon returns None from cache or load_coupon). This contradicts validate_coupon which does normalize. + _(verified, source: exploration, labels: [bug])_ + Also involves: `inventory.py::validate_coupon` +- `if coupon_code:` is falsy for empty string "", None, 0, and False — all skip coupon lookup silently. No distinction between "no coupon" and "invalid coupon". + _(verified, source: exploration)_ +- Accepts negative prices and quantities without validation. Negative subtotals still have 8% tax applied, producing negative totals (e.g., price=-10, qty=1 → total=-10.80). + _(verified, source: exploration, labels: [bug])_ +- Tax rate (0.08 = 8%) is hardcoded with no configuration mechanism. Changing tax requires editing source code. + _(verified, source: exploration, labels: [tech-debt])_ +- Coupon min_total check uses the actual subtotal computed from items at call time. Since apply_bulk_discount mutates prices in-place before calculate_total runs, the min_total check sees post-bulk-discount prices. If bulk discount drops subtotal below min_total, the coupon is correctly rejected. + _(verified, source: exploration)_ + Dream report: `_dreams/20260420-141000Z-bulk-min-total-interaction/` + Also involves: `inventory.py::apply_bulk_discount` +- Non-string coupon_code values (True, 42, etc.) pass the `if coupon_code:` truthiness check, reach get_coupon, cache None under the non-string key, and silently produce no discount. No type validation exists. + _(verified, source: exploration, labels: [security])_ + Dream report: `_dreams/20260420-142000Z-adversarial-inputs/` + Also involves: `cart.py::get_coupon`, `cart.py::COUPON_CACHE` + +## Cross-References + +- [coupon-case-normalization-mismatch](_cross/coupon-case-normalization-mismatch.md) + (involves `cart.py::calculate_total`, `inventory.py::validate_coupon`, `cart.py::load_coupon`) +- [global-coupon-cache-side-effects](_cross/global-coupon-cache-side-effects.md) + (involves `cart.py::COUPON_CACHE`, `cart.py::get_coupon`, `inventory.py::validate_coupon`) +- [mutation-through-discount-pipeline](_cross/mutation-through-discount-pipeline.md) + (involves `inventory.py::apply_bulk_discount`, `cart.py::calculate_total`, `test_cart.py::test_bulk_then_coupon`) diff --git a/examples/coupon-demo/.shadow/inventory.py.md b/examples/coupon-demo/.shadow/inventory.py.md new file mode 100644 index 0000000..2dea944 --- /dev/null +++ b/examples/coupon-demo/.shadow/inventory.py.md @@ -0,0 +1,46 @@ +# Shadow: inventory.py + +**Language**: Python | **Lines**: 15 | **Last modified**: 2026-04-20 + +## File-Level + +- Imports get_coupon from cart, creating a dependency on cart's COUPON_CACHE. Any call to validate_coupon pollutes the shared cache as a side effect. + _(verified, source: exploration)_ + Also involves: `cart.py::get_coupon`, `cart.py::COUPON_CACHE` + +## `validate_coupon` + +- Normalizes code to uppercase via `code.upper()` before lookup, but calculate_total does NOT — so a code that validates successfully may still produce no discount when passed directly to calculate_total. + _(verified, source: exploration, labels: [bug])_ + Also involves: `cart.py::calculate_total`, `cart.py::load_coupon` +- Side effect: populates COUPON_CACHE with the uppercased code. Calling validate_coupon("save20") caches under key "SAVE20", but a later calculate_total("save20") looks up lowercase "save20" — a cache miss that then caches None under "save20". + _(verified, source: exploration, labels: [bug])_ + Also involves: `cart.py::COUPON_CACHE`, `cart.py::get_coupon` +- Will crash with AttributeError if code is None (None has no .upper() method). No guard against non-string input. + _(verified, source: exploration, labels: [bug])_ +- Also crashes on any non-string type: int, list, dict all raise AttributeError on .upper(). Needs `isinstance(code, str)` guard. + _(verified, source: exploration, labels: [security])_ + Dream report: `_dreams/20260420-142000Z-adversarial-inputs/` + +## `apply_bulk_discount` + +- Mutates items in-place — modifies the original dict objects' "price" keys. The caller's list is permanently altered. Returns the same list object (not a copy). + _(verified, source: exploration)_ +- Crashes with KeyError if items lack 'qty' key, and TypeError if 'qty' is a string. No input validation. + _(verified, source: exploration, labels: [security])_ + Dream report: `_dreams/20260420-142000Z-adversarial-inputs/` +- Calling apply_bulk_discount twice on the same items compounds the discount: first call gives 0.90×, second gives 0.81×, third gives 0.729×. No idempotency guard. + _(verified, source: exploration, labels: [bug])_ +- Only applies discount to items with qty >= 5. Items with qty 4 or below are untouched, even if the total quantity across all items exceeds 5. + _(verified, source: exploration)_ +- Uses round(price * 0.90, 2) which can produce floating-point artifacts on certain prices. For example, 33.33 * 0.90 = 29.997 → rounds to 30.0, not 29.997. + _(verified, source: exploration)_ + +## Cross-References + +- [coupon-case-normalization-mismatch](_cross/coupon-case-normalization-mismatch.md) + (involves `cart.py::calculate_total`, `inventory.py::validate_coupon`, `cart.py::load_coupon`) +- [global-coupon-cache-side-effects](_cross/global-coupon-cache-side-effects.md) + (involves `cart.py::COUPON_CACHE`, `cart.py::get_coupon`, `inventory.py::validate_coupon`) +- [mutation-through-discount-pipeline](_cross/mutation-through-discount-pipeline.md) + (involves `inventory.py::apply_bulk_discount`, `cart.py::calculate_total`, `test_cart.py::test_bulk_then_coupon`) diff --git a/examples/coupon-demo/.shadow/test_cart.py.md b/examples/coupon-demo/.shadow/test_cart.py.md new file mode 100644 index 0000000..ffcdbe4 --- /dev/null +++ b/examples/coupon-demo/.shadow/test_cart.py.md @@ -0,0 +1,43 @@ +# Shadow: test_cart.py + +**Language**: Python | **Lines**: 31 | **Last modified**: 2026-04-20 + +## File-Level + +- Uses a manual `if __name__ == "__main__"` test runner, not pytest or unittest. Tests are plain functions with assert statements and no setup/teardown. + _(verified, source: exploration)_ +- No test isolation: COUPON_CACHE is a shared global. test_coupon populates the cache with "SAVE20", and test_bulk_then_coupon populates it with "HALF". If test order changes or tests are rerun in the same process, cached values persist from earlier tests. + _(verified, source: exploration, labels: [bug])_ + Also involves: `cart.py::COUPON_CACHE` +- No negative-path tests: no test for invalid coupon codes, negative prices, empty items, or the case-sensitivity mismatch between validate_coupon and calculate_total. + _(verified, source: exploration, labels: [feature-gap])_ + +## `test_basic_total` + +- Verifies: 25.00 × 3 = 75.00 subtotal + 8% tax = 81.00. Tests the simplest happy path with no coupon. + _(verified, source: exploration)_ + +## `test_coupon` + +- Verifies SAVE20 on 75.00 subtotal: 75.00 − 20% = 60.00 + 4.80 tax = 64.80. The 75.00 subtotal exceeds SAVE20's min_total of 50. + _(verified, source: exploration)_ +- Only tests with uppercase coupon code "SAVE20". Does not test lowercase, revealing nothing about the case-normalization bug. + _(verified, source: exploration)_ + Also involves: `cart.py::calculate_total` + +## `test_bulk_then_coupon` + +- Tests the full pipeline: apply_bulk_discount mutates items (25.00 → 22.50), then calculate_total applies HALF coupon on 112.50 subtotal → 56.25 + 4.50 tax = 60.75. + _(verified, source: exploration)_ +- Creates fresh items list, avoiding the mutation-persistence issue. But if this test's items object were reused in a subsequent test, prices would already be 22.50, not 25.00. + _(verified, source: exploration)_ + Also involves: `inventory.py::apply_bulk_discount`, `cart.py::calculate_total` +- Imports apply_bulk_discount inside the function body (lazy import), unlike the module-level import of calculate_total. This is inconsistent but functionally irrelevant. + _(verified, source: exploration, labels: [tech-debt])_ + +## Cross-References + +- [global-coupon-cache-side-effects](_cross/global-coupon-cache-side-effects.md) + (involves `cart.py::COUPON_CACHE`, `cart.py::get_coupon`, `inventory.py::validate_coupon`) +- [mutation-through-discount-pipeline](_cross/mutation-through-discount-pipeline.md) + (involves `inventory.py::apply_bulk_discount`, `cart.py::calculate_total`, `test_cart.py::test_bulk_then_coupon`) diff --git a/examples/coupon-demo/README.md b/examples/coupon-demo/README.md new file mode 100644 index 0000000..696694a --- /dev/null +++ b/examples/coupon-demo/README.md @@ -0,0 +1,76 @@ +# Coupon Demo — what a real `.shadow/` looks like + +A tiny, self-contained example of what ShadowFrog's `.shadow/` knowledge +base looks like on a real (very small) codebase. **Not** an eval — the +systematic eval lives in [`eval/`](../../eval/). This is here so you can +see the shape of the artifact ShadowFrog produces before installing +anything yourself. + +## The codebase + +Three files, ~75 lines total: + +- **`cart.py`** — `load_coupon()` defines coupons (uppercase keys), + `get_coupon()` caches them, `calculate_total()` applies a discount and + 8 % tax. +- **`inventory.py`** — `validate_coupon()` checks existence (calls + `.upper()` before lookup), `apply_bulk_discount()` mutates item prices + in place when qty ≥ 5. +- **`test_cart.py`** — three happy-path tests using `assert` (no pytest). + +These three files contain several real defects (case-normalization +mismatch, cache poisoning, missing input validation, in-place mutation +contracts). All of them appear in `.shadow/` as `verified` discoveries +labeled `bug`, `security`, or `performance`. + +## What's in `.shadow/` + +``` +.shadow/ +├── _index.md Top-level summary (files / symbols / counts) +├── _prefs.md Project preferences (empty in this demo) +├── _meta/state.json Tracking state (HEAD sha, last update, totals) +├── cart.py.md Per-symbol discoveries for cart.py +├── inventory.py.md Per-symbol discoveries for inventory.py +├── test_cart.py.md Per-symbol discoveries for test_cart.py +├── _cross/ Cross-cutting discoveries (3 files) +│ ├── coupon-case-normalization-mismatch.md +│ ├── global-coupon-cache-side-effects.md +│ └── mutation-through-discount-pipeline.md +└── _dreams/ Dream experiment archive (3 experiments) + ├── _index.md + ├── 20260420-140000Z-cache-poison-sequence/ + │ ├── report.md Narrative + verdict + │ ├── manifest.json Machine-readable discoveries + │ └── patch.diff Demonstration script (vs. base_commit) + ├── 20260420-141000Z-bulk-min-total-interaction/ + └── 20260420-142000Z-adversarial-inputs/ +``` + +## Exploring it + +Use the viewer to inspect the shadow exactly the way an installed agent +would: + +```bash +# Summary across the whole shadow +python3 ../../skills/shadow-frog-viewer/shadow-viewer.py --summary + +# Actionable discoveries for a single file (what the preToolUse hook +# inlines before edits to cart.py) +python3 ../../skills/shadow-frog-viewer/shadow-viewer.py --top cart.py + +# All discoveries labeled `bug` +python3 ../../skills/shadow-frog-viewer/shadow-viewer.py --labels bug +``` + +Or just open `.shadow/cart.py.md` in your editor — every per-symbol +discovery is a plain markdown bullet. + +## Where the real eval lives + +See [`eval/README.md`](../../eval/README.md) and +[`eval/results_dashboard.html`](../../eval/results_dashboard.html) for +the systematic SWE-Smith eval (100 bugs × 5 models × ablations). That is +the source of truth for "does the shadow help"; this folder exists only +to show what the shadow itself looks like. diff --git a/examples/coupon-demo/cart.py b/examples/coupon-demo/cart.py new file mode 100644 index 0000000..f8c2544 --- /dev/null +++ b/examples/coupon-demo/cart.py @@ -0,0 +1,28 @@ +COUPON_CACHE = {} + + +def load_coupon(code): + """Simulate loading coupon from database.""" + coupons = { + "SAVE20": {"discount": 0.20, "min_total": 50}, + "HALF": {"discount": 0.50, "min_total": 100}, + } + return coupons.get(code) + + +def get_coupon(code): + if code not in COUPON_CACHE: + COUPON_CACHE[code] = load_coupon(code) + return COUPON_CACHE[code] + + +def calculate_total(items, coupon_code=None): + subtotal = sum(item["price"] * item["qty"] for item in items) + + if coupon_code: + coupon = get_coupon(coupon_code) + if coupon and subtotal >= coupon["min_total"]: + subtotal -= subtotal * coupon["discount"] + + tax = subtotal * 0.08 + return round(subtotal + tax, 2) diff --git a/examples/coupon-demo/inventory.py b/examples/coupon-demo/inventory.py new file mode 100644 index 0000000..6f2d465 --- /dev/null +++ b/examples/coupon-demo/inventory.py @@ -0,0 +1,15 @@ +from cart import get_coupon + + +def validate_coupon(code): + """Check if a coupon code is valid. Returns True/False.""" + coupon = get_coupon(code.upper()) + return coupon is not None + + +def apply_bulk_discount(items): + """Apply 10% discount if buying 5+ of any single item.""" + for item in items: + if item["qty"] >= 5: + item["price"] = round(item["price"] * 0.90, 2) + return items diff --git a/examples/coupon-demo/test_cart.py b/examples/coupon-demo/test_cart.py new file mode 100644 index 0000000..3af4d2d --- /dev/null +++ b/examples/coupon-demo/test_cart.py @@ -0,0 +1,31 @@ +from cart import calculate_total + + +def test_basic_total(): + items = [{"name": "Widget", "price": 25.00, "qty": 3}] + total = calculate_total(items) + assert total == 81.00, f"Expected 81.00, got {total}" + + +def test_coupon(): + items = [{"name": "Widget", "price": 25.00, "qty": 3}] + total = calculate_total(items, coupon_code="SAVE20") + assert total == 64.80, f"Expected 64.80, got {total}" + + +def test_bulk_then_coupon(): + from inventory import apply_bulk_discount + items = [{"name": "Widget", "price": 25.00, "qty": 5}] + items = apply_bulk_discount(items) + total = calculate_total(items, coupon_code="HALF") + assert total == 60.75, f"Expected 60.75, got {total}" + + +if __name__ == "__main__": + test_basic_total() + print("pass: test_basic_total") + test_coupon() + print("pass: test_coupon") + test_bulk_then_coupon() + print("pass: test_bulk_then_coupon") + print("All tests passed!") diff --git a/froggy_logo.png b/froggy_logo.png new file mode 100644 index 0000000..3a3892f Binary files /dev/null and b/froggy_logo.png differ diff --git a/hook-templates/check-hook-failopen.py b/hook-templates/check-hook-failopen.py new file mode 100755 index 0000000..a156d47 --- /dev/null +++ b/hook-templates/check-hook-failopen.py @@ -0,0 +1,276 @@ +#!/usr/bin/env python3 +"""Guard the fail-open contract of advisory hook scripts. + +Copilot CLI >= 1.0.57 treats any non-zero exit from a preToolUse command hook +as a tool-call DENY. Our hooks therefore MUST run fail-open under every +condition. This script enforces the invariants required for that guarantee. + +Invariants (failing any of these blocks the PR): + + 1. NO `set -e`, `set -u`, `set -o errexit|nounset|pipefail|errtrace` in any + spelling. Original v1.0.57 bug was `set -euo pipefail`. Long-form + spellings (`set -o errexit`) plus four bypasses must also be rejected: + `[[ X ]] && set -e`, `set \\\n -e` (line continuation), `; set -e` + (after semicolon), and `eval 'set -e'`. The matcher scans substrings, + not just line starts, and pre-joins line-continuations. + + 2. NO `source` / `.` of external files. A sourced file with `set -e` would + bypass an inline-only check. + + 3. `trap 'exit 0' EXIT` present, on its own non-comment line. The EXIT trap + converts clean non-zero exits to 0. + + 4. `trap 'exit 0' TERM HUP INT` (or equivalent covering at least TERM) + present, on its own non-comment line. EXIT alone returns 143/-15 under + SIGTERM — verified empirically on bash 3.2 (macOS) and bash 5+ (Linux). + Without this, the runner's timeout-kill still denies the tool. Signal + aliases (`SIGTERM`, numeric `15`) are normalized to canonical names. + + 5. NO raw `git` or unbounded raw `python3` invocations at the bash level + outside of bounded Python `subprocess.run(timeout=...)` wrappers. An + unbounded foreground child queues signals (the trap pyramid cannot fire + while bash is in `wait`) — confirmed by empirical reproduction of a 31s + hang on `git rev-parse --show-toplevel`. + +Usage: python3 check-hook-failopen.py path/to/hook1.sh path/to/hook2.sh ... +""" +import re +import sys + + +# Block: any form of strict-mode flags that would propagate non-zero exits. +# Four bypasses of the naive `^\s*set\s+` anchor must be caught: +# 1. `[[ X ]] && set -e` (combinator) +# 2. `set \\\n -e` (line continuation — joined before regex) +# 3. `; set -e` (semicolon) +# 4. `eval 'set -e'` (literal inside string) +# Substring matching catches all four. The leading boundary requires that +# `set` is either at line start (post-strip), or after a non-word context: +# whitespace, shell control char, or quote. +FORBIDDEN_SET_RE = re.compile( + r"(?:^|[\s;&|'\"`])set\s+(" + # Short form: any combination containing e, u, or E (errtrace). + r"-[a-zA-Z]*[euE][a-zA-Z]*" + # Long form: -o errexit / nounset / pipefail / errtrace + r"|-o\s+(errexit|nounset|pipefail|errtrace)" + r")\b" +) + +# Block: source / . of external files. +FORBIDDEN_SOURCE_RE = re.compile(r"^\s*(source|\.)\s+[^\s#]+") + +# Require: trap '...exit 0...' that includes EXIT, on a non-comment line. +# Signal list allows alphanumerics so `SIGTERM` and numeric `15` are +# captured (then normalized below). +TRAP_EXIT_RE = re.compile( + r"^\s*trap\s+['\"][^'\"]*exit\s+0[^'\"]*['\"]\s+([A-Za-z0-9\s]+)\s*(#.*)?$" +) + +# Require: same trap form covering at least TERM (signal-kill defense). +# The set of signals we require to be covered: +REQUIRED_SIGNALS = {"TERM"} + +# Map common signal aliases to their canonical short name. Bash accepts +# `SIGTERM`, `TERM`, and `15` interchangeably; the checker must too, +# else it false-rejects perfectly valid hooks. +SIGNAL_ALIASES = { + "SIGTERM": "TERM", "15": "TERM", + "SIGINT": "INT", "2": "INT", + "SIGHUP": "HUP", "1": "HUP", + "SIGPIPE": "PIPE", "13": "PIPE", + "SIGQUIT": "QUIT", "3": "QUIT", + "SIGEXIT": "EXIT", "0": "EXIT", +} + +def _canonicalize_signal(s: str) -> str: + return SIGNAL_ALIASES.get(s, s) + +# Detect raw `git` in bash code (outside Python heredocs). +RAW_GIT_RE = re.compile(r"^\s*(?:[A-Z_0-9]+=)?\$?\(?\s*git\s+\S") + +# Detect raw `python3` script invocations at bash level (outside heredocs) +# that aren't `python3 -c ""` AND aren't `python3 - < str: + """Drop trailing comment, respecting single-quoted strings.""" + in_squote = False + out = [] + for ch in line: + if ch == "'" and not in_squote: + in_squote = True + elif ch == "'" and in_squote: + in_squote = False + elif ch == "#" and not in_squote: + break + out.append(ch) + return "".join(out) + + +def _join_continuations(text: str) -> str: + """Join lines ending in `\\` so a continuation cannot smuggle `set -e` + past a per-line regex.""" + # Replace backslash-newline with a single space, preserving overall + # structure. We do NOT touch backslashes that aren't followed by + # newline (those are escape sequences inside strings or paths). + return re.sub(r"\\\n", " ", text) + + +def _is_in_heredoc(lines: list[str], idx: int) -> bool: + """Return True if line `idx` is inside a heredoc (e.g., python3 -< idx: + break + if in_heredoc: + stripped = line.strip() + if stripped == heredoc_term: + in_heredoc = False + heredoc_term = None + continue + # Look for heredoc start: <<'TERM' or < list[str]: + """Return list of error messages (empty if clean).""" + errors: list[str] = [] + try: + text = open(path).read() + except OSError as e: + return [f"{path}: cannot read: {e}"] + + # Join line continuations so a `set \\\n -e` cannot evade per-line regex. + text = _join_continuations(text) + lines = text.splitlines() + + has_exit_trap = False + trap_signals_covered: set[str] = set() + + for lineno, raw_line in enumerate(lines, 1): + # Skip lines inside heredocs — those are Python/awk/etc., not bash. + if _is_in_heredoc(lines, lineno - 1): + continue + + # Strip trailing comment to avoid false positives on comment text. + line = _strip_comments(raw_line) + if not line.strip(): + continue + + # Invariant 1: strict-mode flags (substring scan; catches + # `&& set -e`, `; set -e`, `eval 'set -e'`, and joined + # `set \\\n -e`). + if FORBIDDEN_SET_RE.search(line): + errors.append( + f"{path}:{lineno}: FORBIDDEN strict-mode flag in advisory hook: {raw_line.strip()!r}\n" + f" Hooks MUST NOT use 'set -e', 'set -u', 'set -o errexit|nounset|pipefail|errtrace'.\n" + f" Copilot CLI >= 1.0.57 treats non-zero preToolUse exit as DENY." + ) + + # Invariant 2: source / dot of external files + if FORBIDDEN_SOURCE_RE.match(line): + errors.append( + f"{path}:{lineno}: FORBIDDEN source/dot of external file: {raw_line.strip()!r}\n" + f" A sourced file could re-introduce 'set -e' and bypass this guard.\n" + f" Inline the needed code in the hook script directly." + ) + + # Capture trap declarations (normalize signal aliases). + m = TRAP_EXIT_RE.match(line) + if m: + signals = [_canonicalize_signal(s) for s in m.group(1).split()] + if "EXIT" in signals: + has_exit_trap = True + for sig in signals: + trap_signals_covered.add(sig) + + # Invariant 5: raw git outside of bounded Python wrapper. + # We allow `git` only inside heredoc'd Python (handled above by skip). + if RAW_GIT_RE.match(line): + errors.append( + f"{path}:{lineno}: UNBOUNDED 'git' call at bash level: {raw_line.strip()!r}\n" + f" Hooks MUST wrap every git call in Python subprocess.run(timeout=...)\n" + f" to prevent runner timeout-kill on slow/locked repos.\n" + f" (bash's signal trap is queued while waiting for a foreground child.)" + ) + + # Invariant 5b: raw python3 script invocation (unbounded blast radius). + # We allow `python3 -c ""` (bash literal) and `python3 - <\"`\n" + f" (small inline literal) or `python3 - < int: + if len(sys.argv) < 2: + print("usage: check-hook-failopen.py [ ...]", file=sys.stderr) + return 2 + + all_errors: list[str] = [] + for path in sys.argv[1:]: + all_errors.extend(check_file(path)) + + if all_errors: + print("\n".join(all_errors)) + print() + print(f"FAIL: {len(all_errors)} fail-open contract violation(s) found.") + print("See CHANGELOG.md (2026-06-02) and claude.md (Hook Format).") + return 1 + + print(f"OK: fail-open contract intact across {len(sys.argv) - 1} hook script(s).") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/hook-templates/claude-settings.json b/hook-templates/claude-settings.json new file mode 100644 index 0000000..966306d --- /dev/null +++ b/hook-templates/claude-settings.json @@ -0,0 +1,27 @@ +{ + "hooks": { + "SessionStart": [ + { + "hooks": [ + { + "type": "command", + "command": "\"${CLAUDE_PROJECT_DIR}/.claude/hooks/scripts/shadow-frog-check-init.sh\"", + "timeout": 5 + } + ] + } + ], + "PreToolUse": [ + { + "matcher": "Edit|Write|MultiEdit|NotebookEdit", + "hooks": [ + { + "type": "command", + "command": "\"${CLAUDE_PROJECT_DIR}/.claude/hooks/scripts/shadow-frog-pre-tool.sh\"", + "timeout": 5 + } + ] + } + ] + } +} diff --git a/hook-templates/scripts/shadow-frog-check-init.sh b/hook-templates/scripts/shadow-frog-check-init.sh new file mode 100755 index 0000000..550298a --- /dev/null +++ b/hook-templates/scripts/shadow-frog-check-init.sh @@ -0,0 +1,145 @@ +#!/usr/bin/env bash +# Hook: sessionStart — check if .shadow/ exists, report status +# Outputs JSON with additionalContext for injection into agent conversation. + +# sessionStart is fail-open by contract, but we still run defensively so an +# internal hiccup never produces noisy stderr or a non-zero exit. See +# shadow-frog-pre-tool.sh for the full defense-in-depth rationale. Same trap +# pyramid: EXIT handles clean non-zero exits; TERM/HUP/INT handles runner- +# initiated signal kills (EXIT alone returns 143/-15 under SIGTERM). +trap 'exit 0' EXIT +trap 'exit 0' TERM HUP INT PIPE + +# Defensive PATH — some runners launch hooks with a stripped PATH (observed +# under sandboxed containers). Append the standard binary dirs so cat, +# python3, and git remain findable. We APPEND (not prepend) to honor any +# custom paths the runner exported intentionally. +PATH="${PATH:+$PATH:}/usr/local/bin:/usr/bin:/bin" +export PATH + +# Consume stdin (the JSON payload) to keep the runner's pipe drained, even +# though check-init doesn't need its content. shellcheck flags INPUT as unused; +# that's intentional. Brace group silences bash's own "ignored null byte" +# warning (bash 5+) if stdin ever contains NULs. +# shellcheck disable=SC2034 +{ INPUT=$(cat 2>/dev/null || echo ""); } 2>/dev/null + +if [ ! -d ".shadow" ]; then + python3 -c "import json; ctx='[ShadowFrog] No .shadow/ directory found. Run /shadow-frog-init to create one.'; print(json.dumps({'additionalContext': ctx, 'hookSpecificOutput': {'hookEventName': 'SessionStart', 'additionalContext': ctx}}))" 2>/dev/null + exit 0 +fi + +# Read state.json in ONE python3 invocation (instead of three separate calls). +# Single startup cost; smaller blast radius if python3 is slow or state.json +# is unusual. Output is three pipe-separated values on stdout: +# || +STATE_INFO=$(python3 -c " +import json +try: + s = json.load(open('.shadow/_meta/state.json')) + if not isinstance(s, dict): + s = {} +except Exception: + s = {} +print(str(s.get('total_files', 0)) + '|' + str(s.get('total_discoveries', 0)) + '|' + str(s.get('last_update_at', 'unknown'))) +" 2>/dev/null || echo "0|0|unknown") +TOTAL_FILES="${STATE_INFO%%|*}" +REST="${STATE_INFO#*|}" +TOTAL_DISCOVERIES="${REST%%|*}" +LAST_UPDATE="${REST#*|}" +[ -z "$TOTAL_FILES" ] && TOTAL_FILES=0 +[ -z "$TOTAL_DISCOVERIES" ] && TOTAL_DISCOVERIES=0 +[ -z "$LAST_UPDATE" ] && LAST_UPDATE="unknown" + +# Garbage-collect old dedup directories from past sessions. Without this, +# /tmp/shadowfrog-hook-* accumulates over months and can hit inode caps on +# long-lived workstations. SessionStart fires once per Copilot session, so +# it's the natural GC trigger (vs pre-tool which fires many times per +# session). Bounded inside Python with top-level try/except so any +# filesystem hiccup is silently absorbed. +python3 - <<'PYEOF' 2>/dev/null || true +import os, time, shutil +tmp_root = os.environ.get('SHADOWFROG_TMP_DIR') or os.environ.get('TMPDIR') or '/tmp' +TTL_SEC = 24 * 60 * 60 +now = time.time() +try: + entries = os.listdir(tmp_root) +except Exception: + entries = [] +for entry in entries: + if not entry.startswith('shadowfrog-hook-'): + continue + p = os.path.join(tmp_root, entry) + try: + if (now - os.path.getmtime(p)) > TTL_SEC: + shutil.rmtree(p, ignore_errors=True) + except Exception: + pass +PYEOF + +# Check staleness. Git work is bounded with per-call subprocess timeouts so a +# large/locked repo can't exceed the hook budget. Timeouts sum to 2.0s, +# leaving >=3s headroom under timeoutSec=5. Any failure -> no warning. +STALE_MSG="" +CHANGED=$(python3 - <<'PYEOF' 2>/dev/null || echo "" +import json, subprocess + +def git(args, timeout): + return subprocess.run(["git"] + args, capture_output=True, text=True, timeout=timeout) + +try: + last = json.load(open(".shadow/_meta/state.json")).get("last_commit", "none") + head = git(["rev-parse", "HEAD"], 0.5).stdout.strip() + if last and last != "none" and last != head \ + and git(["rev-parse", "--verify", last], 0.5).returncode == 0: + r = git(["diff", "--name-only", last, "HEAD", "--", ":!.shadow"], 1.0) + n = len([ln for ln in r.stdout.splitlines() if ln.strip()]) + if n > 0: + print(n) +except Exception: + pass +PYEOF +) +if [ -n "$CHANGED" ]; then + STALE_MSG=" WARNING: ${CHANGED} file(s) changed since last update — consider running /shadow-frog-update." +fi + +# Read top preferences +PREFS="" +if [ -f ".shadow/_prefs.md" ]; then + PREFS=$(grep '^\- ' .shadow/_prefs.md 2>/dev/null | head -3 | sed 's/"/\\"/g' | tr '\n' ' ' || echo "") +fi + +# Output JSON — additionalContext gets injected into the conversation +export SF_FILES="$TOTAL_FILES" +export SF_DISC="$TOTAL_DISCOVERIES" +export SF_UPDATED="$LAST_UPDATE" +export SF_STALE="$STALE_MSG" +export SF_PREFS="$PREFS" +python3 << 'PYEOF' 2>/dev/null +import json, os + +files = os.environ.get("SF_FILES", "0") +disc = os.environ.get("SF_DISC", "0") +updated = os.environ.get("SF_UPDATED", "unknown") +stale = os.environ.get("SF_STALE", "") +prefs = os.environ.get("SF_PREFS", "") + +parts = [ + f"[ShadowFrog] Shadow loaded — {files} files, {disc} discoveries, last updated {updated}.", + stale, + "Before editing any file, check .shadow/.md for known bugs, edge cases, and implicit contracts.", + "Read .shadow/_prefs.md for project conventions the user wants followed.", +] +if prefs: + parts.append(f"Top preferences: {prefs}") + +ctx = " ".join(p for p in parts if p) +# Emit both shapes: top-level `additionalContext` (Copilot CLI) and the +# nested `hookSpecificOutput` form (Claude Code). Each agent reads its own +# key and ignores the other, so one payload drives both platforms. +print(json.dumps({ + "additionalContext": ctx, + "hookSpecificOutput": {"hookEventName": "SessionStart", "additionalContext": ctx}, +})) +PYEOF diff --git a/hook-templates/scripts/shadow-frog-pre-tool.sh b/hook-templates/scripts/shadow-frog-pre-tool.sh new file mode 100755 index 0000000..52adb33 --- /dev/null +++ b/hook-templates/scripts/shadow-frog-pre-tool.sh @@ -0,0 +1,242 @@ +#!/usr/bin/env bash +# Hook: preToolUse — always remind agent to consult the shadow knowledge base +# Outputs JSON with additionalContext before every tool execution. +# Includes file-specific hints for edit/create and staleness warnings when behind. +# +# When a mutation tool (edit/create/str_replace/write) targets a file with +# a shadow that has actionable discoveries (bug/security labels), the +# top entries are inlined into additionalContext via +# shadow-viewer.py --top. Per-session dedup ensures the same file's +# content is injected at most once per Copilot CLI process. + +# This hook is ADVISORY — it only injects shadow context, it is NOT a security +# gate. Copilot CLI >= 1.0.57 treats a non-zero preToolUse exit as a DENY of the +# user's tool call, so this script MUST run fail-open under every condition. +# +# Defense in depth: +# 1. No `set -e`/`-u`/`pipefail` — a failing sub-step doesn't abort the script. +# 2. trap on EXIT — converts clean non-zero exits to 0. +# 3. trap on TERM/HUP/INT — converts runner-initiated signal kills to 0. +# (bash 3.2+ on macOS and bash 5+ on Linux verified: EXIT alone is NOT +# enough — SIGTERM still produces exit 143/-15 without a TERM trap.) +# 4. Every external call (git, python3, shadow-viewer.py) MUST be wrapped in +# a bounded subprocess timeout. If the foreground child hangs, bash will +# queue the signal until the child returns, so the trap can't save us +# unless boundedness holds. CI enforces this via .github/workflows/shellcheck.yml. +# 5. Total bounded work budget is ~3.5s, leaving >=1.5s headroom under the +# hook's 5s timeoutSec configured in shadow-frog-hooks.json. +trap 'exit 0' EXIT +trap 'exit 0' TERM HUP INT PIPE + +# Defensive PATH — some runners launch hooks with a stripped PATH (observed +# under sandboxed containers). Append the standard binary dirs so cat, +# python3, and git remain findable. We APPEND (not prepend) to honor any +# custom paths the runner exported intentionally. Without this, bash itself +# emits `cat: No such file or directory` to stderr. +PATH="${PATH:+$PATH:}/usr/local/bin:/usr/bin:/bin" +export PATH + +# Wrap in a brace group so bash's own "ignored null byte in input" warning +# (emitted by bash 5+ when command substitution captures NULs) is silenced +# alongside cat's stderr. +{ INPUT=$(cat 2>/dev/null || echo ""); } 2>/dev/null + +[ ! -d ".shadow" ] && exit 0 + +# Base reminder (always injected — agent decides whether to act on it). +# Kept terse: this fires on every tool call, so every char counts against +# the per-call prompt budget. Details live in /shadow-frog. +MSG="[ShadowFrog] .shadow/ mirrors the repo (src/x.py -> .shadow/src/x.py.md). Check before edits; capture user-shared knowledge as source: user; preferences -> .shadow/_prefs.md. /shadow-frog" + +# Extract tool name + target file in ONE python3 invocation (instead of two +# separate `python3 -c` calls). Single startup cost; smaller blast radius if +# python3 is slow/broken. Output is two pipe-separated values on stdout: +# | +# Support both Copilot CLI (camelCase fields, lowercase tool names like +# `edit`/`create`/`str_replace`/`write`) and Claude Code (snake_case fields, +# PascalCase tool names like `Edit`/`Write`/`MultiEdit`/`NotebookEdit`). +# Normalize tool name to lowercase so one case statement handles both. +# Prefer `file_path` over `path` — both Copilot CLI and Claude Code use +# `file_path` as the canonical key; `path` is a less-common alias. When both +# are present, `file_path` wins. +TOOL_INFO=$(echo "$INPUT" | python3 -c " +import json, sys +try: + d = json.load(sys.stdin) +except Exception: + d = {} +if not isinstance(d, dict): + d = {} +name = (d.get('toolName') or d.get('tool_name') or '') +if not isinstance(name, str): + name = '' +args = d.get('toolInput') or d.get('tool_input') or d.get('toolArgs') or d.get('tool_args') or {} +target = '' +if isinstance(args, dict): + target = args.get('file_path') or args.get('path') or '' + if not isinstance(target, str): + target = '' +# Strip NUL and newline bytes — bash 5+ warns when command substitution +# captures NULs, and a newline in target would break the pipe-delimited +# output format on parse. These bytes never appear in legitimate tool +# names or file paths; an adversarial JSON \u0000 escape is the only way +# they reach this point. +def _scrub(s): + return s.replace('\x00', '').replace('\n', '').replace('\r', '') +print(_scrub(name).lower() + '|' + _scrub(target)) +" 2>/dev/null || echo "|") +TOOL_NAME="${TOOL_INFO%%|*}" +TARGET_FILE="${TOOL_INFO#*|}" +case "$TOOL_NAME" in + edit|create|str_replace|write|multiedit|notebookedit) + IS_MUTATION=1 + ;; + *) + IS_MUTATION=0 + ;; +esac + +if [ "$IS_MUTATION" = "1" ]; then + # Oversized path → fall back to base reminder. A 200KB path took + # ~5.3s in benchmarks, exceeding the production deny threshold. Real + # file paths are well under 1KB; anything larger is malformed or + # adversarial. The base reminder still fires. + if [ -n "$TARGET_FILE" ] && [ "${#TARGET_FILE}" -le 1024 ]; then + # CWD is where .shadow/ lives (verified above). Compute the + # file's path relative to CWD. This works for both + # at-repo-root and subdirectory shadow setups (e.g., + # examples/coupon-demo within a parent monorepo). + CWD=$(pwd) + REL_PATH="${TARGET_FILE#"$CWD"/}" + SHADOW_FILE=".shadow/${REL_PATH}.md" + if [ -f "$SHADOW_FILE" ]; then + # Default pointer message if --top is unavailable or empty + MSG="[ShadowFrog] .shadow/${REL_PATH}.md has discoveries — check before changes. Capture user-shared knowledge as source: user. /shadow-frog" + + # Per-session dedup — keyed on PPID + parent process start time so the + # bucket doesn't collide across (a) PID-wrap on long-running systems, + # (b) Claude Code sessions sharing PPID=1 under a process manager, or + # (c) different shells of the same user transiently sharing PPID. Falls + # back to PPID alone if `ps` is unavailable (some sandboxed containers). + # Hooks must remain pure-read w.r.t. .shadow/, so the + # dedup marker stays outside the shadow tree. + TMP_ROOT="${SHADOWFROG_TMP_DIR:-/tmp}" + PARENT_START=$(ps -p "$PPID" -o lstart= 2>/dev/null | tr -d ' :' || echo "") + SESSION_KEY="${PPID}${PARENT_START:+-${PARENT_START}}" + DEDUP_DIR="${TMP_ROOT}/shadowfrog-hook-${SESSION_KEY}" + mkdir -p "$DEDUP_DIR" 2>/dev/null || true + SAFE_PATH=$(echo "$REL_PATH" | tr '/' '_' | tr -cd '[:alnum:]_.-') + DEDUP_FILE="${DEDUP_DIR}/${SAFE_PATH}.injected" + + if [ ! -f "$DEDUP_FILE" ]; then + SCRIPT_DIR="" + if [ -n "$0" ]; then + SCRIPT_DIR="$(cd "$(dirname "$0")" 2>/dev/null && pwd -P)" || SCRIPT_DIR="" + fi + # Resolve viewer + run it inside ONE Python block. Every + # external call (git rev-parse, viewer subprocess) is + # bounded with subprocess.run(timeout=...) so a hung git + # or hung viewer can't blow the hook's 5s budget — even + # the trap pyramid can't help if bash is blocked waiting + # on an unbounded foreground child (signals are queued). + # Total bounded work here is ~1.5s (rev-parse 0.5s + viewer 1.0s). + TOP_OUTPUT=$(SF_SCRIPT_DIR="$SCRIPT_DIR" SF_REL_PATH="$REL_PATH" python3 - <<'PYEOF' 2>/dev/null || true +import os, subprocess, sys + +script_dir = os.environ.get('SF_SCRIPT_DIR', '') +rel_path = os.environ.get('SF_REL_PATH', '') + +def _git(args, timeout): + try: + r = subprocess.run(['git'] + args, capture_output=True, text=True, timeout=timeout) + if r.returncode == 0: + return r.stdout.strip() + except Exception: + pass + return '' + +# Locate shadow-viewer.py. Order: +# 1. Script-relative — works for source repo dev AND project installs +# (hooks at .github/hooks/scripts/ co-located with .github/skills/). +# 2-3. Parent repo's .github/ or .claude/ skills — useful when the hook +# is run from a subdir of a monorepo with an installed parent. +repo_root = _git(['rev-parse', '--show-toplevel'], 0.5) + +candidates = [] +if script_dir: + candidates.append(os.path.join(script_dir, '..', '..', 'skills', 'shadow-frog-viewer', 'shadow-viewer.py')) +if repo_root: + candidates.append(os.path.join(repo_root, '.github', 'skills', 'shadow-frog-viewer', 'shadow-viewer.py')) + candidates.append(os.path.join(repo_root, '.claude', 'skills', 'shadow-frog-viewer', 'shadow-viewer.py')) + +viewer = '' +for candidate in candidates: + if os.path.isfile(candidate): + viewer = candidate + break + +if viewer: + try: + r = subprocess.run( + ['python3', viewer, + '--shadow-dir', '.shadow', + '--top', rel_path, + '--top-labels', 'bug,security', + '--top-limit', '3', + '--top-max-chars', '600'], + capture_output=True, text=True, timeout=1.0, + ) + if r.returncode == 0: + sys.stdout.write(r.stdout.strip()) + except Exception: + pass +PYEOF +) + # Only inline when the viewer produced an actionable + # response (non-empty and not the "no discoveries" sentinel). + if [ -n "$TOP_OUTPUT" ] && [[ "$TOP_OUTPUT" != "No actionable"* ]]; then + MSG="[ShadowFrog] Actionable discoveries for ${REL_PATH} (verify against source before acting): +${TOP_OUTPUT} +Capture user-shared knowledge as source: user. /shadow-frog" + touch "$DEDUP_FILE" 2>/dev/null || true + fi + fi + fi + fi +fi + +# Staleness warning (appended when shadow is behind HEAD). +# All git work is bounded with per-call subprocess timeouts so a huge/locked +# repo can't blow past the hook's 5s budget. Timeouts sum to 2.0s here, +# matched with the viewer's ~1.5s above + bash overhead = ~4s worst case, +# leaving 1s headroom under timeoutSec=5. Any failure/timeout -> no warning. +CHANGED=$(python3 - <<'PYEOF' 2>/dev/null || echo "" +import json, subprocess + +def git(args, timeout): + return subprocess.run(["git"] + args, capture_output=True, text=True, timeout=timeout) + +try: + last = json.load(open(".shadow/_meta/state.json")).get("last_commit", "none") + head = git(["rev-parse", "HEAD"], 0.5).stdout.strip() + if last and last != "none" and last != head \ + and git(["rev-parse", "--verify", last], 0.5).returncode == 0: + r = git(["diff", "--name-only", last, "HEAD", "--", ":!.shadow"], 1.0) + n = len([ln for ln in r.stdout.splitlines() if ln.strip()]) + if n > 0: + print(n) +except Exception: + pass +PYEOF +) +if [ -n "$CHANGED" ]; then + MSG="${MSG} Shadow is behind HEAD — ${CHANGED} file(s) changed since last update. Run /shadow-frog-update when ready." +fi + +export SF_MSG="$MSG" +# Emit both shapes: top-level `additionalContext` (Copilot CLI) and the +# nested `hookSpecificOutput` form (Claude Code). Each agent reads its own +# key and ignores the other. +python3 -c "import json,os; ctx=os.environ['SF_MSG']; print(json.dumps({'additionalContext': ctx, 'hookSpecificOutput': {'hookEventName': 'PreToolUse', 'additionalContext': ctx}}))" 2>/dev/null || true + +exit 0 diff --git a/hook-templates/shadow-frog-hooks.json b/hook-templates/shadow-frog-hooks.json new file mode 100644 index 0000000..40b03d6 --- /dev/null +++ b/hook-templates/shadow-frog-hooks.json @@ -0,0 +1,20 @@ +{ + "version": 1, + "hooks": { + "sessionStart": [ + { + "type": "command", + "bash": ".github/hooks/scripts/shadow-frog-check-init.sh", + "timeoutSec": 5 + } + ], + "preToolUse": [ + { + "matcher": "^(edit|create|str_replace|write|multiedit|notebookedit|Edit|Write|MultiEdit|NotebookEdit)$", + "type": "command", + "bash": ".github/hooks/scripts/shadow-frog-pre-tool.sh", + "timeoutSec": 5 + } + ] + } +} diff --git a/install.sh b/install.sh new file mode 100755 index 0000000..d68bff2 --- /dev/null +++ b/install.sh @@ -0,0 +1,260 @@ +#!/usr/bin/env bash +# ShadowFrog Installer +# Installs ShadowFrog skills and hooks for GitHub Copilot CLI and Claude Code. +# +# Usage: +# ./install.sh --project /path/to/repo Install skills + hooks + context into a repo +# ./install.sh --project /path/to/repo --no-hooks --no-context Skip hooks/context + +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +SKILLS_DIR="$SCRIPT_DIR/skills" +HOOKS_DIR="$SCRIPT_DIR/hook-templates" +CONTEXT_FILE="$SCRIPT_DIR/agent-context.md" + +# Colors +GREEN='\033[0;32m' +RED='\033[0;31m' +YELLOW='\033[1;33m' +BLUE='\033[0;34m' +NC='\033[0m' # No Color + +# Parse arguments +PROJECT_DIR="" +NO_HOOKS=false +NO_CONTEXT=false +AGENT="copilot" # which agent's conventions to target: copilot (default) | claude +USAGE="Usage: $0 --project [--agent copilot|claude] [--no-hooks] [--no-context]" +while [[ $# -gt 0 ]]; do + case "$1" in + --agent) + if [[ $# -lt 2 ]]; then + echo "Error: --agent requires a value (copilot or claude)." >&2 + echo "$USAGE" >&2 + exit 1 + fi + AGENT="$2" + shift 2 + ;; + --project) + if [[ $# -lt 2 ]]; then + echo "Error: --project requires a directory argument." >&2 + echo "$USAGE" >&2 + exit 1 + fi + PROJECT_DIR="$2" + shift 2 + ;; + --no-hooks) + NO_HOOKS=true + shift + ;; + --no-context) + NO_CONTEXT=true + shift + ;; + -h|--help) + echo "$USAGE" + echo "" + echo " --agent AGENT Target agent conventions: 'copilot' (default) or" + echo " 'claude'. Copilot uses .github/ + copilot-instructions.md;" + echo " Claude uses .claude/ + CLAUDE.md." + echo " --project DIR (Required) Install skills, hooks, and context into DIR" + echo " for local + cloud agent use." + echo " --no-hooks Skip hook installation." + echo " --no-context Skip agent-context injection." + exit 0 + ;; + *) + echo -e "${YELLOW}Unknown argument: $1${NC}" + echo "$USAGE" + exit 1 + ;; + esac +done + +if [[ "$AGENT" != "copilot" && "$AGENT" != "claude" ]]; then + echo "Error: --agent must be 'copilot' or 'claude' (got '$AGENT')." >&2 + echo "$USAGE" >&2 + exit 1 +fi + +if [ -z "$PROJECT_DIR" ]; then + echo "Error: --project is required." >&2 + echo "ShadowFrog installs into a specific repository, not globally — its" >&2 + echo "shadow-edit hooks should only fire inside projects you opted in." >&2 + echo "$USAGE" >&2 + exit 1 +fi + +echo -e "${GREEN}ShadowFrog Installer${NC}" +echo "====================" +echo "" + +# ============================================================================= +# Project install: copy skills, hooks, and context into a repo +# ============================================================================= + +if [ -n "$PROJECT_DIR" ]; then + if [ ! -d "$PROJECT_DIR" ]; then + echo -e "${YELLOW}Error: $PROJECT_DIR does not exist.${NC}" + exit 1 + fi + + echo -e "${BLUE}Project install ($AGENT) → $PROJECT_DIR${NC}" + echo "" + + # Resolve per-agent conventions. Copilot CLI uses .github/ and + # copilot-instructions.md; Claude Code uses .claude/ and CLAUDE.md. + if [ "$AGENT" = "claude" ]; then + SKILLS_TARGET="$PROJECT_DIR/.claude/skills" + HOOKS_SCRIPTS_DIR="$PROJECT_DIR/.claude/hooks/scripts" + CONTEXT_TARGET="$PROJECT_DIR/CLAUDE.md" + else + SKILLS_TARGET="$PROJECT_DIR/.github/skills" + HOOKS_SCRIPTS_DIR="$PROJECT_DIR/.github/hooks/scripts" + CONTEXT_TARGET="$PROJECT_DIR/.github/copilot-instructions.md" + fi + + # --- Copy skills --- + echo -e "${BLUE}Installing skills to ${SKILLS_TARGET#"$PROJECT_DIR"/}/...${NC}" + mkdir -p "$SKILLS_TARGET" + for skill_dir in "$SKILLS_DIR"/shadow-frog*; do + [[ -e "$skill_dir" ]] || continue + skill_name=$(basename "$skill_dir") + target="$SKILLS_TARGET/$skill_name" + + # Sync: remove old and copy fresh (project skills are committed, not symlinked) + rm -rf "$target" + cp -r "$skill_dir" "$target" + echo -e " ${GREEN}✓${NC} $skill_name → $target" + done + + # --- Copy hooks (default: yes) --- + if [ "$NO_HOOKS" = false ]; then + echo "" + echo -e "${BLUE}Installing hooks...${NC}" + mkdir -p "$HOOKS_SCRIPTS_DIR" + cp -f "$HOOKS_DIR"/scripts/*.sh "$HOOKS_SCRIPTS_DIR/" || { echo -e " ${RED}✗${NC} Failed to copy hook scripts"; exit 1; } + chmod +x "$HOOKS_SCRIPTS_DIR"/*.sh + + if [ "$AGENT" = "claude" ]; then + # Claude Code: hooks live in .claude/settings.json (the shared, + # committable project settings), referencing the scripts above via + # ${CLAUDE_PROJECT_DIR}. Merge (not overwrite) to preserve any + # existing project hooks. + CLAUDE_SETTINGS="$PROJECT_DIR/.claude/settings.json" + SF_CLAUDE_SETTINGS="$CLAUDE_SETTINGS" SF_CLAUDE_TEMPLATE="$HOOKS_DIR/claude-settings.json" python3 - <<'PYEOF' || { echo -e " ${RED}✗${NC} Failed to write .claude/settings.json"; exit 1; } +import json, os + +settings_path = os.environ["SF_CLAUDE_SETTINGS"] +template_path = os.environ["SF_CLAUDE_TEMPLATE"] + +with open(template_path) as f: + template = json.load(f) + +if os.path.isfile(settings_path): + with open(settings_path) as f: + try: + settings = json.load(f) + except json.JSONDecodeError: + raise SystemExit(f"Existing {settings_path} is not valid JSON; refusing to overwrite.") +else: + settings = {} + +settings.setdefault("hooks", {}) + +def command_of(handler): + return (handler or {}).get("command", "") + +for event, groups in template["hooks"].items(): + existing_groups = settings["hooks"].setdefault(event, []) + existing_cmds = { + command_of(h) + for g in existing_groups + for h in g.get("hooks", []) + } + for group in groups: + new_handlers = [h for h in group.get("hooks", []) if command_of(h) not in existing_cmds] + if new_handlers: + merged = dict(group) + merged["hooks"] = new_handlers + existing_groups.append(merged) + +with open(settings_path, "w") as f: + json.dump(settings, f, indent=2) + f.write("\n") +PYEOF + echo -e " ${GREEN}✓${NC} hooks → $CLAUDE_SETTINGS (+ scripts)" + else + # Copilot CLI: hooks.json + scripts under .github/hooks/ + PROJECT_HOOKS_DIR="$PROJECT_DIR/.github/hooks" + [ -L "$PROJECT_HOOKS_DIR/hooks.json" ] && rm "$PROJECT_HOOKS_DIR/hooks.json" + cp -f "$HOOKS_DIR/shadow-frog-hooks.json" "$PROJECT_HOOKS_DIR/hooks.json" || { echo -e " ${RED}✗${NC} Failed to copy hooks.json"; exit 1; } + echo -e " ${GREEN}✓${NC} hooks → $PROJECT_HOOKS_DIR" + fi + fi + + # --- Inject agent-context (default: yes) --- + if [ "$NO_CONTEXT" = false ] && [ -f "$CONTEXT_FILE" ]; then + echo "" + echo -e "${BLUE}Injecting agent-context into ${CONTEXT_TARGET#"$PROJECT_DIR"/}...${NC}" + INSTRUCTIONS_FILE="$CONTEXT_TARGET" + mkdir -p "$(dirname "$INSTRUCTIONS_FILE")" + + MARKER_START="" + MARKER_END="" + CONTEXT_CONTENT=$(cat "$CONTEXT_FILE") + + # Remove existing block if present (idempotent re-apply). + # Pass the path via the environment (not string interpolation) so + # project paths containing quotes don't break the Python source. + if [ -f "$INSTRUCTIONS_FILE" ]; then + SF_INSTRUCTIONS_FILE="$INSTRUCTIONS_FILE" python3 - <<'PYEOF' +import os, re +path = os.environ["SF_INSTRUCTIONS_FILE"] +with open(path) as f: + content = f.read() +pattern = r'\n*.*?\n*' +content = re.sub(pattern, '\n', content, flags=re.DOTALL).strip() +with open(path, 'w') as f: + f.write(content + '\n') +PYEOF + fi + + # Append marked block + { + [ -f "$INSTRUCTIONS_FILE" ] && [ -s "$INSTRUCTIONS_FILE" ] && echo "" + echo "$MARKER_START" + echo "$CONTEXT_CONTENT" + echo "$MARKER_END" + } >> "$INSTRUCTIONS_FILE" + echo -e " ${GREEN}✓${NC} agent-context → $INSTRUCTIONS_FILE" + fi + + echo "" + echo -e "${GREEN}Project install complete!${NC}" + echo "" + echo "Next steps:" + echo " 1. Commit and push so future agent sessions find the skills:" + echo " cd $PROJECT_DIR" + if [ "$AGENT" = "claude" ]; then + GIT_ADD_PATHS=".claude/skills/" + [ "$NO_HOOKS" = false ] && GIT_ADD_PATHS="$GIT_ADD_PATHS .claude/settings.json .claude/hooks/" + [ "$NO_CONTEXT" = false ] && GIT_ADD_PATHS="$GIT_ADD_PATHS CLAUDE.md" + else + GIT_ADD_PATHS=".github/skills/" + [ "$NO_HOOKS" = false ] && GIT_ADD_PATHS="$GIT_ADD_PATHS .github/hooks/" + [ "$NO_CONTEXT" = false ] && GIT_ADD_PATHS="$GIT_ADD_PATHS .github/copilot-instructions.md" + fi + echo " git add $GIT_ADD_PATHS" + echo " git commit -m 'Add ShadowFrog skills, hooks, and context'" + echo "" + echo " 2. Run /shadow-frog-init in your agent session to create the shadow" + echo "" + echo " 3. To run dream via delegate:" + echo " /delegate Run /shadow-frog-dream on this codebase" + echo "" + exit 0 +fi diff --git a/pytest.ini b/pytest.ini new file mode 100644 index 0000000..f405d85 --- /dev/null +++ b/pytest.ini @@ -0,0 +1,6 @@ +[pytest] +testpaths = tests +addopts = -ra --strict-markers --tb=short +markers = + slow: tests that touch real git or subprocesses (deselect with -m "not slow") + integration: end-to-end tests that exercise full CLI paths diff --git a/requirements-dev.txt b/requirements-dev.txt new file mode 100644 index 0000000..d1e23ed --- /dev/null +++ b/requirements-dev.txt @@ -0,0 +1,7 @@ +# Test and development dependencies for ShadowFrog. +# Runtime ShadowFrog skills themselves have no Python dependencies beyond +# the standard library; this file is what `tests/` and CI need. + +pytest>=9.1.0 +pytest-cov>=7.1.0 +pathspec>=1.1.1 diff --git a/shadow_repo.png b/shadow_repo.png new file mode 100644 index 0000000..26b8a07 Binary files /dev/null and b/shadow_repo.png differ diff --git a/skills/shadow-frog-dream/SKILL.md b/skills/shadow-frog-dream/SKILL.md new file mode 100644 index 0000000..237fea3 --- /dev/null +++ b/skills/shadow-frog-dream/SKILL.md @@ -0,0 +1,1203 @@ +--- +name: shadow-frog-dream +description: >- + Run autonomous experimentation while the user is AFK. Uses 6 investigation + categories (investigation, bug hunting, feature design, refactoring, + optimization, security audit) to systematically discover non-obvious + behaviors. Every task is an experiment — implement in worktrees, commit + to persistent dream branches, and push to the fork. Dreams compound + across sessions: future experiments branch from prior dream branches, + building a tree of progressively deeper work. Invoke when the user is + AFK or asks for a dream run. +scripts: + - dream-coverage.py + - dream-validate.py + - dream-reconcile.py + - dream-setup.sh + - dream-cleanup.sh + - dream-gc.sh +--- + +# ShadowFrog Dream + +Autonomous experimentation while the user is away. Every task is an +**experiment** — implement real code in a worktree, run it, persist as a +**named git branch** pushed to the fork. Dream's unique value is +implementation experience that **compounds across sessions**. + +## Critical Invariants + +These rules are stated ONCE here and enforced by helper scripts. Violating +any of them is a completion criteria failure. + +### Prerequisite: `.shadow/` must be git-tracked + +Dream moves `.shadow/` content **through git** — artifacts are committed onto +the dream branch, pushed to the remote, then read back by the reconciler via +`git show origin/ .shadow/...`. If `.shadow/` is gitignored (the +"local only" option in `shadow-frog-init`), `git add -A` silently skips those +files, nothing reaches the remote, and the reconciler finds no manifest — +**every discovery is lost without warning**. `dream-setup.sh` runs +`git check-ignore .shadow` up front and refuses to start if it's ignored. +Use `shadow-frog-update` instead for local-only shadows. + +### Path Isolation + +``` +WORKTREE_BASE = /tmp/shadowfrog-dreams// +WORKTREE_DIR = $WORKTREE_BASE/dream- +``` + +- Worktrees are ALWAYS in `/tmp/shadowfrog-dreams//`, NEVER in + the project directory. This prevents conflicts between parallel agents + and keeps the main repo clean. +- `DREAM_NS` (namespace) isolates branches per task/instance. Resolved + from: `DREAM_NAMESPACE` env → `TASK_INFO.json` → `.env` → repo basename. +- Override only with `DREAM_WORKTREE_BASE` env var if `/tmp` is too small. +- `dream-setup.sh` computes and enforces all paths. Use it. + +### Branch Naming + +``` +BRANCH_NAME = dream// +DREAM_ID = YYYYMMDD-HHMMSSZ- +``` + +### Artifact Format + +``` +.shadow/_dreams//report.md +.shadow/_dreams//manifest.json +.shadow/_dreams//patch.diff +``` + +NEVER flat files (`_dreams/.md`). Flat files break the pipeline. + +### RUN_PREFIX + +All python/pytest commands MUST use the `RUN_PREFIX` resolved during +preflight. When `RUN_PREFIX="uv run"`, use `$RUN_PREFIX python3 ...`. +Bare `python3` or `pytest` without prefix is a violation when non-empty. + +### Reconciliation is Mandatory + +Every dream branch must reconcile to main before the session ends. The +most common failure mode is agents pushing dream branches but never +reconciling — losing all discoveries. + +**Two modes:** +- **Parallel mode (default):** Launch a batch of 3-4 sub-agents → wait + for all to push → run `dream-reconcile.py "$REPO_ROOT"` ONCE at the end + of the batch. The reconciler auto-discovers every un-reconciled dream + branch in the namespace — you do NOT pass branch names. One call merges + every pushed branch. +- **Sequential mode (fallback when sub-agents unavailable):** + Complete dream → push → reconcile → verify → next dream. Adds ~30s + per dream but guarantees zero data loss if the session crashes + mid-batch. + +Never queue multiple un-reconciled batches; reconcile at the end of +each batch or each individual dream. + +### Script Failure Recovery + +All helper scripts (`dream-setup.sh`, `dream-reconcile.py`, +`dream-validate.py`) are **self-documenting**. If a script fails or is +unavailable: **read the script source**, understand what it does, and adapt +its logic manually for your situation. Never skip steps just because a +script errored — the steps still need to happen. + +``` +Fork repo (user/target-repo) + main --- .shadow/ (accumulated ALL discoveries) + | + +-- dream// (cycle 1, agent A, from main) + +-- dream// (cycle 1, agent B, from main) + +-- dream// (cycle 2, from dream//, compounding) +``` + +- Branches are live — `git checkout dream/` runs the code +- Shadow follows lineage — ancestor chain, not sibling branches +- Main is the accumulator — reconciliation merges ALL discoveries + +### Prerequisites + +1. Forked repo cloned locally with pushable remote +2. ShadowFrog skills installed (`install.sh --project /path/to/fork`) +3. `.shadow/` initialized (`/shadow-frog-init`) + +## Helper Scripts + +This skill bundles 6 helper scripts. Find them in the skill directory: + +```bash +SKILL_DIR="" +for DIR in .github/skills/shadow-frog-dream .claude/skills/shadow-frog-dream; do + [ -d "$DIR" ] && SKILL_DIR="$DIR" && break +done +``` + +| Script | Purpose | When to use | +|--------|---------|-------------| +| `dream-setup.sh` | Creates worktree + branch with namespace isolation | **Phase 3** — start of every experiment | +| `dream-validate.py` | Validates artifacts before push (hard gate) | **Phase 5** — before `git push` | +| `dream-reconcile.py` | Merges dream branches into main's `.shadow/` | **Phase 6** — after all experiments done | +| `dream-coverage.py` | Computes exploration coverage map | **Phase 2** — task planning for diversity | +| `dream-cleanup.sh` | Safely removes ONE dream worktree (with safety gate) | **After push** — replaces the old inline cleanup snippet | +| `dream-gc.sh` | Sweeps orphan dream worktrees from `$DREAM_WORKTREE_BASE` | **Auto** — triggered by `dream-setup.sh` (per-namespace throttle, default 1× / hour) in orphan-only mode; also `--task-complete --namespace "$DREAM_NS" --min-age-min 0` for end-of-session sweep of registered-but-stale dirs | + +**Usage patterns:** + +```bash +# Setup: creates worktree, prints export vars. +# IMPORTANT: capture the output FIRST, then eval it. Writing +# `eval "$(dream-setup.sh ...)" || exit 1` does NOT catch failures: if the +# command substitution exits non-zero and prints nothing, `eval ""` still +# succeeds (exit 0) and the agent silently proceeds with empty env vars. +# Assigning to a variable makes `|| exit 1` fire on the script's real exit code. +SETUP_OUT="$("$SKILL_DIR/dream-setup.sh" --slug t01-my-experiment)" || exit 1 +eval "$SETUP_OUT" +# → exports (keep in sync with dream-setup.sh emit_export block): +# REPO_ROOT, DEFAULT_BRANCH, DREAM_NS, DREAM_ID, BRANCH_NAME, PARENT_BRANCH, +# WORKTREE_DIR, WORKTREE_BASE, BASE_COMMIT, RUN_PREFIX, SLUG + +# Validate: hard gate before push +python3 "$SKILL_DIR/dream-validate.py" "$DREAM_ID" "$WORKTREE_DIR" + +# Reconcile: merge all dream branches into main +python3 "$SKILL_DIR/dream-reconcile.py" "$REPO_ROOT" +# After `git push` succeeds, optionally clean up reconciled branches. +# Cleanup REFUSES to run if `.shadow/` has uncommitted changes, or unless +# HEAD is already on origin/ — so the canonical flow is: +# reconcile → git add .shadow/ && git commit && git push → re-run --cleanup-branches +python3 "$SKILL_DIR/dream-reconcile.py" "$REPO_ROOT" --cleanup-branches + +# Coverage: show which files still need exploration +python3 "$SKILL_DIR/dream-coverage.py" "$REPO_ROOT" + +# All scripts support --help. +``` + +If a script is not found or fails, read its source — they are +self-documenting. Adapt the steps manually if needed (see each phase for +inline fallback instructions). + +## Phase 1: Preflight and Assess + +### Preflight Validation + +```bash +REPO_ROOT=$(git rev-parse --show-toplevel) +cd "$REPO_ROOT" + +# 0. Auto-detect DREAM_NAMESPACE +if [ -z "${DREAM_NAMESPACE:-}" ]; then + if [ -f TASK_INFO.json ]; then + DREAM_NAMESPACE=$(python3 -c "import json; print(json.load(open('TASK_INFO.json')).get('dream_namespace',''))" 2>/dev/null) + export DREAM_NAMESPACE + elif [ -f .env ]; then + DREAM_NAMESPACE=$(grep '^DREAM_NAMESPACE=' .env | head -1 | cut -d'=' -f2-) + export DREAM_NAMESPACE + fi +fi +[ -n "${DREAM_NAMESPACE:-}" ] && echo "Dream namespace: $DREAM_NAMESPACE" + +# 1. Detect default branch +DEFAULT_BRANCH=$(git symbolic-ref refs/remotes/origin/HEAD 2>/dev/null | sed 's|refs/remotes/origin/||') +if [ -z "$DEFAULT_BRANCH" ]; then + if git show-ref --verify refs/remotes/origin/main >/dev/null 2>&1; then + DEFAULT_BRANCH="main" + elif git show-ref --verify refs/remotes/origin/master >/dev/null 2>&1; then + DEFAULT_BRANCH="master" + else + echo "ERROR: Cannot detect default branch. Fix: git remote set-head origin " + fi +fi +echo "Default branch: $DEFAULT_BRANCH" + +# 2-5. Validate environment +echo "Current branch: $(git branch --show-current)" +echo "Remote: $(git remote get-url origin)" +git ls-remote origin HEAD >/dev/null 2>&1 || echo "ERROR: Cannot reach remote." +[ -d .shadow ] || echo "ERROR: .shadow/ not found. Run /shadow-frog-init first." +mkdir -p .shadow/_dreams +git diff --quiet && git diff --cached --quiet || echo "ERROR: Uncommitted changes." + +# 6. Fetch all remote branches (ONE fetch for all agents) +git fetch origin --prune + +# 7. List dream branches (namespace-filtered) +DREAM_NS="${DREAM_NAMESPACE:-}" +BRANCH_PATTERN="${DREAM_NS:+origin/dream/${DREAM_NS}/}" +BRANCH_PATTERN="${BRANCH_PATTERN:-origin/dream/}" +echo "Available dream branches:" +git branch -r | grep "$BRANCH_PATTERN" | sed 's|origin/||' || echo " (none)" + +# 8. Detect RUN_PREFIX from lockfiles +if [ -f uv.lock ]; then + uv sync --all-groups 2>&1 | tail -3 + RUN_PREFIX="uv run" +elif [ -f package-lock.json ]; then + npm install --quiet 2>&1 | tail -3 + RUN_PREFIX="npx" +elif [ -f yarn.lock ]; then + yarn install --silent 2>&1 | tail -3 + RUN_PREFIX="npx" +else + RUN_PREFIX="" +fi +echo "RUN_PREFIX='$RUN_PREFIX'" + +# 9. List compoundable experiments +echo "" +echo "=== COMPOUNDABLE EXPERIMENTS ===" +if [ -f .shadow/_dreams/_index.md ]; then + awk -F'|' 'NR>2 && /useful/ { + gsub(/ /,"",$2); gsub(/ /,"",$4); gsub(/ /,"",$6); + gsub(/^ +| +$/,"",$5); + if ($2 != "" && $6 != "") print $6 " | " $3 " | " $5 + }' .shadow/_dreams/_index.md + COMPOUNDABLE=$(awk -F'|' 'NR>2 && /useful/ {gsub(/ /,"",$2); if ($2 != "") c++} END {print c+0}' .shadow/_dreams/_index.md) + echo "Total compoundable: $COMPOUNDABLE" +else + echo "(none — first dream session)" +fi +``` + +**If any check prints ERROR, STOP.** Do not use `exit 1` — check output +and stop at the agent level. + +(`RUN_PREFIX` MUST be threaded into every subagent prompt — see Critical +Invariants above. Bare `python3`/`pytest` without prefix = completion +criteria violation.) + +### Snapshot Branch State + +After the single `git fetch`, capture dream branches and pass to all +sub-agents — they do NOT fetch independently. + +```bash +DREAM_NS="${DREAM_NAMESPACE:-}" +BRANCH_FILTER="${DREAM_NS:+origin/dream/${DREAM_NS}/}" +BRANCH_FILTER="${BRANCH_FILTER:-origin/dream/}" +git branch -r --format='%(refname:short) %(objectname:short)' \ + | grep -F "$BRANCH_FILTER" \ + | sed 's|origin/||' > .shadow/_dreams/.branch-map.txt +cat .shadow/_dreams/.branch-map.txt + +# Initialize session tracking (orchestrator-only; agents do NOT write here) +: > .shadow/_dreams/.session-branches.txt +``` + +### Assess Codebase + +Read `_meta/state.json`, `_index.md`, existing discoveries, and **past +dream reports** in `_dreams/`. + +### Build Exploration Coverage Map + +File-level coverage breadth is the strongest predictor of dream success +(r²=0.63 vs bugs found), NOT dream count (r²=0.04). + +```bash +# Find and run the coverage script +COVERAGE_SCRIPT="" +for DIR in .github/skills/shadow-frog-dream .claude/skills/shadow-frog-dream; do + [ -f "$DIR/dream-coverage.py" ] && COVERAGE_SCRIPT="$DIR/dream-coverage.py" && break +done +[ -n "$COVERAGE_SCRIPT" ] && python3 "$COVERAGE_SCRIPT" "$REPO_ROOT" || echo "WARNING: dream-coverage.py not found" +``` + +**Coverage definition:** A file is "covered" only when its shadow has +≥1 behavioral discovery (line starting with `- `). Placeholder-only = NOT covered. + +**Scoped exploration (`--scope`)** — pass `--scope ` (repeatable) +to restrict the coverage map to a specific subtree. Use this when the +broader repo is well-explored but a particular area (e.g., a known +frontier of bugs, a newly-added module, a subsystem the user just +flagged) deserves a focused dream session. All counts (totals, %, +saturated, fan-in, per-dir) are computed over the scoped subset only. + +```bash +# Scope to one subtree +python3 "$COVERAGE_SCRIPT" "$REPO_ROOT" --scope src/auth/ + +# Scope to multiple subtrees in one pass +python3 "$COVERAGE_SCRIPT" "$REPO_ROOT" --scope src/auth/ --scope src/db/ +``` + +When using `--scope`, the per-category task quotas (Phase 2) still apply +but are interpreted against the scoped subset. Don't use scoped +exploration as the default — pick it only when there's a concrete reason +to concentrate effort. Unscoped diversity remains the strongest +predictor of useful discoveries. + +### Review Past Dreams (Required) + +When compoundable experiments exist (preflight step 9): +- Read each report: `cat .shadow/_dreams//report.md` +- Choose which to continue (extending, fixing, integrating) +- Note `dead_end` experiments to avoid repeating +- Trace lineage via the `parent` column in `_dreams/_index.md` + +**Compounding quality gate** — before choosing to compound from a parent: + +1. Read the parent's `report.md` AND `manifest.json` +2. Verify the parent has a non-empty `patch.diff` (prose-only parents + are low-value — prefer parents with working code) +3. Identify at least one specific file or function you plan to modify/extend +4. Check the parent's area isn't saturated (8+ discoveries) — if it is, + start fresh from main unless you have a concrete new angle +5. Log your compounding intent: "I will extend parent's retry logic in + `src/http.py` to handle connection timeouts" — vague "continue + exploring" is NOT compounding + +**First dream session:** if preflight step 9 shows `(none)`, all tasks +branch from main. + +## Phase 2: Plan + +Generate a concrete plan. **Target 12 tasks (2 per category).** On small +codebases (<30 source files), minimum 6 tasks across 4+ categories. + +### The 6 Investigation Categories + +| Category | What to look for | Priority signals | +|----------|-----------------|-----------------| +| **Investigation** | Under-explored files, shallow coverage, uncertain discoveries | Files with 0-2 discoveries, import chains not traced, `uncertain` entries | +| **Bug hunting** | Defects, edge cases, race conditions | Error-handling code, concurrency, unvalidated inputs | +| **Feature design** | New capabilities, missing functionality | TODOs, FIXMEs, user-facing gaps, integration opportunities | +| **Refactoring** | Structural improvements, duplication | God classes, copy-paste patterns, high-coupling files | +| **Optimization** | Algorithmic efficiency, performance | Hot paths, nested loops, repeated I/O, missing caches | +| **Security audit** | Vulnerabilities, unsafe patterns | Auth code, data handling, deserialization, user inputs | + +| Category | What to experiment | +|----------|-------------------| +| **Investigation** | Write assertion-based tests proving/disproving behavior hypotheses | +| **Bug hunting** | Fuzz inputs, trigger error paths, reproduce race conditions | +| **Feature design** | Implement the feature, run it, evaluate integration | +| **Refactoring** | Do the refactor, run existing tests, measure complexity | +| **Optimization** | Benchmark, profile, implement optimization, measure before/after | +| **Security audit** | Craft adversarial inputs, test injection vectors (local only) | + +**Exception — user-directed focus**: If the user specifies a focus area +(e.g., "dream focus on security"), allocate ALL tasks to that category. + +### Task Plan Format + +Each task specifies **base branch**, **primary target file(s)**, and **why**: + +``` +Tasks (by category): + Investigation: + 1. [title] — write tracing tests for [target] + Base: main + Target: src/auth/validator.py (UNCOVERED, 12 refs) + Why: High fan-in utility with no shadow coverage + Bug hunting: + 1. [title] — fuzz [target] + Base: dream// (compounds prior) + Target: src/parsers/csv.py (extending parent's failing tests) + Why: Parent found 2 crashes, need to verify fixes + ... +``` + +### Diversity Rules + +Prevent fixation (exploring the same files while leaving most untouched): + +1. **Max 2 tasks per source file** (unless prior dream left concrete follow-up) +2. **≥30% of tasks on uncovered files** (from coverage map) +3. **≥2 tasks on "deep" files** (utilities, internals, converters) +4. **Vary directories** — no 3+ consecutive tasks in same dir + +**Self-check before finalizing:** unique target files ≥ 60% of task count, +uncovered file tasks ≥ 30%, no file in > 2 tasks. Swap if failing. + +**Escape hatches** (document justification): prior dream's failing test, +concrete untested hypothesis, file is 500+ lines with unexplored sections, +codebase has <20 source files. + +### Task Design + +Each task needs: **category**, **hypothesis**, **what to implement**, +**base branch**, **primary target** (with coverage status), **why this +target**, **scope** (hours, not days), and **success criteria**. + +Good examples (one per category): +- **Investigation**: "Write assertion harness for request lifecycle — + instrument each layer to log entry/exit and reveal implicit contracts" +- **Bug hunting**: "Fuzz the CSV parser with malformed inputs — what + crashes or silently corrupts?" +- **Feature design**: "Implement retry logic with exponential backoff — + does it handle transient failures without masking permanent ones?" +- **Refactoring**: "Extract 5 duplicate auth checks into middleware — + run tests, measure if it simplifies without breaking special cases" +- **Optimization**: "Benchmark the hot path, implement LRU cache for + repeated lookups — measure before/after wall time" +- **Security audit**: "Craft SQL injection payloads for user-facing + endpoints — does parameterized query hold under nested quotes?" + +Bad examples: "Look at the code", "Trace the flow", "Review error +handling", "Improve code quality" + +### Feature Design: Motivation Required + +Feature experiments must address a real gap identified in existing code. +Answer: "Why would maintainers want this?" with a specific code reference. +The feature must connect to the existing codebase (imports, modifies, +replaces duplication). Standalone modules with only stdlib don't qualify. + +### Vary Your Approach + +Each experiment should have unique structure driven by its hypothesis. If +you find yourself copying the same module layout (one source file + one +test file, identical importlib hack) across experiments, you're optimizing +for throughput over insight. Vary your approach: some experiments modify +existing files, some add tests for existing code, some create minimal +scripts, some refactor existing modules. + +### File Selection Guidance + +Agents gravitate toward entry points. Evaluation shows this causes missed bugs. + +**High-value targets typically missed:** +- High fan-in files (imported by many, rarely explored directly) +- Internal/private modules (`_internal/`, `_utils/`, `_compat/`) +- Conversion/serialization code (parse, encode, format, marshal) +- Error handling paths (exception hierarchies, fallback logic) + +**Avoid:** Starting from `__init__.py`, skipping "boring" files, same +directory 3+ times, ignoring files with few public symbols. + +## Phase 3: Execute Tasks + +Work through the plan. **Launch 3-4 experiments in parallel** via +sub-agents. Each handles the full lifecycle: create worktree → implement +→ test → write shadow + manifest + report → commit → push → clean up. +If sub-agents are unavailable, fall back to sequential execution. + +**Each experiment runs in a separate git worktree.** The worktree IS the +dream branch (created with `-b`). After pushing, the worktree is removed +but the branch persists on the remote. + +### Reading Before Implementing + +You must understand the code before changing it. For each task: + +1. Read the source file(s) and their shadows (existing discoveries) +2. Read shadows of referenced/referencing files +3. Understand the current behavior, edge cases, and implicit contracts + +Reading is *preparation*, not the deliverable. The deliverable is code +written, code run, results recorded. + +### Experiment Setup + +Use `dream-setup.sh` to create worktrees (handles all path computation, +namespace resolution, worktree creation, and validation): + +```bash +# Find the setup script +SETUP_SCRIPT="" +for DIR in .github/skills/shadow-frog-dream .claude/skills/shadow-frog-dream; do + [ -f "$DIR/dream-setup.sh" ] && SETUP_SCRIPT="$DIR/dream-setup.sh" && break +done + +# Fresh experiment from main: +SETUP_OUT="$("$SETUP_SCRIPT" --slug t01-csv-fuzzer)" || exit 1 +eval "$SETUP_OUT" + +# Compounding from prior dream: +SETUP_OUT="$("$SETUP_SCRIPT" --slug t03-extend --base-branch dream//)" || exit 1 +eval "$SETUP_OUT" +``` + +Capture into `SETUP_OUT` first, then `eval` it — see Helper Scripts § +usage patterns (above) for why bare `eval "$(…)" || exit 1` silently +swallows the script's exit code. + +This exports (keep in sync with `dream-setup.sh`): `REPO_ROOT`, +`DEFAULT_BRANCH`, `DREAM_NS`, `DREAM_ID`, `BRANCH_NAME`, `PARENT_BRANCH`, +`WORKTREE_DIR`, `WORKTREE_BASE`, `BASE_COMMIT`, `RUN_PREFIX`, `SLUG`. + +**If `dream-setup.sh` fails or is not found:** Apply the Script Failure +Recovery rule (read the script source, adapt its logic). Common causes: +missing git remote, branch already exists, `/tmp` permissions. + +**Note:** Shell variables don't persist across tool calls. Either run +multi-step setup in a single shell, or re-derive values. From inside a +worktree, get main repo with: +`git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel` + +**If worktree creation fails:** mark task `blocked`, replace with another. + +### What Meaningful Compounding Looks Like + +Compounding means **actively engaging with the parent's code**, not just +sitting on its branch. Valid compounding approaches: +- **Extend**: import or call the parent's modules and build on them +- **Modify**: edit the parent's code to fix limitations noted in its report +- **Refactor**: restructure the parent's implementation for better design +- **Integrate**: wire the parent's standalone module into the real codebase +- **Test deeper**: add edge-case tests for the parent's implementation + +Don't assume the parent dream's code is complete or frozen — iterative +improvement is the whole point. If you can't find anything meaningful to +build on, start fresh from main instead. + +Compounding that only adds a new standalone module beside the parent's +code (with no imports, edits, or integration) is NOT compounding — it's +a fresh experiment on the wrong branch. + +### Run + +1. Implement the experiment — write real code, run tests/builds +2. Note what worked, broke, surprised +3. Debug if needed — the struggle produces the best discoveries +4. Record results as you go + +### Write Shadow Discoveries + +**On the dream branch** (in the worktree), NOT on main. + +**Every experiment MUST write ≥1 discovery to a per-file `.shadow/*.md`.** +Authoring order: write the human-readable shadow first, then mirror every +discovery into `manifest.json`. For reconciliation the **manifest is the +source of truth** — the reconciler merges manifest entries into main, so a +discovery that is missing from the manifest never reaches main. The per-file +shadow is the human-readable copy (and a required validate gate), not the +propagation path. + +Follow the dedup and writing rules in `/shadow-frog`. Dream discoveries +are typically `source: exploration`. Mark `verified` when confirmed by +running code; `uncertain` if not fully testable. + +#### How to Append + +Find the `##`/`###` heading for the symbol, then: +- Placeholder `_No discoveries yet._` → replace with discovery +- Existing discoveries → append after last bullet +- No heading → create before `## Cross-References` + +#### Label Triage (REQUIRED) + +After writing each discovery, evaluate whether it deserves any of the +five actionable labels from `/shadow-frog` (`bug`, `security`, +`performance`, `feature-gap`, `tech-debt`). Apply labels when: + +| Label | Apply when the discovery describes... | +|-------|---------------------------------------| +| `bug` | A defect, silent failure, off-by-one, race, incorrect result, edge case that misbehaves, validate-then-use ordering hazard | +| `security` | Injection vector, unsafe default, missing auth/authz check, sensitive value logged, untrusted input reaching unsafe sink | +| `performance` | Measured bottleneck, O(N²) where N is large, repeated I/O that could batch, missing cache, blocking call on hot path | +| `feature-gap` | Missing capability the codebase clearly needs, asymmetric API (e.g., reads but no writes) | +| `tech-debt` | Duplication, dead code, leaky abstraction, vestigial parameter, inconsistent naming | + +Rules: +- Apply labels to BOTH the in-file discovery markdown AND the + `manifest.json` discovery entry (`"labels": ["bug"]`). The reconciler + uses the manifest as source of truth; the in-file copy is for humans + reading the shadow directly. +- Multiple labels are fine when accurate: `labels: [bug, security]`. +- Omit labels for pure behavioral observations ("retries N times before + giving up", "default timeout is 30s") — these are knowledge, not + action items. +- Do not apply labels speculatively. The label says "an engineer should + act on this." If you wouldn't act on it, don't label it. + +Examples: + +``` +- /api/upload accepts paths from request body without normalization, + allowing `../` traversal into /etc/. + _(verified, source: exploration, labels: [bug, security])_ + Dream report: `_dreams/20260518-161200Z-upload-traversal/` +``` + +``` +- HttpClient.send retries 3x on transient failures. + _(verified, source: exploration)_ + Dream report: `_dreams/20260518-163000Z-retry-audit/` +``` +(No label — pure behavioral knowledge, no action implied.) + +`dream-validate.py` emits non-blocking warnings when discovery text +contains label-signal keywords but no label is set. Treat those +warnings as a prompt to re-check the triage, not as a directive. + +#### Anchor Rules + +- About existing code → anchor to that symbol +- Spans 3+ files → `_cross/.md` +- Project-wide convention → `_prefs.md` +- **Only create shadows for base-codebase files** — experiment-only files + don't get shadows (the branch IS the artifact). Anchor findings to the + existing code they relate to. + +#### Cross-Cutting Discoveries + +When you see the same behavior in 3+ files, create a `_cross/.md` +rather than repeating the discovery in each per-file shadow. Add +back-pointers in each file's `## Cross-References` section. + +#### Discovery Quality + +Discoveries must be **self-contained process knowledge** — understandable +with just the base codebase. Someone reading main's shadow should understand +the insight without checking out the dream branch. Capture **how to do it**, +**what you learned**, and **what to avoid** — not what was built. The branch +preserves the artifact; the shadow preserves the wisdom. + +Good (behavioral insights about existing code): +- "functools.lru_cache is not thread-safe for initialization — two + threads can trigger duplicate expensive computations on first call." +- "agent.py's retry loop catches all exceptions including OOM, masking + fatal errors that should crash immediately." +- "To add a new eval metric, register in METRIC_MAP at metrics.py:25 + and implement the Metric interface — missing either causes a silent + no-op in the pipeline." + +Bad (descriptions of new code): +- "The implemented PluginFramework has PluginRegistry, PluginManager, + and 7 lifecycle hooks." — describes branch-only artifact. +- "Provides RewardShaper with 4 methods, Welford normalizer, and + GAE-lambda estimation." — feature spec, not behavioral insight. +- "Complete tested module with 58 passing tests." — verdict, not discovery. + +Per-file discoveries should reference the dream report: + +``` +- Retrying with exponential backoff recovers from 99% of transient errors, + but must exclude 4xx or it retries bad requests for 30s. + _(verified, source: exploration)_ + Dream report: `_dreams/20250612-143012Z-retry-logic/` +``` + +### Write Discovery Manifest + +After shadow writes, create `.shadow/_dreams/$DREAM_ID/manifest.json`: + +```json +{ + "dream_id": "", + "branch": "", + "parent_branch": "main", + "category": "bug hunting", + "verdict": "useful", + "title": "CSV Parser Edge Cases", + "discoveries": [ + { + "op": "add", + "anchor": "src/parsers/csv.py::parse_row", + "text": "Unescaped quotes in fields cause silent truncation.", + "status": "verified", + "source": "exploration", + "labels": ["bug"], + "also_involves": ["src/parsers/utils.py::unescape"], + "dream_report": "_dreams//" + } + ], + "cross_cutting": [] +} +``` + +**Anchor format:** `file::symbol` with bare names (no backticks). The +reconciler handles normalization. + +Manifest `op` values: only `add` is supported by the reconciler today. +`update` and `refute` are reserved keywords — `dream-validate.py` will +reject any discovery whose `op` is not `add`. To revise or contradict an +existing discovery, run a meditate session against main's `.shadow/` +instead of trying to do it from a dream branch. + +**Hard gate — discoveries must be mirrored into per-file shadows.** The +reconciler merges `manifest.json` entries into main directly (so discoveries +are not lost at merge time), but the branch's per-file shadows must ALSO be +updated so human PR reviewers can read the discoveries in context. If +`manifest.json` declares discoveries but no `.shadow/*.md` files outside +`_dreams/` are modified in the branch diff vs `base_commit`, +`dream-validate.py` rejects the dream. Always write each discovery into BOTH +the corresponding per-file shadow (or `.shadow/_cross/`) AND the manifest +before staging. + +### Save Dream Report + +Save as `.shadow/_dreams/$DREAM_ID/report.md`: + +```markdown +--- +dream_id: "" +category: bug hunting +verdict: useful +base_commit: "" +branch: "" +parent_branch: "main" +remote: "origin" +related_symbols: + - "src/parsers/csv.py::parse_row" +builds_on: [] +--- + +# CSV Parser Edge Cases + +## Motivation + + +## Compounding Delta + + +## Hypothesis + + +## Implementation + + +## Commands Run + + +## Evaluation + + +## Takeaways + + +## Verdict Details + +``` + +| Field | Required | Values | +|-------|----------|--------| +| `dream_id` | yes | `YYYYMMDD-HHMMSSZ-slug` | +| `category` | yes | one of the 6 categories | +| `verdict` | yes | `useful` or `dead_end` | +| `base_commit` | yes | SHA branched from | +| `branch` | yes | full branch name | +| `parent_branch` | yes | `main` or prior branch path | +| `related_symbols` | yes | `file::symbol` refs | + +**`tip_commit` is NOT in the report.** Including the final commit SHA +creates a chicken-and-egg problem (SHA changes when report is committed). +The reconciler derives it via `git rev-parse origin/$BRANCH` and records +it in `_dreams/_index.md`. + +**Verdict** is the agent's assessment (set once, immutable): +- `useful` — produced actionable findings, working code, or valuable lessons +- `dead_end` — approach doesn't work; documented why so future dreams skip + +### Validate, Commit, Push + +```bash +cd "$WORKTREE_DIR" + +# 1. Generate diff (exclude .shadow/ and common build artifacts) +git add -A -- ':!.dream_parent' ':!__pycache__/' ':!.pytest_cache/' +git commit -m "dream: $SLUG" +mkdir -p .shadow/_dreams/"$DREAM_ID" +git diff "$BASE_COMMIT" HEAD -- \ + ':!.shadow/' ':!__pycache__/' ':!*.pyc' ':!.pytest_cache/' \ + ':!node_modules/' ':!*.lock' ':!dist/' ':!build/' \ + > .shadow/_dreams/"$DREAM_ID"/patch.diff +[ ! -s .shadow/_dreams/"$DREAM_ID"/patch.diff ] && echo "WARNING: Empty diff" + +# 2. Validate (hard gate — must pass before push) +VALIDATE_SCRIPT="" +for DIR in .github/skills/shadow-frog-dream .claude/skills/shadow-frog-dream; do + [ -f "$DIR/dream-validate.py" ] && VALIDATE_SCRIPT="$DIR/dream-validate.py" && break +done +if [ -n "$VALIDATE_SCRIPT" ]; then + python3 "$VALIDATE_SCRIPT" "$DREAM_ID" "$WORKTREE_DIR" || { + # If validate fails: read the script to understand what checks failed, + # fix the issues it reports, then re-run. The script checks artifact + # structure, manifest schema, and report frontmatter. + echo "FIX ERRORS"; exit 1 + } +else + # Script not found — inline fallback (minimal checks) + [ ! -d ".shadow/_dreams/$DREAM_ID" ] && echo "ERROR: Missing dir" && exit 1 + for F in report.md manifest.json patch.diff; do + [ ! -f ".shadow/_dreams/$DREAM_ID/$F" ] && echo "ERROR: Missing $F" && exit 1 + done +fi + +# 3. Final commit and push +git add -A -- ':!.dream_parent' +git commit -m "dream: $SLUG — final with report and manifest" +if git push origin "$BRANCH_NAME"; then + echo "Pushed: $BRANCH_NAME" +else + echo "ERROR: Push failed. Keep worktree for recovery." + exit 1 +fi +``` + +**Do NOT write to `.session-branches.txt`** — that is managed by the +orchestrator after all agents complete. Agents only push their branch; +the orchestrator discovers pushed branches from the remote. + +### Worktree Cleanup + +```bash +bash "$SKILL_DIR/dream-cleanup.sh" "$WORKTREE_DIR" --repo-root "$REPO_ROOT" +``` + +`dream-cleanup.sh` does the equivalent of `git worktree remove --force` +followed by `git worktree prune`, but ALSO falls back to a safety-gated +`rm -rf` if `git worktree remove` silently fails — the failure mode that +leaked tens of dream worktrees per AFK session under the previous inline +snippet (see bug-worktree-leak.md). The rm fallback ONLY fires for paths +that match `${DREAM_WORKTREE_BASE:-/tmp/shadowfrog-dreams}//dream-` +exactly; any other path is refused. + +Remove as you go. If push failed, keep the worktree. + +### Mid-Session Diversity Check + +After completing roughly half of your planned tasks, pause and review: + +1. **Count unique primary target files** explored so far. If fewer than + 50% of completed tasks targeted distinct files, remaining tasks MUST + target new files. +2. **Check for re-exploration** — are any completed tasks exploring files + already well-covered before this session? Swap remaining tasks for + uncovered ones. +3. **Review coverage map delta** — if fewer than 2 previously uncovered + files explored, prioritize uncovered files for remaining tasks. +4. **Adjust the plan** — swap, add, or reorder remaining tasks. The plan + is a starting point, not a contract. + +This prevents the fixation failure mode where the first half discovers a +rich area and the second half keeps digging there instead of spreading. + +## Phase 4: AFK-Safe Patterns + +1. Worktrees are outside the repo — writes don't trigger approval +2. Temp scripts go in `/tmp/shadow-dream-.*` +3. Never modify main directly — only during reconciliation +4. Clean up worktrees after push +5. Shadow writes on dream branches are safe + +## Phase 5: Parallel Agent Rules + +1. Each agent targets different files (orchestrator assigns non-overlapping sets) +2. Each agent gets its own branch (inherently isolated) +3. Each agent writes its own manifest in its `$DREAM_ID/` directory +4. Do NOT write to main or shared files (`_index.md`, `state.json`) +5. Do NOT update metadata — reconciled post-dream by orchestrator +6. Fetch once, branch from Phase 1 snapshot (no independent fetches) +7. Manifest anchors use bare symbol names (reconciler normalizes) +8. Thread `RUN_PREFIX` into every subagent prompt +9. Include `WORKTREE_BASE` and `DREAM_NS` in every subagent prompt +10. Dream artifacts MUST use subdirectory format — flat files are a + completion criteria violation (see Critical Invariants → Artifact Format) + +## Phase 6: Reconcile to Main + +**⚠️ CRITICAL: Reconciliation is MANDATORY at the end of every dream batch.** +See Critical Invariants → Reconciliation is Mandatory (above) for the +parallel-vs-sequential mode definitions and the auto-discover rule. Do +NOT defer reconciliation across batches. + +The `_index.md` entry is your sequential-mode checkpoint — any dream +listed there is safe if the session crashes. + +Use the reconciliation script: + +```bash +cd "$REPO_ROOT" +git checkout "$DEFAULT_BRANCH" +git fetch origin --prune + +# Find and run the reconciliation script +RECONCILE_SCRIPT="" +for DIR in .github/skills/shadow-frog-dream .claude/skills/shadow-frog-dream; do + [ -f "$DIR/dream-reconcile.py" ] && RECONCILE_SCRIPT="$DIR/dream-reconcile.py" && break +done + +if [ -n "$RECONCILE_SCRIPT" ]; then + python3 "$RECONCILE_SCRIPT" "$REPO_ROOT" +else + echo "WARNING: dream-reconcile.py not found. Apply Script Failure Recovery: read dream-reconcile.py source, adapt its 9 steps manually." +fi +``` + +**If the reconciler script fails or errors:** Read `dream-reconcile.py` source +to understand which step broke and why. The script is structured as 9 +sequential, idempotent steps (see below). You can often fix the issue and +re-run the script (it skips dreams already in `_index.md`), or perform the +failing step manually and then re-run the remaining steps. Common failures: +missing manifest, corrupt report frontmatter, merge conflict in shadow file. +Adapt based on the error message. + +### What the Reconciler Does + +1. **Discovers** new branches (namespace-filtered, not in `_index.md`) +2. **Reads/validates** manifests from remote branches +3. **Merges** discoveries into main's per-file shadows (semantic dedup; on an exact-text duplicate it upgrades the existing entry's metadata — unions labels, raises source trust, promotes `uncertain`→`verified` — but never alters a `refuted` status) +4. **Mirrors** reports, manifests, patches to main's `_dreams/` +5. **Updates** `_dreams/_index.md` with new entries +6. **Updates** `_meta/state.json` +7. **Rebuilds** top-level `.shadow/_index.md` (per-file discovery counts) +8. **Verifies** all artifacts present (hard gate) +9. **(Optional)** Deletes reconciled branches — only with `--cleanup-branches`, and only after the reconciliation has been committed and pushed (refuses on a dirty `.shadow/` or when HEAD is not yet on `origin/`) + +### After Reconciliation: Commit, Push, and Cleanup + +```bash +cd "$REPO_ROOT" +git add .shadow/ +git commit -m "dream: reconcile $(date -u +%Y%m%d-%H%M%SZ) — N experiments" +git pull --rebase origin "$DEFAULT_BRANCH" || { + echo "ERROR: Rebase failed. Abort and retry manually." + git rebase --abort 2>/dev/null + exit 1 +} +if git push origin "$DEFAULT_BRANCH"; then + echo "✓ Pushed reconciliation" +else + echo "ERROR: Push failed. Retry: git pull --rebase && git push" + echo "⚠️ Do NOT clean up branches until push succeeds." + exit 1 +fi +``` + +### Post-Reconciliation Branch Cleanup + +After reconciliation is **committed AND pushed**, clean up dream branches +to prevent repo pollution. Only delete branches whose artifacts are safely +on main. + +```bash +for BRANCH in $RECONCILED_BRANCHES; do + DREAM_ID="${BRANCH#dream/${DREAM_NS}/}" + # Safety check: verify artifacts exist on main BEFORE deleting. + # Match the _index.md entry by EXACT cell (column 2), not substring — + # `grep -qF "$DREAM_ID"` would false-match when one dream_id is a prefix + # of another, deleting an un-reconciled branch. + if [ -f .shadow/_dreams/"$DREAM_ID"/report.md ] && \ + [ -f .shadow/_dreams/"$DREAM_ID"/manifest.json ] && \ + [ -f .shadow/_dreams/"$DREAM_ID"/patch.diff ] && \ + awk -F'|' -v id="$DREAM_ID" \ + '{gsub(/ /,"",$2); if ($2==id) f=1} END {exit !f}' \ + .shadow/_dreams/_index.md 2>/dev/null; then + # Safe to delete — all artifacts are on main + git push origin --delete "$BRANCH" 2>/dev/null && \ + echo " 🗑 Deleted remote: $BRANCH" + git branch -D "$BRANCH" 2>/dev/null && \ + echo " 🗑 Deleted local: $BRANCH" + else + echo " ⚠️ KEEPING $BRANCH — artifacts not verified on main" + fi +done +``` + +**Rules:** +- NEVER delete branches before push to main succeeds +- NEVER delete branches that have un-reconciled descendants +- If `SHADOWFROG_KEEP_BRANCHES=1` is set, skip cleanup (for eval harness) +- `dead_end` branches are cleaned up too — `patch.diff` + `tip_commit` in + index preserves recoverability +- Branches with compounding descendants: delete ONLY after descendants are + also reconciled (check `_index.md` for entries listing this branch as parent) + +**Worktree cleanup** happens separately (Phase 7 — see "Worktree Pruning" +below). Worktrees can be removed immediately after branch push regardless +of reconciliation status. Reconciled branches also have their worktree +GC'd automatically by `dream-reconcile.py --cleanup-branches`. + +### Recovery + +If reconciliation is interrupted: branches are already pushed (no data +loss). Re-run reconciliation — it's idempotent. The reconciler uses +`_dreams/_index.md` as its journal: any branch already listed there is +skipped, any branch not listed is reprocessed. The reconciler's own +step 8 verifies all artifacts on main; if verification fails the script +exits non-zero — fix the cause and re-run. + +## Phase 7: Summary, Review, and Pruning + +### End-of-Session Cleanup + +Before the summary, sweep leftover worktrees from the mid-batch leak +(dreams that pushed but weren't `dream-cleanup.sh`'d before the loop +exited). Only run this once the agent has asserted no more dreams are +starting **in this namespace**: + +```bash +bash "$SKILL_DIR/dream-gc.sh" \ + --task-complete --namespace "$DREAM_NS" \ + --repo-root "$REPO_ROOT" --min-age-min 0 +``` + +See "Worktree Pruning" below for `--namespace` rationale, `--min-age-min` +semantics, and the other three cleanup paths. + +### Summary + +``` +Dream session complete. + Results (by category): + Investigation: N tasks, M discoveries + Bug hunting: ... + Branches pushed: K + Branch tree: + main + +-- dream/ (useful) + +-- dream/ (dead_end) + Top findings: + - -- +``` + +### Experiment Review + +Walk through each experiment with the user. Actions: +- **Keep** (default) — branch and report stay +- **Delete** — remove from `_dreams/`, delete remote branch +- **Checkout** — inspect the code live + +Wait for user confirmation before deleting any remote branch. + +### Branch Pruning + +Reconciled branches are cleaned up automatically after push. For +branches not auto-cleaned (push failed, or `SHADOWFROG_KEEP_BRANCHES=1` +set), apply the rules in Phase 6 → Post-Reconciliation Branch Cleanup +(above). + +### Worktree Pruning + +Dream worktrees live OUTSIDE the repo at +`${DREAM_WORKTREE_BASE:-/tmp/shadowfrog-dreams}//dream-/`. There +are four places they get cleaned up: + +1. **`dream-cleanup.sh`** — called by the agent after each `git push` (see + "Worktree Cleanup" earlier in this skill). Removes ONE worktree. +2. **`dream-reconcile.py --cleanup-branches`** — after deleting a merged + branch, also `rm -rf`s its worktree directory. No extra command needed. +3. **`dream-gc.sh` (auto-triggered)** — `dream-setup.sh` invokes this + sweeper at the start of each new dream, throttled by a per-namespace + `.last-gc` tombstone to run at most once per `DREAM_GC_INTERVAL_MIN` + minutes (default 60). Catches orphans from crashed dreams, machine + reboots, OOM-killed agents — the long tail of cleanup failures that + accumulated GBs of leaked worktrees on long-running fleets. + + Env knobs (all optional, sensible defaults): + - `DREAM_GC_AUTO=0` — disable the auto-trigger entirely + - `DREAM_GC_INTERVAL_MIN` — how often the trigger fires (default 60) + - `DREAM_GC_AGE_MIN` — min worktree age to sweep (default 60) + +4. **`dream-gc.sh --task-complete --namespace "$DREAM_NS"`** — + end-of-session sweep, run by the agent when it stops dreaming (dream + count reached, or genuinely blocked). Unlike the auto-trigger, this + mode ALSO removes `stale-registered` worktrees (valid `.git` pointer + but no `dream-cleanup.sh` ever ran on them — the mid-batch + `task_complete` leak). The agent's assertion "I'm done dreaming + **in this namespace**" is what makes this safe. + + **Required:** `--namespace` (or `DREAM_NAMESPACE` env). The script + refuses with exit 2 if neither is given — that prevents a multi-repo + fleet sharing one `$DREAM_WORKTREE_BASE` from one agent's + `task_complete` destroying another agent's live worktrees. + + `--min-age-min` is an **mtime gate, not a liveness check**. Pass `0` + at end-of-session to catch the freshly-pushed final batch; pass a + higher value (e.g. `10`) if you can't fully assert that no other + dream in the same namespace is in flight. Locked worktrees + (`git worktree lock`) are always respected — the sweeper WARNs and + leaves them in place. + + ```bash + # At the end of the dream loop, before the final summary: + bash "$SKILL_DIR/dream-gc.sh" \ + --task-complete \ + --namespace "$DREAM_NS" \ + --repo-root "$REPO_ROOT" \ + --min-age-min 0 + ``` + +All four paths share a single safety gate (`_worktree_safety.py`) that +refuses ANY path which is not strictly under `$DREAM_WORKTREE_BASE` and +doesn't match the exact `//dream-` shape. The gate is +unconditional — even an attacker-controlled `$DREAM_WORKTREE_BASE` cannot +cause `rm -rf /`. + +### Applying Dream Code + +```bash +# Option 1: Merge the dream branch +git checkout "$DEFAULT_BRANCH" +git merge dream/ --no-ff -m "Adopt dream: " + +# Option 2: Cherry-pick specific commits +git cherry-pick <tip_commit> + +# Option 3: Apply the patch (if branch was pruned but commit exists) +git show <tip_commit> | git apply +``` + +## Guidance + +- **Always experiment.** Implementation reveals what reading cannot. +- **Small tasks, big lessons.** 30-minute experiment > 3 hours reading. +- **Fail forward.** "Tried X, broke because Y" is extremely valuable. +- **Breadth over depth.** More files with 2-3 discoveries > one file with 20. +- **No descriptions.** "Catches all exceptions including OOM" yes. + "This function authenticates users" no. +- **Compound deliberately.** Read parent's report. Build on findings. +- **Branches are cheap, shadow is expensive.** Push freely, write carefully. + +## Experiment Completion Criteria + +A task is complete ONLY when ALL of these hold: + +1. Code was written or modified (non-empty `patch.diff`) +2. At least one command was executed with exit code recorded in `Commands Run` +3. At least one finding tied to running code (not just reading) +4. At least one per-file `.shadow/*.md` edit made +5. Discoveries are behavioral insights, not feature descriptions +6. `report.md` saved with all required fields +7. Dream branch pushed to remote +8. Reconciliation completed and verified (`report.md`, `manifest.json`, + `patch.diff` exist on main, `_index.md` has entry) + +Additional gates: +- **Feature design:** `## Motivation` cites specific existing code +- **Compounding:** `## Compounding Delta` explains what parent code was modified + +A task that fails these criteria is **not completed**. If setup fails or +the experiment produces nothing, mark it `blocked` in the summary and +replace it with another experiment. Blocked tasks do not count toward the +category minimum. + +> **After reconciliation:** run `/shadow-frog-meditate` to consolidate +> discoveries and repair the index. + +## Curating Dream Experiments for Upstream PRs + +Once dream branches accumulate, you (or the user) may want to surface a +few worth submitting to the upstream project. The default AI failure mode +is sycophancy — approving too many experiments because they look like +work. Resist that. Apply these heuristics: + +1. **The maintainer test.** For each experiment, ask: *If I submitted this + as a PR to an open-source repo I don't maintain, would the maintainer + merge it — or politely close it?* This is the only question that matters. +2. **Devil's advocate framing.** Your job is to find reasons NOT to PR + each experiment. Recommend only when you cannot find a compelling + reason to reject. +3. **70% rejection quota.** If you approve more than 30% of experiments + reviewed, your standards are too low. Re-evaluate. +4. **The "so what?" test.** Would a human engineer read the report and + say "so what?" If yes, reject. +5. **The 30-minute test.** Could a competent developer have produced + this in 30 minutes with a linter, TODO grep, or quick docs read? If + yes, it's maintenance work, not a contribution. Reject. +6. **The novelty test.** Does the experiment reveal a non-obvious + behavior, hidden assumption, or unexpected interaction? If not, reject. +7. **No credit for effort.** A 10-experiment chain that produces a minor + tweak is still a minor tweak. Judge the result, not the journey. + +When you do submit a PR, write the body for someone who has never seen +the fork. Include: one-paragraph "what it does" derived from the dream +report, the experiment's evidence (test output, before/after metric), +and an honest "what we did not verify" note. diff --git a/skills/shadow-frog-dream/_worktree_safety.py b/skills/shadow-frog-dream/_worktree_safety.py new file mode 100644 index 0000000..05c0b19 --- /dev/null +++ b/skills/shadow-frog-dream/_worktree_safety.py @@ -0,0 +1,207 @@ +"""Single source of truth for the "is it safe to rm -rf this dream worktree?" gate. + +Imported by `dream-reconcile.py` and invoked as a subprocess by +`dream-cleanup.sh` + `dream-gc.sh`. Both bash callers pass the candidate +path + the configured base via argv (NOT via string interpolation), so the +caller can never inject Python source through this module. + +A path is considered safe to remove ONLY when ALL of these hold: + +1. Path and base are both non-empty strings. +2. Path and base are both absolute (start with "/"). +3. Neither contains a ".." component in the literal input. +4. The resolved (symlink-followed) base is not a sensitive filesystem root + (e.g. /, /tmp, /home, $HOME, /Users, /etc, /usr). +5. After resolving symlinks in the path's parent directory, the resolved + path is STRICTLY INSIDE the resolved base — not equal to it, and not + above it. +6. The path matches the exact dream-worktree shape `<base>/<ns>/dream-<slug>` + where `<ns>` and `<slug>` are each `[A-Za-z0-9._-]+` (the same `SAFE_RE` + that `dream-setup.sh` already enforces on the inputs). + +These rules are deliberately strict: they reject anything that doesn't look +like a dream worktree created by `dream-setup.sh`. That means we will NEVER +`rm -rf` a path the user happens to point us at — only paths that match the +namespace's own creation contract. + +CLI invocation (used by the bash scripts): + python3 _worktree_safety.py <worktree-dir> <base> [<ns>] + exit codes: + 0 → safe AND the path currently exists (caller should rm) + 2 → safe AND the path does NOT exist (caller should treat as no-op) + 1 → UNSAFE; do not rm (error printed to stderr) +""" +from __future__ import annotations + +import os +import re +import sys +from pathlib import Path + +# Same regex `dream-setup.sh` validates --slug and --namespace against. +# Keep these in lockstep — if one widens, the other must follow. +_SAFE_RE = re.compile(r"^[A-Za-z0-9._-]+$") + +# Defense-in-depth: even if rule 5 (strictly under base) holds, refuse +# outright if the BASE itself lands on one of these. Checked against BOTH +# the literal input AND the symlink-resolved value, with macOS's `/private` +# prefix stripped — otherwise `/tmp` → `/private/tmp` would silently bypass +# the check on macOS (verified empirically: macOS `realpath /tmp` returns +# `/private/tmp` which is not in the list, so the literal `/tmp` check is +# what catches it). +_FORBIDDEN_BASES = frozenset({ + "/", + "/bin", "/boot", "/dev", "/etc", "/home", "/lib", "/lib32", "/lib64", + "/Library", "/mnt", "/media", "/opt", "/private", "/proc", "/root", + "/run", "/sbin", "/srv", "/sys", "/System", "/tmp", "/Users", "/usr", + "/Users/Shared", "/var", "/var/folders", "/var/tmp", +}) + +# macOS-specific: `/private` is the real location for `/tmp`, `/etc`, +# `/var`. After `realpath` these all gain the `/private` prefix. +_MACOS_PRIVATE_PREFIX = "/private" + + +def _strip_macos_private(p: str) -> str: + """Strip the macOS `/private` prefix so resolved paths can be compared + against the bare forbidden roots. `/private` itself stays `/private`.""" + if p.startswith(_MACOS_PRIVATE_PREFIX + "/"): + return p[len(_MACOS_PRIVATE_PREFIX):] + return p + + +class UnsafePath(ValueError): + """Raised when the candidate path fails any of the safety rules.""" + + +def safe_worktree_path(path: str, base: str) -> Path: + """Validate `path` is a safe-to-remove dream worktree under `base`. + + Returns the symlink-resolved `Path` on success. + Raises `UnsafePath` on any rule violation. + Does NOT touch the filesystem beyond `os.path.realpath` resolution. + """ + # Rule 1: non-empty. + if not isinstance(path, str) or not path.strip(): + raise UnsafePath(f"empty or non-string worktree path: {path!r}") + if not isinstance(base, str) or not base.strip(): + raise UnsafePath(f"empty or non-string base: {base!r}") + + # Rule 2: absolute. + if not path.startswith("/"): + raise UnsafePath(f"worktree path is not absolute: {path!r}") + if not base.startswith("/"): + raise UnsafePath(f"base is not absolute: {base!r}") + + # Rule 3: no ".." traversal in the literal input. Catches things that + # would otherwise normalize past the base. + for part in path.split("/"): + if part == "..": + raise UnsafePath(f"worktree path contains '..': {path!r}") + for part in base.split("/"): + if part == "..": + raise UnsafePath(f"base contains '..': {base!r}") + + # Resolve the base: this is what we compare against. + base_res = os.path.realpath(base) + + # Rule 4: refuse sensitive bases. Compare against three normalizations + # so a symlink shenanigan (`$HOME/my-worktrees → /`) AND macOS's + # implicit `/tmp → /private/tmp` redirection both fail closed. + base_norm = os.path.normpath(base) + candidates = {base_norm, base_res, _strip_macos_private(base_res)} + forbidden = set(_FORBIDDEN_BASES) + home = os.path.expanduser("~") + if home and home != "~": + forbidden.add(home) + forbidden.add(_strip_macos_private(os.path.realpath(home))) + if candidates & forbidden: + raise UnsafePath( + f"refusing sweep: base resolves to a sensitive root: " + f"{base!r} (literal={base_norm!r} resolved={base_res!r})" + ) + + # Resolve the path's PARENT (not the path itself — the leaf may not + # exist, which is fine for the rm-is-a-no-op case). os.path.realpath + # walks all symlinks; combining with the literal leaf prevents a + # symlink AT the leaf from escaping the base after a successful check. + stripped = path.rstrip("/") + parent_lit = os.path.dirname(stripped) or "/" + leaf = os.path.basename(stripped) + parent_res = os.path.realpath(parent_lit) + resolved = os.path.join(parent_res, leaf) + + # If the leaf is itself a symlink, follow it AFTER constructing + # `resolved` so the strict-inside-base check uses the real target. + if os.path.islink(resolved): + resolved = os.path.realpath(resolved) + + # Rule 5: strictly under base. + try: + rel = os.path.relpath(resolved, base_res) + except ValueError as exc: + raise UnsafePath( + f"worktree path {resolved!r} not relatable to base " + f"{base_res!r}: {exc}" + ) from None + if rel == "." or rel.startswith(".."): + raise UnsafePath( + f"worktree path {resolved!r} is not strictly under base " + f"{base_res!r} (relpath={rel!r})" + ) + + # Rule 6: exact dream-worktree shape: <base>/<ns>/dream-<slug>. + parts = rel.split(os.sep) + if len(parts) != 2: + raise UnsafePath( + f"worktree path {resolved!r} is not exactly 2 levels under " + f"base {base_res!r} (got parts={parts!r})" + ) + ns_part, leaf_part = parts + if not _SAFE_RE.match(ns_part): + raise UnsafePath( + f"namespace component {ns_part!r} does not match {_SAFE_RE.pattern}" + ) + if not leaf_part.startswith("dream-"): + raise UnsafePath( + f"leaf {leaf_part!r} does not start with 'dream-'" + ) + slug_part = leaf_part[len("dream-"):] + if not slug_part or not _SAFE_RE.match(slug_part): + raise UnsafePath( + f"slug component {slug_part!r} does not match {_SAFE_RE.pattern}" + ) + + return Path(resolved) + + +def _cli() -> int: + if len(sys.argv) < 3: + print( + "usage: _worktree_safety.py <worktree-dir> <base>", + file=sys.stderr, + ) + return 1 + path, base = sys.argv[1], sys.argv[2] + try: + resolved = safe_worktree_path(path, base) + except UnsafePath as exc: + print(f"ERROR: {exc}", file=sys.stderr) + return 1 + except (ValueError, OSError) as exc: + # Defensive: e.g. embedded NUL would raise ValueError from + # os.path.realpath. Treat any non-UnsafePath gate failure as + # "refused" rather than crashing with a traceback that the bash + # caller would interpret as exit-code-1 anyway, but noisily. + print(f"ERROR: gate failure: {exc}", file=sys.stderr) + return 1 + # Check existence on the RESOLVED path. islink() catches dangling + # symlinks (which `exists()` reports as False but which we should + # still remove). + if resolved.exists() or resolved.is_symlink(): + return 0 + return 2 + + +if __name__ == "__main__": + sys.exit(_cli()) diff --git a/skills/shadow-frog-dream/dream-cleanup.sh b/skills/shadow-frog-dream/dream-cleanup.sh new file mode 100755 index 0000000..1abf41b --- /dev/null +++ b/skills/shadow-frog-dream/dream-cleanup.sh @@ -0,0 +1,166 @@ +#!/usr/bin/env bash +# Dream worktree cleanup — safe removal of ONE dream worktree. +# +# Usage: +# dream-cleanup.sh <worktree-dir> [--repo-root DIR] [--quiet] +# +# Replaces the previous inline shell snippet: +# git worktree remove "$WORKTREE_DIR" --force 2>/dev/null +# git worktree prune +# which silently leaked the worktree directory whenever `git worktree remove` +# failed (the `2>/dev/null` swallowed the error AND the next prune only +# cleared git's bookkeeping — not the directory on disk). +# +# This script first tries the polite path (`git worktree remove --force`), +# and ONLY if git fails AND the path passes a strict safety gate, falls back +# to `rm -rf`. The safety gate is enforced by `_worktree_safety.py` (see +# that module for the full rule set). +# +# This script is INTERACTIVE (run by the dream agent) so it FAILS FAST on +# bad input — unlike the always-exit-0 hook scripts. Exit codes: +# 0 → worktree removed (or never existed) +# 1 → safety gate refused the path (no removal attempted) +# 2 → usage error +# 3 → fallback rm -rf itself failed +# 4 → safety module (_worktree_safety.py) is missing +# +# Flags: +# --repo-root DIR Where to invoke `git worktree remove` from. +# If omitted, falls back to $REPO_ROOT env var, then +# derives from the worktree's `.git` pointer +# (rev-parse --git-common-dir). +# --quiet Suppress progress output (errors still print). +# --help, -h Show this help message. +# +# Environment: +# DREAM_WORKTREE_BASE Override the base path the safety gate checks +# against (default: /tmp/shadowfrog-dreams). + +set -euo pipefail + +show_help() { + sed -n '2,/^$/p' "$0" | sed 's/^# \{0,1\}//' + exit 0 +} + +WORKTREE_DIR="" +REPO_ROOT_OVERRIDE="" +QUIET=false + +while [[ $# -gt 0 ]]; do + case "$1" in + --repo-root) REPO_ROOT_OVERRIDE="$2"; shift 2 ;; + --quiet|-q) QUIET=true; shift ;; + --help|-h) show_help ;; + --*) echo "ERROR: unknown flag: $1" >&2; exit 2 ;; + *) + if [[ -z "$WORKTREE_DIR" ]]; then + WORKTREE_DIR="$1"; shift + else + echo "ERROR: unexpected positional arg: $1" >&2; exit 2 + fi + ;; + esac +done + +if [[ -z "$WORKTREE_DIR" ]]; then + echo "ERROR: <worktree-dir> is required" >&2 + echo "Usage: dream-cleanup.sh <worktree-dir> [--repo-root DIR] [--quiet]" >&2 + exit 2 +fi + +say() { [[ "$QUIET" == true ]] || echo "$@"; } + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +SAFETY="$SCRIPT_DIR/_worktree_safety.py" +BASE="${DREAM_WORKTREE_BASE:-/tmp/shadowfrog-dreams}" + +# --- Pre-flight: safety module must exist --- +# If `_worktree_safety.py` is missing, `python3` itself exits 2 — the same +# code the gate uses for "safe but path missing". Detect this BEFORE +# invoking python so a missing module fails loudly instead of silently +# pretending nothing needs cleanup (which would re-introduce the leak +# this script exists to fix). +if [[ ! -f "$SAFETY" ]]; then + echo "ERROR: safety module not found: $SAFETY" >&2 + echo " refusing to operate without a safety gate." >&2 + exit 4 +fi + +# --- SAFETY GATE — enforced BEFORE any destructive action --- +# Exit codes from _worktree_safety.py: +# 0 = safe AND path exists (we proceed) +# 2 = safe AND path missing (idempotent no-op — exit 0) +# 1 = refused (we abort) +guard_rc=0 +python3 "$SAFETY" "$WORKTREE_DIR" "$BASE" || guard_rc=$? +case "$guard_rc" in + 0) ;; + 2) say "Nothing to clean (path doesn't exist): $WORKTREE_DIR"; exit 0 ;; + 1) + echo "ERROR: dream-cleanup.sh refuses to operate on '$WORKTREE_DIR'" >&2 + echo " (base=$BASE — see _worktree_safety.py for rules)" >&2 + exit 1 + ;; + *) + echo "ERROR: safety gate returned unexpected code $guard_rc for '$WORKTREE_DIR'" >&2 + exit 1 + ;; +esac + +# --- Resolve repo root (where the bare repo lives) for `git worktree …` --- +# Precedence: --repo-root flag > $REPO_ROOT env var > worktree's gitdir. +# Parameter expansion preserves env-inherited REPO_ROOT instead of +# clobbering it (the previous unconditional assignment lost env values). +REPO_ROOT="${REPO_ROOT_OVERRIDE:-${REPO_ROOT:-}}" +# Fall back to the dream worktree's recorded gitdir when --repo-root and +# $REPO_ROOT are both absent. A dead worktree (broken .git pointer) will +# fail this — that's fine, we'll just skip the git step and rm -rf below. +if [[ -z "$REPO_ROOT" ]] && [[ -e "$WORKTREE_DIR/.git" ]]; then + REPO_ROOT="$(git -C "$WORKTREE_DIR" rev-parse --show-superproject-working-tree 2>/dev/null || true)" + if [[ -z "$REPO_ROOT" ]]; then + # Single-repo case: superproject is empty; derive from common-dir. + COMMON_DIR="$(git -C "$WORKTREE_DIR" rev-parse --git-common-dir 2>/dev/null || true)" + if [[ -n "$COMMON_DIR" ]]; then + REPO_ROOT="$(cd "$COMMON_DIR/.." 2>/dev/null && pwd || true)" + fi + fi +fi + +# Helper: is $1 a usable git repo root (regular OR worktree OR bare)? +# `-d .git` was too strict — when REPO_ROOT is itself a worktree, `.git` +# is a file, not a directory, and we'd silently skip the polite git path. +_is_git_root() { + [[ -n "${1:-}" ]] && git -C "$1" rev-parse --git-dir >/dev/null 2>&1 +} + +# --- Step 1: try `git worktree remove --force` (lets git clean its bookkeeping) --- +git_cleaned=false +if _is_git_root "$REPO_ROOT"; then + if git -C "$REPO_ROOT" worktree remove "$WORKTREE_DIR" --force 2>/dev/null; then + git_cleaned=true + say "Removed via git worktree: $WORKTREE_DIR" + fi +fi + +# --- Step 2: fallback rm -rf (only because safety gate passed) --- +# The gate enforces: absolute path, strictly under $DREAM_WORKTREE_BASE, +# exact `<base>/<ns>/dream-<slug>` shape, every component matches SAFE_RE. +# We re-check existence in case `git worktree remove` actually succeeded +# (some failure-modes of git's exit code don't reflect rm success). +if [[ "$git_cleaned" != true ]] && [[ -d "$WORKTREE_DIR" || -L "$WORKTREE_DIR" ]]; then + say "git worktree remove failed (or no repo-root) — rm -rf fallback: $WORKTREE_DIR" + if rm -rf -- "$WORKTREE_DIR"; then + say "Removed: $WORKTREE_DIR" + else + echo "ERROR: rm -rf '$WORKTREE_DIR' failed" >&2 + exit 3 + fi +fi + +# --- Step 3: prune stale worktree refs (cheap, safe, no-op if nothing stale) --- +if _is_git_root "$REPO_ROOT"; then + git -C "$REPO_ROOT" worktree prune 2>/dev/null || true +fi + +exit 0 diff --git a/skills/shadow-frog-dream/dream-coverage.py b/skills/shadow-frog-dream/dream-coverage.py new file mode 100644 index 0000000..860c712 --- /dev/null +++ b/skills/shadow-frog-dream/dream-coverage.py @@ -0,0 +1,262 @@ +#!/usr/bin/env python3 +"""Dream coverage map — compute exploration coverage for task planning. + +Usage: + python3 dream-coverage.py [REPO_ROOT] + python3 dream-coverage.py [REPO_ROOT] --scope src/auth/ --scope src/db/ + +Outputs: + - Coverage summary (total, covered, uncovered) + - Saturated files (8+ discoveries) + - High-value uncovered files (sorted by fan-in) + - Per-directory coverage + +Coverage definition: A file is "covered" if its shadow file has at least 1 +behavioral discovery (line starting with "- "). Files with only placeholder +text ("_No discoveries yet._") are NOT covered. + +Scoped exploration (--scope): + Restricts the coverage map to files whose path starts with one of the + given prefixes. Repeatable. Use this to focus a dream session on a + specific subtree (e.g., a frontier area where bug density is high). + All counts (totals, %, saturated, fan-in, per-dir) are computed over + the scoped subset only. Exits nonzero if a non-empty --scope list + matches zero files so callers notice typos. + +Replaces the inline bash coverage map block in SKILL.md, fixing: + - Space-delimited filename handling (now uses Python lists) + - O(n²) fan-in computation (now batched with cap) + - Division by zero on empty repos + - Hardcoded temp file collisions + - Coverage = file existence (now = discovery_count > 0) +""" + +import argparse +import os +import subprocess +import sys +from collections import defaultdict + +# Directory components and suffixes to exclude. Kept in lockstep with +# shadow-init.py's EXCLUDE_DIRS / EXCLUDE_PATTERNS_SUFFIX so coverage's +# denominator is exactly the set of files init shadows (tests INCLUDED — +# init shadows them, so excluding them here would hide shadowed test files +# from the "uncovered" list and inflate the coverage percentage). +EXCLUDE_DIRS = { + "node_modules", "vendor", "venv", ".venv", "__pycache__", + "dist", "build", "target", "out", ".shadow", +} +EXCLUDE_SUFFIXES = (".min.js", ".min.css", ".map", ".lock") + + +def _is_excluded_path(rel_path): + """Match shadow-init.py: excluded directory component or suffix.""" + parts = rel_path.replace(os.sep, "/").split("/") + if any(part in EXCLUDE_DIRS for part in parts): + return True + return any(rel_path.endswith(suffix) for suffix in EXCLUDE_SUFFIXES) + + +# Source file selection — must mirror shadow-init.py's _is_source_file so +# coverage's denominator matches the set of files init actually shadows. +# Without this, non-source tracked files (READMEs, images, docs) are counted +# as "uncovered source files" and skew dream planning. +SOURCE_EXTENSIONS = { + ".py", + ".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs", + ".java", ".kt", ".kts", ".scala", + ".go", + ".rs", + ".rb", + ".c", ".h", ".cpp", ".hpp", ".cc", ".hh", ".cxx", ".hxx", + ".cs", + ".php", + ".sh", ".bash", ".zsh", + ".swift", + ".yaml", ".yml", ".toml", ".json", +} +SOURCE_BASENAMES = {"Makefile", "Dockerfile", "Containerfile", "Rakefile", "Gemfile"} + + +def _is_source_file(rel_path): + """Match shadow-init.py: known source extensions or basenames.""" + base = os.path.basename(rel_path) + if base in SOURCE_BASENAMES: + return True + _, ext = os.path.splitext(base) + return ext.lower() in SOURCE_EXTENSIONS + + +def get_source_files(repo_root, scopes=None): + """Get all tracked source files (same selection rules as shadow-init.py). + + If `scopes` is a non-empty list of path prefixes, only files whose + path starts with at least one prefix are returned. Prefix matching + is literal (use a trailing '/' to scope a directory cleanly). + """ + result = subprocess.run( + ['git', 'ls-files', '-z'], + capture_output=True, text=True, cwd=repo_root + ) + files = [] + for f in result.stdout.split('\0'): + if not f or _is_excluded_path(f): + continue + if not _is_source_file(f): + continue + if scopes and not any(f.startswith(p) for p in scopes): + continue + files.append(f) + return files + + +def _count_discoveries(shadow_path): + """Count `- ` bullets that are real discoveries. + + Bullets inside a `## Cross-References` section are back-pointer links + to `_cross/*.md`, NOT discoveries. Skip them. + """ + count = 0 + in_xref = False + with open(shadow_path) as sf: + for line in sf: + stripped = line.rstrip('\n') + if stripped.startswith('## ') or stripped.startswith('### '): + in_xref = stripped.strip().lower().startswith('## cross-references') + continue + if not in_xref and line.startswith('- '): + count += 1 + return count + + +def check_coverage(repo_root, files): + """Check shadow coverage. Covered = discovery_count > 0.""" + covered = [] + uncovered = [] + saturated = [] + + for f in files: + shadow = os.path.join(repo_root, '.shadow', f + '.md') + if os.path.isfile(shadow): + count = _count_discoveries(shadow) + if count > 0: + covered.append((f, count)) + if count >= 8: + saturated.append((f, count)) + else: + # Shadow exists but no real discoveries (placeholder only) + uncovered.append(f) + else: + uncovered.append(f) + + return covered, uncovered, saturated + + +def compute_fan_in(repo_root, uncovered_files, max_files=200): + """Compute fan-in for uncovered files. Capped at max_files to avoid timeout.""" + if not uncovered_files: + return {} + + # Deduplicate basenames and batch + basenames = {} + for f in uncovered_files[:max_files]: + base = os.path.splitext(os.path.basename(f))[0] + if base not in basenames: + basenames[base] = [] + basenames[base].append(f) + + fan_in = defaultdict(int) + for base, files_for_base in basenames.items(): + try: + result = subprocess.run( + ['git', 'grep', '-F', '-w', '-l', '--', base], + capture_output=True, text=True, cwd=repo_root, + timeout=10 + ) + count = len([l for l in result.stdout.strip().split('\n') if l]) if result.stdout.strip() else 0 + for f in files_for_base: + fan_in[f] = count + except (subprocess.TimeoutExpired, OSError): + for f in files_for_base: + fan_in[f] = 0 + + return fan_in + + +def main(): + parser = argparse.ArgumentParser( + description="Dream coverage map — compute exploration coverage for task planning.", + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=__doc__, + ) + parser.add_argument( + 'repo_root', nargs='?', default=os.getcwd(), + help='Repository root (defaults to current working directory).', + ) + parser.add_argument( + '--scope', action='append', default=[], metavar='PREFIX', + help='Restrict coverage to files whose path starts with PREFIX. ' + 'Repeatable (e.g., --scope src/auth/ --scope src/db/). ' + 'Trailing slash recommended for directory prefixes.', + ) + args = parser.parse_args() + repo_root = args.repo_root + scopes = args.scope + + print("=== EXPLORATION COVERAGE MAP ===") + if scopes: + print(f"Scope filters: {', '.join(scopes)}") + + files = get_source_files(repo_root, scopes=scopes) + total = len(files) + print(f"Total source files: {total}") + + if total == 0: + if scopes: + print(f"No source files matched scope filters: {scopes}") + print("Check prefix spelling (trailing slash matters).") + sys.exit(2) + print("No source files found after filtering.") + print("Coverage: N/A") + return + + covered, uncovered, saturated = check_coverage(repo_root, files) + + for f, count in saturated: + print(f"SATURATED ({count}): {f}") + + pct = len(covered) * 100 // total + print(f"Covered: {len(covered)} / {total} ({pct}%)") + print(f"Uncovered: {len(uncovered)}") + + # High-value uncovered files + if uncovered: + print() + print("=== HIGH-VALUE UNCOVERED FILES ===") + fan_in = compute_fan_in(repo_root, uncovered) + ranked = sorted(uncovered, key=lambda f: fan_in.get(f, 0), reverse=True) + for f in ranked[:30]: + refs = fan_in.get(f, 0) + print(f"{refs} {f}") + print("(sorted by reference count — high fan-in files should be explored first)") + + # Per-directory coverage + if uncovered: + print() + print("=== DIRECTORY COVERAGE ===") + dir_stats = defaultdict(lambda: {'total': 0, 'covered': 0}) + for f in files: + d = os.path.dirname(f) or '.' + dir_stats[d]['total'] += 1 + for f, _ in covered: + d = os.path.dirname(f) or '.' + dir_stats[d]['covered'] += 1 + + for d in sorted(dir_stats.keys()): + s = dir_stats[d] + if s['covered'] < s['total']: + print(f"{d}: {s['covered']} / {s['total']}") + + +if __name__ == '__main__': + main() diff --git a/skills/shadow-frog-dream/dream-gc.sh b/skills/shadow-frog-dream/dream-gc.sh new file mode 100755 index 0000000..884db0c --- /dev/null +++ b/skills/shadow-frog-dream/dream-gc.sh @@ -0,0 +1,333 @@ +#!/usr/bin/env bash +# Dream worktree garbage collector — sweep ORPHAN worktrees from the base. +# +# Usage: +# dream-gc.sh [--repo-root DIR] [--min-age-min N] [--dry-run] [--quiet] +# dream-gc.sh --task-complete --namespace NS [--repo-root DIR] [--min-age-min N] +# +# Defense-in-depth complement to dream-cleanup.sh: even when individual +# `dream-cleanup.sh` calls fail (e.g. machine crash, OOM-killed agent), +# this sweeper finds and removes orphan dream worktrees that have outlived +# their git registration. Safe to cron / run manually. +# +# IMPORTANT: `--task-complete` is always namespace-scoped (refuses without +# `--namespace`). Default orphan-only mode walks the whole base, which is +# safe because it only sweeps dirs whose `.git` pointer is already broken. +# +# A directory under $DREAM_WORKTREE_BASE/<ns>/dream-<slug>/ is an "orphan" +# when BOTH of these hold: +# (a) It is older than --min-age-min minutes (default: 10) by mtime. +# Avoids racing with a worktree being created RIGHT NOW. +# (b) Its git registration is broken: either `.git` is missing, OR the +# `.git` file points at a `gitdir:` path that no longer exists. +# +# Every candidate is then re-validated through `_worktree_safety.py` before +# removal. This script can NEVER rm a path that doesn't match the +# `<base>/<ns>/dream-<slug>` shape, even if the base is misconfigured. +# +# Flags: +# --repo-root DIR Where the bare repo lives (for `git worktree prune` +# at the end). Default: $REPO_ROOT, then $PWD. +# --min-age-min N Only consider dirs whose MTIME is older than N +# minutes (default: 10). NOT a liveness check — an +# active dream that hasn't written to disk in N +# minutes is still eligible. +# --dry-run Print what would be removed; remove nothing. +# --quiet, -q Suppress progress output (errors still print). +# --task-complete Sweep registered-but-stale dream-* dirs too (not +# just orphans), in the named namespace ONLY. +# REQUIRES --namespace. For each candidate, +# `git worktree remove --force` runs first; if git +# refuses (e.g. the worktree is `git worktree +# lock`'d), we WARN and skip — we do NOT fall back +# to `rm -rf` for registered candidates, since git's +# refusal is a liveness signal we must respect. +# --namespace NS, -n NS +# Required for `--task-complete`. Restricts the +# sweep to `<base>/<ns>/`. Default source: +# $DREAM_NAMESPACE env. No auto-derivation from +# REPO_ROOT — that could silently sweep the wrong +# namespace if REPO_ROOT came from cwd inference. +# --help, -h Show this help message. +# +# Environment: +# DREAM_WORKTREE_BASE Base path to sweep (default: /tmp/shadowfrog-dreams). +# Refuses sensitive bases (/, /tmp, /home, $HOME, …). +# +# Exit codes: +# 0 → sweep completed (anything from 0..N worktrees removed) +# 1 → base path is unsafe to sweep (no removal attempted) +# 2 → usage error +# 4 → safety module (_worktree_safety.py) is missing + +set -euo pipefail + +show_help() { + sed -n '2,/^$/p' "$0" | sed 's/^# \{0,1\}//' + exit 0 +} + +MIN_AGE_MIN=10 +DRY_RUN=false +QUIET=false +TASK_COMPLETE=false +REPO_ROOT_OVERRIDE="" +NAMESPACE_OVERRIDE="" + +while [[ $# -gt 0 ]]; do + case "$1" in + --repo-root) REPO_ROOT_OVERRIDE="$2"; shift 2 ;; + --min-age-min) MIN_AGE_MIN="$2"; shift 2 ;; + --dry-run) DRY_RUN=true; shift ;; + --quiet|-q) QUIET=true; shift ;; + --task-complete) TASK_COMPLETE=true; shift ;; + --namespace|-n) NAMESPACE_OVERRIDE="$2"; shift 2 ;; + --help|-h) show_help ;; + *) echo "ERROR: unknown flag: $1" >&2; exit 2 ;; + esac +done + +# Validate --min-age-min is a non-negative integer (prevents shell injection +# via the `find -mmin +N` arg). +if ! [[ "$MIN_AGE_MIN" =~ ^[0-9]+$ ]]; then + echo "ERROR: --min-age-min must be a non-negative integer (got: $MIN_AGE_MIN)" >&2 + exit 2 +fi + +# --- Resolve namespace (REQUIRED when --task-complete is set) --- +# No auto-derivation from `basename "$REPO_ROOT"` — REPO_ROOT can come +# from cwd inference, so that fallback could silently sweep the wrong ns. +NAMESPACE="${NAMESPACE_OVERRIDE:-${DREAM_NAMESPACE:-}}" +if [[ "$TASK_COMPLETE" == true ]]; then + if [[ -z "$NAMESPACE" ]]; then + echo "ERROR: --task-complete requires --namespace NS (or DREAM_NAMESPACE env)" >&2 + echo " Refusing to sweep registered worktrees across all namespaces." >&2 + exit 2 + fi + # First char non-`.` to reject bare `.`/`..` (depth-1 + dream-* basename + + # _worktree_safety.py already defend in depth; this just makes intent clear). + if ! [[ "$NAMESPACE" =~ ^[A-Za-z0-9_-][A-Za-z0-9._-]*$ ]]; then + echo "ERROR: --namespace must match [A-Za-z0-9_-][A-Za-z0-9._-]* (got: $NAMESPACE)" >&2 + exit 2 + fi +fi + +# Validate --min-age-min as a non-negative integer. We need this before the +# find branching below: `0` means "no age gate" (omit -mmin entirely) while +# any positive N maps to `-mmin +N`. Without this check, a non-integer would +# silently flow to find and bypass age gating. +if ! [[ "$MIN_AGE_MIN" =~ ^[0-9]+$ ]]; then + echo "ERROR: --min-age-min must be a non-negative integer (got: $MIN_AGE_MIN)" >&2 + exit 2 +fi + +say() { [[ "$QUIET" == true ]] || echo "$@"; } + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +SAFETY="$SCRIPT_DIR/_worktree_safety.py" +BASE="${DREAM_WORKTREE_BASE:-/tmp/shadowfrog-dreams}" + +# --- Pre-flight: safety module must exist --- +# A missing `_worktree_safety.py` makes `python3` exit 2 (same code as +# "missing path"); without this guard the probe + per-candidate gate would +# both fail in non-distinguishable ways. Refuse loudly. +if [[ ! -f "$SAFETY" ]]; then + echo "ERROR: safety module not found: $SAFETY" >&2 + echo " refusing to sweep without a safety gate." >&2 + exit 4 +fi + +# --- Refuse upfront if the base itself is unsafe --- +# Use a sentinel path to probe the base: `<base>/__probe__/dream-_probe` +# violates the shape only if BASE itself is sensitive (rule 4 fires first). +# A successful gate on a real path implies BASE is non-sensitive. +PROBE="$BASE/__probe__/dream-_probe_" +probe_rc=0 +python3 "$SAFETY" "$PROBE" "$BASE" >/dev/null 2>&1 || probe_rc=$? +case "$probe_rc" in + 0|2) ;; # gate passed (path-exists or path-missing both fine for probe) + 1) + echo "ERROR: dream-gc.sh refuses to sweep base '$BASE'" >&2 + echo " (sensitive root, or fails safety check — set DREAM_WORKTREE_BASE)" >&2 + exit 1 + ;; + *) + echo "ERROR: safety gate returned unexpected code $probe_rc for base '$BASE'" >&2 + exit 1 + ;; +esac + +# Nothing to sweep if base doesn't exist (idempotent no-op). +if [[ ! -d "$BASE" ]]; then + say "Base does not exist (nothing to sweep): $BASE" + exit 0 +fi + +# --- Resolve repo root for the final `git worktree prune` --- +REPO_ROOT="${REPO_ROOT_OVERRIDE:-${REPO_ROOT:-}}" +if [[ -z "$REPO_ROOT" ]]; then + REPO_ROOT="$(git rev-parse --show-toplevel 2>/dev/null || true)" +fi + +# Helper: is $1 a usable git repo root (regular OR worktree OR bare)? +_is_git_root() { + [[ -n "${1:-}" ]] && git -C "$1" rev-parse --git-dir >/dev/null 2>&1 +} + +say "Sweeping dream worktree base: $BASE" +say " min age: ${MIN_AGE_MIN} minutes" +if [[ "$TASK_COMPLETE" == true ]]; then + say " TASK-COMPLETE MODE — sweeping registered worktrees in namespace '$NAMESPACE' only" +fi +[[ "$DRY_RUN" == true ]] && say " DRY RUN — no removal" + +# --- Enumerate candidates --- +# Shape: $BASE/<ns>/dream-<slug>. We use `find` to keep this fast on large +# bases. `-mindepth 2 -maxdepth 2` matches exactly that level. +# `-mmin +N` requires modification time older than N minutes (avoids racing +# with a fresh `dream-setup.sh` mid-creation). +removed=0 +kept=0 +refused=0 + +# NUL-delimited to handle exotic chars even though SAFE_RE forbids them. +while IFS= read -r -d '' candidate; do + # Only consider dirs whose leaf starts with "dream-". + [[ "$(basename "$candidate")" == dream-* ]] || continue + + # Re-validate via the safety gate. This guarantees the candidate matches + # `<base>/<ns>/dream-<slug>` exactly. + gate_rc=0 + python3 "$SAFETY" "$candidate" "$BASE" >/dev/null 2>&1 || gate_rc=$? + if [[ "$gate_rc" -ne 0 ]]; then + say " refused (gate): $candidate" + refused=$((refused + 1)) + continue + fi + + # Orphan check: missing .git OR broken gitdir pointer. + # In --task-complete mode we ALSO sweep registered (non-orphan) dirs, + # so the orphan check becomes informational rather than a gate. + git_path="$candidate/.git" + is_orphan=false + if [[ ! -e "$git_path" ]]; then + is_orphan=true + elif [[ -f "$git_path" ]]; then + # `.git` is a file (the standard worktree layout). Parse the + # `gitdir:` line robustly: + # - sed strips ONLY the `gitdir:` prefix + leading whitespace, + # preserving any `:` chars in the rest of the path. + # - `tr -d '\r'` tolerates CRLF line endings. + # - `head -1` defends against multi-line files (only first matters). + # Earlier `awk -F': *'` truncated on every `:`, so a path like + # `/repo:with-colon/.git/...` was misclassified as orphan and the + # live worktree would be deleted. + gitdir_target="$(sed -n 's|^gitdir:[[:space:]]*||p' "$git_path" 2>/dev/null | head -1 | tr -d '\r' || true)" + if [[ -z "$gitdir_target" ]]; then + is_orphan=true + else + # Relative gitdir paths (rare but legal) resolve relative to the + # .git file's directory — NOT the cwd. + if [[ "$gitdir_target" != /* ]]; then + gitdir_target="$candidate/$gitdir_target" + fi + if [[ ! -e "$gitdir_target" ]]; then + is_orphan=true + fi + fi + fi + + # Decide whether to sweep this candidate. + if [[ "$is_orphan" != true ]] && [[ "$TASK_COMPLETE" != true ]]; then + kept=$((kept + 1)) + continue + fi + + label="orphan" + [[ "$is_orphan" != true ]] && label="stale-registered" + + # Action: dry-run logs, real run removes. + if [[ "$DRY_RUN" == true ]]; then + say " WOULD REMOVE ($label): $candidate" + removed=$((removed + 1)) + continue + fi + + say " removing $label: $candidate" + + # For registered worktrees, try git's polite path first so its + # bookkeeping is cleaned up too. Capture stderr to distinguish + # success from refusal — locked worktrees MUST NOT be force-deleted + # via the rm fallback. + cleaned=false + git_stderr="" + if [[ "$is_orphan" != true ]] && _is_git_root "$REPO_ROOT"; then + git_stderr="$(git -C "$REPO_ROOT" worktree remove "$candidate" --force 2>&1 >/dev/null || true)" + if [[ -z "$git_stderr" ]]; then + cleaned=true + fi + fi + + if [[ "$cleaned" != true ]] && [[ -d "$candidate" || -L "$candidate" ]]; then + if [[ "$is_orphan" == true ]]; then + # Orphan: `.git` already broken, no ownership concern. + if rm -rf -- "$candidate"; then + cleaned=true + fi + else + # Registered worktree git refused to remove (locked, wrong + # repo, REPO_ROOT unset). DO NOT fall back to rm -rf — git's + # refusal is a liveness signal. + if [[ -n "$git_stderr" ]]; then + echo " WARN: git refused to remove registered worktree: $candidate" >&2 + echo " ($git_stderr)" >&2 + else + echo " WARN: no usable REPO_ROOT — skipping registered worktree: $candidate" >&2 + fi + refused=$((refused + 1)) + continue + fi + fi + + if [[ "$cleaned" == true ]]; then + removed=$((removed + 1)) + else + echo " ERROR: failed to remove '$candidate'" >&2 + refused=$((refused + 1)) + fi +done < <( + if [[ "$TASK_COMPLETE" == true ]]; then + # Namespace-scoped: silently no-op if the ns subtree is missing. + # mindepth/maxdepth 1 matches dream-* one level below the ns dir + # (the walk root is the ns dir itself). + # MIN_AGE_MIN=0 omits -mmin entirely: -mmin +0 means older-than-0 + # minutes, which still EXCLUDES files modified in the last ~1 + # minute. Without this special case the end-of-session sweep would + # silently miss the freshly-pushed final batch it must catch. + if [[ -d "$BASE/$NAMESPACE" ]]; then + if [[ "$MIN_AGE_MIN" -gt 0 ]]; then + find "$BASE/$NAMESPACE" -mindepth 1 -maxdepth 1 -type d -mmin "+$MIN_AGE_MIN" -print0 + else + find "$BASE/$NAMESPACE" -mindepth 1 -maxdepth 1 -type d -print0 + fi + fi + else + # Default orphan-only mode walks the whole base. Safe because the + # orphan gate (.git missing or pointer broken) cannot fire on live + # sibling-repo worktrees. Same MIN_AGE_MIN=0 special case. + if [[ "$MIN_AGE_MIN" -gt 0 ]]; then + find "$BASE" -mindepth 2 -maxdepth 2 -type d -mmin "+$MIN_AGE_MIN" -print0 + else + find "$BASE" -mindepth 2 -maxdepth 2 -type d -print0 + fi + fi +) + +say "Summary: removed=$removed kept=$kept refused=$refused" + +# --- Prune stale worktree refs (in case our rm beat git's bookkeeping) --- +if [[ "$DRY_RUN" != true ]] && _is_git_root "$REPO_ROOT"; then + git -C "$REPO_ROOT" worktree prune 2>/dev/null || true +fi + +exit 0 diff --git a/skills/shadow-frog-dream/dream-reconcile.py b/skills/shadow-frog-dream/dream-reconcile.py new file mode 100755 index 0000000..f006547 --- /dev/null +++ b/skills/shadow-frog-dream/dream-reconcile.py @@ -0,0 +1,1783 @@ +#!/usr/bin/env python3 +"""Dream reconciliation — merge dream branches into main's shadow. + +Usage: + python3 dream-reconcile.py [REPO_ROOT] [OPTIONS] + python3 dream-reconcile.py --help + +Options: + --dry-run Show what would be done without modifying files + --verify-only Only run post-reconciliation verification (checks + all dreams already in _index.md, not just unreconciled) + --namespace NS Override DREAM_NAMESPACE + --cleanup-branches Delete reconciled branches after verification. + REFUSES to run unless the reconciliation commit is + already on origin/<default-branch>. Run AFTER push. + --help, -h Show this help message + +Steps (idempotent, safe to rerun): + 1. Discover new branches (namespace-filtered, not in _index.md) + 2. Read/validate manifests from remote branches + 3. Merge discoveries into main's per-file shadows (semantic dedup) + 4. Mirror reports, manifests, patches to main's _dreams/ + 5. Update _dreams/_index.md + 6. Update _meta/state.json + 7. Rebuild top-level .shadow/_index.md (per-file discovery counts) + 8. Verify all artifacts present + 9. (Optional) Delete reconciled branches — only after push + +Exits 0 on success, 1 on any verification failure. +""" + +import json +import os +import re +import shutil +import subprocess +import sys +from datetime import datetime, timezone + +# Shared safety gate for `rm -rf <worktree>`. Lives next to this script so +# bash callers (dream-cleanup.sh, dream-gc.sh) and this module share ONE +# source of truth for the "is this path safe to remove?" rules. Imported +# lazily inside _gc_worktree_after_merge() — top-level `from .` would fail +# when this script is run directly (no package context). + +# --- Configuration --- + +EXCLUDE_PATTERNS = re.compile( + r'(^\.|/\.)' + r'|\btest[s]?/' + r'|\btest_' + r'|_test\.' + r'|\.lock$' + r'|node_modules/' + r'|vendor/' + r'|dist/' + r'|build/' + r'|__pycache__' + r'|\.min\.' +) + +# Semantic-dedup thresholds. Word-overlap heuristics over short discoveries +# are notoriously lossy ("returns None on EXPIRED tokens" vs "...REVOKED tokens" +# share 5/6 words → 83% overlap). Use a tight threshold AND require a minimum +# length so 1-word differences in short claims aren't auto-merged. +DEDUP_THRESHOLD = 0.95 +DEDUP_MIN_WORDS = 12 + +# Trust/strength orderings for metadata-merge on EXACT-text duplicates (B15). +# When two dreams independently record the SAME claim with different metadata, +# the stronger metadata must survive instead of being silently dropped. Only +# applied on exact-text matches — fuzzy matches stay skip-only. +SOURCE_TRUST = {'exploration': 1, 'interaction': 2, 'user': 3} + +# Language detection for reconciler-created shadows. Mirrors +# shadow-init.py's EXTENSION_TO_LANG / BASENAME_TO_LANG so files bootstrapped +# during reconciliation carry the same canonical `**Language**:` header as +# init-created files. +EXTENSION_TO_LANG = { + ".py": "Python", + ".js": "JavaScript", ".jsx": "JavaScript", + ".mjs": "JavaScript", ".cjs": "JavaScript", + ".ts": "TypeScript", ".tsx": "TypeScript", + ".java": "Java", + ".kt": "Kotlin", ".kts": "Kotlin", + ".scala": "Scala", + ".go": "Go", + ".rs": "Rust", + ".rb": "Ruby", + ".c": "C", ".h": "C", + ".cpp": "C++", ".hpp": "C++", ".cc": "C++", + ".hh": "C++", ".cxx": "C++", ".hxx": "C++", + ".cs": "C#", + ".php": "PHP", + ".sh": "Shell", ".bash": "Shell", ".zsh": "Shell", + ".swift": "Swift", + ".yaml": "YAML", ".yml": "YAML", + ".toml": "TOML", + ".json": "JSON", +} +BASENAME_TO_LANG = { + "Makefile": "Makefile", + "Dockerfile": "Dockerfile", + "Containerfile": "Dockerfile", + "Rakefile": "Ruby", + "Gemfile": "Ruby", +} + + +def _detect_language(rel_path): + """Detect language from a source path, mirroring shadow-init.py.""" + base = os.path.basename(rel_path) + if base in BASENAME_TO_LANG: + return BASENAME_TO_LANG[base] + _, ext = os.path.splitext(base) + return EXTENSION_TO_LANG.get(ext.lower(), "Unknown") + + +def _source_rel_from_shadow(shadow_path): + """Derive the source path (e.g. src/foo.py) from a shadow path.""" + norm = shadow_path.replace(os.sep, '/') + marker = '/.shadow/' + idx = norm.rfind(marker) + rel = norm[idx + len(marker):] if idx >= 0 else os.path.basename(norm) + if rel.endswith('.md'): + rel = rel[:-3] + return rel + + +def _canonical_header_lines(shadow_path): + """Return canonical per-file shadow header lines. + + Mirrors shadow-init.py's per-file template (`# Shadow: <path>`, + `**Language**: <lang>`, `## File-Level`) so reconciler-created shadows + match init-created ones. Without this, downstream tools read `Unknown` + for the index Language column and `--check-invariants` flags the file + as missing its metadata block. + """ + rel = _source_rel_from_shadow(shadow_path) + lang = _detect_language(rel) + return [ + f'# Shadow: {rel}\n', '\n', + f'**Language**: {lang}\n', '\n', + '## File-Level\n', '\n', '_No discoveries yet._\n', '\n', + ] + + +# --- Git helpers --- + +def git(*args, cwd=None, check=True): + """Run a git command and return stdout.""" + result = subprocess.run( + ['git'] + list(args), + capture_output=True, text=True, cwd=cwd + ) + if check and result.returncode != 0: + raise RuntimeError(f"git {' '.join(args)} failed: {result.stderr.strip()}") + return result.stdout.strip() + + +def git_show(ref, path, cwd=None): + """Read a file from a git ref. Returns None if not found.""" + result = subprocess.run( + ['git', 'show', f'{ref}:{path}'], + capture_output=True, text=True, cwd=cwd + ) + if result.returncode != 0: + return None + return result.stdout + + +# --- Step 1: Discover branches --- + +def discover_branches(repo_root, dream_ns): + """Find dream branches not yet in _index.md.""" + # Get all remote dream branches for this namespace. + # Use startswith on the short branch name to avoid false positives from + # any ref that merely contains "dream/<ns>/" as a substring. + raw = git('branch', '-r', '--format=%(refname:short)', cwd=repo_root) + prefix = f'dream/{dream_ns}/' + all_branches = [] + for b in raw.split('\n'): + b = b.strip() + if not b: + continue + short = b[len('origin/'):] if b.startswith('origin/') else b + if short.startswith(prefix): + all_branches.append(short) + + existing_ids = _read_indexed_dream_ids(repo_root) + + # Filter to new branches (dream_id not in index) + new_branches = [] + for branch in all_branches: + # Extract dream_id from branch name (everything after the prefix) + dream_id = branch[len(prefix):] + if dream_id and dream_id not in existing_ids: + new_branches.append((branch, dream_id)) + + return new_branches + + +def _read_indexed_dream_ids(repo_root): + """Return the set of dream_ids already present in _index.md.""" + index_path = os.path.join(repo_root, '.shadow', '_dreams', '_index.md') + existing = set() + if not os.path.isfile(index_path): + return existing + with open(index_path) as f: + for line in f: + if line.startswith('|') and not line.startswith('| dream_id') and not line.startswith('|---'): + parts = [p.strip() for p in line.split('|')] + if len(parts) > 1 and parts[1]: + existing.add(parts[1]) + return existing + + +def _read_indexed_branches(repo_root): + """Return list of (branch, dream_id) tuples from _index.md (column 5).""" + index_path = os.path.join(repo_root, '.shadow', '_dreams', '_index.md') + rows = [] + if not os.path.isfile(index_path): + return rows + with open(index_path) as f: + for line in f: + if line.startswith('|') and not line.startswith('| dream_id') and not line.startswith('|---'): + parts = [p.strip() for p in line.split('|')] + # Leading '|' creates an empty parts[0]; columns start at parts[1]. + if len(parts) >= 6 and parts[1]: + dream_id = parts[1] + branch = parts[5] + if branch: + rows.append((branch, dream_id)) + return rows + + +# --- Step 2: Read/validate manifests --- + +def load_manifests(repo_root, branches): + """Read and validate manifests from remote branches.""" + manifests = [] + skipped = [] + + for branch, dream_id in branches: + manifest_path = f'.shadow/_dreams/{dream_id}/manifest.json' + raw = git_show(f'origin/{branch}', manifest_path, cwd=repo_root) + + if not raw: + skipped.append((branch, dream_id, "no manifest found")) + continue + + try: + manifest = json.loads(raw) + except json.JSONDecodeError as e: + skipped.append((branch, dream_id, f"invalid JSON: {e}")) + continue + + # Validate dream_id consistency + m_did = manifest.get('dream_id', '') + if m_did != dream_id: + skipped.append((branch, dream_id, f"dream_id mismatch: {m_did}")) + continue + + # Validate required fields + if not manifest.get('category') or not manifest.get('verdict'): + skipped.append((branch, dream_id, "missing category or verdict")) + continue + + manifests.append((branch, dream_id, manifest)) + + return manifests, skipped + + +# --- Step 3: Semantic merge --- + +def find_heading(lines, symbol): + """Find the line index of a heading matching the symbol (bare or backtick-wrapped).""" + patterns = [ + f'## `{symbol}`', + f'### `{symbol}`', + f'## {symbol}', + f'### {symbol}', + ] + for i, line in enumerate(lines): + stripped = line.rstrip() + for pat in patterns: + if stripped == pat: + return i + return -1 + + +def find_cross_references_heading(lines): + """Find the ## Cross-References heading. + + Case-insensitive: meditate/user rewrites that lowercase the heading + must still be detected, otherwise `_ensure_cross_references_section` + appends a duplicate section and back-pointer dedup misses entirely. + """ + for i, line in enumerate(lines): + if line.strip().lower() == '## cross-references': + return i + return -1 + + +def is_duplicate_discovery(existing_lines, new_text): + """Check if a discovery with similar text already exists. + + Uses word-overlap (Jaccard-ish, scaled by the new text). Tight + threshold (DEDUP_THRESHOLD) and a minimum length (DEDUP_MIN_WORDS) avoid + falsely merging short discoveries that differ by a single keyword + (e.g., "expired" vs "revoked"). Exact-match (after whitespace + normalization) is always treated as duplicate regardless of length. + """ + normalized_new = re.sub(r'\s+', ' ', new_text.lower().strip()) + new_words = normalized_new.split() + if not new_words: + return False + + for line in existing_lines: + if not line.startswith('- '): + continue + normalized_existing = re.sub(r'\s+', ' ', line[2:].lower().strip()) + + # Exact-match short-circuit (independent of word count). + if normalized_existing == normalized_new: + return True + + existing_words = normalized_existing.split() + # Skip the fuzzy heuristic for short discoveries — it has high false + # positive rate on 1-word differences (5/6 ≈ 83% but distinct meaning). + if len(new_words) < DEDUP_MIN_WORDS or len(existing_words) < DEDUP_MIN_WORDS: + continue + + new_set = set(new_words) + existing_set = set(existing_words) + overlap = len(new_set & existing_set) / len(new_set) + if overlap >= DEDUP_THRESHOLD: + return True + return False + + +def _ensure_cross_references_section(lines): + """Append a ## Cross-References section if missing. Returns updated lines.""" + if find_cross_references_heading(lines) >= 0: + return lines + # Ensure trailing newline before adding section + if lines and not lines[-1].endswith('\n'): + lines[-1] = lines[-1] + '\n' + if lines and lines[-1].strip(): + lines.append('\n') + lines.extend(['## Cross-References\n', '\n', '_No cross-cutting discoveries yet._\n']) + return lines + + +def _find_exact_discovery_index(section_lines, new_text): + """Return the index (within section_lines) of a `- ` discovery line whose + text EXACTLY matches new_text (whitespace/case normalized), else -1. + + Exact match is the only case eligible for metadata-merge — the fuzzy + overlap heuristic is too lossy to trust for silently rewriting metadata. + """ + normalized_new = re.sub(r'\s+', ' ', new_text.lower().strip()) + if not normalized_new: + return -1 + for i, line in enumerate(section_lines): + if not line.startswith('- '): + continue + if re.sub(r'\s+', ' ', line[2:].lower().strip()) == normalized_new: + return i + return -1 + + +def _parse_meta_line(meta_line): + """Parse a ` _(<status>, source: <src>[, labels: [..]])_` line. + + Returns (status, source, labels) or None if the line isn't a metadata + line in the canonical shape. + """ + m = re.match(r'\s*_\((.*)\)_\s*$', meta_line.rstrip('\n')) + if not m: + return None + inner = m.group(1) + sm = re.search(r'\b(verified|uncertain|refuted)\b', inner) + status = sm.group(1) if sm else None + src_m = re.search(r'source:\s*([A-Za-z]+)', inner) + source = src_m.group(1) if src_m else None + lbl_m = re.search(r'labels:\s*\[([^\]]*)\]', inner) + labels = [] + if lbl_m: + labels = [l.strip() for l in lbl_m.group(1).split(',') if l.strip()] + if status is None or source is None: + return None + return status, source, labels + + +def _merge_meta(existing, new_status, new_source, new_labels): + """Compute the upgraded (status, source, labels) for an exact-text dup. + + Rules (B15): union labels; upgrade source to the higher-trust value; + upgrade `uncertain`->`verified`; NEVER silently change to/from `refuted` + (status conflicts are meditate's job). Returns (status, source, labels, + changed). + """ + e_status, e_source, e_labels = existing + + status = e_status + if e_status != 'refuted' and new_status != 'refuted': + if e_status == 'uncertain' and new_status == 'verified': + status = 'verified' + + source = e_source + if SOURCE_TRUST.get(new_source, 0) > SOURCE_TRUST.get(e_source, 0): + source = new_source + + labels = list(e_labels) + for l in new_labels: + if l not in labels: + labels.append(l) + + changed = (status != e_status or source != e_source or labels != e_labels) + return status, source, labels, changed + + +def _format_meta_line(status, source, labels): + parts = [status, f'source: {source}'] + if labels: + parts.append(f"labels: [{', '.join(labels)}]") + return f' _({", ".join(parts)})_\n' + + +def merge_discovery_into_file(shadow_path, anchor_symbol, discovery, dream_id): + """Merge a single discovery into a shadow file. Returns True if written.""" + text = discovery.get('text', '').strip() + if not text: + return False + + status = discovery.get('status', 'verified') + source = discovery.get('source', 'exploration') + labels = discovery.get('labels', []) + also_involves = discovery.get('also_involves', []) + + # Build the discovery line + meta_parts = [status, f'source: {source}'] + if labels: + meta_parts.append(f"labels: [{', '.join(labels)}]") + meta_line = f' _({", ".join(meta_parts)})_' + + lines_to_add = [f'- {text}\n', f'{meta_line}\n'] + if also_involves: + refs = ', '.join(f'`{r}`' for r in also_involves) + lines_to_add.append(f' Also involves: {refs}\n') + lines_to_add.append(f' Dream report: `_dreams/{dream_id}/`\n') + + # Read or create shadow file + new_file = not os.path.isfile(shadow_path) + if new_file: + # Bootstrap with the canonical layout: file header + File-Level + # section + symbol heading + Cross-References footer, matching + # shadow-init.py's per-file template. + os.makedirs(os.path.dirname(shadow_path), exist_ok=True) + lines = _canonical_header_lines(shadow_path) + [ + f'## `{anchor_symbol}`\n', '\n', + '## Cross-References\n', '\n', '_No cross-cutting discoveries yet._\n', + ] + else: + with open(shadow_path) as f: + lines = f.readlines() + lines = _ensure_cross_references_section(lines) + + # Check for duplicate + heading_idx = find_heading(lines, anchor_symbol) + if heading_idx >= 0: + # Find the section content (until next heading or end) + section_end = len(lines) + for i in range(heading_idx + 1, len(lines)): + if lines[i].startswith('## ') or lines[i].startswith('### '): + section_end = i + break + section_lines = lines[heading_idx:section_end] + + # Exact-text duplicate: don't drop the new discovery's metadata — + # merge it into the existing line (union labels, upgrade source trust, + # upgrade uncertain->verified; never touch refuted). B15. + exact_rel = _find_exact_discovery_index(section_lines, text) + if exact_rel >= 0: + text_idx = heading_idx + exact_rel + meta_idx = text_idx + 1 + if meta_idx < section_end: + existing_meta = _parse_meta_line(lines[meta_idx]) + else: + existing_meta = None + if existing_meta is None: + # No canonical metadata line to upgrade — nothing safe to do. + return False + merged = _merge_meta(existing_meta, status, source, labels) + m_status, m_source, m_labels, changed = merged + if not changed: + return False + lines[meta_idx] = _format_meta_line(m_status, m_source, m_labels) + with open(shadow_path, 'w') as f: + f.writelines(lines) + return True + + # Fuzzy near-duplicate: skip (heuristic too lossy to merge metadata). + if is_duplicate_discovery(section_lines, text): + return False + + # Remove placeholder if present + for i in range(heading_idx + 1, section_end): + if '_No discoveries yet._' in lines[i]: + lines[i] = '' + break + + # Insert discovery before next heading or cross-references + insert_at = section_end + lines[insert_at:insert_at] = ['\n'] + lines_to_add + else: + # Create heading before Cross-References (or at end) + xref_idx = find_cross_references_heading(lines) + if xref_idx >= 0: + insert_at = xref_idx + else: + insert_at = len(lines) + + new_section = ['\n', f'## `{anchor_symbol}`\n', '\n'] + lines_to_add + lines[insert_at:insert_at] = new_section + + with open(shadow_path, 'w') as f: + f.writelines(lines) + + return True + + +def add_cross_reference_backpointer(repo_root, file_part, slug, title, dream_id): + """Add a back-pointer to per-file shadow's ## Cross-References section. + + Required by the bidirectional-reference invariant: every entry in + `_cross/<slug>.md` must have a matching entry in each referenced file's + ## Cross-References section. Idempotent (skips if back-pointer exists). + """ + shadow_path = os.path.join(repo_root, '.shadow', file_part + '.md') + # Relative link from .shadow/<file_part>.md back up to .shadow/_cross/<slug>.md. + # For a top-level file (no slashes) the prefix is empty; each directory of + # depth adds one "../". Otherwise the markdown link is broken and the + # bidirectional-reference invariant fails on any subdir shadow. + depth = file_part.count('/') + prefix = '../' * depth + backpointer = f'- [{title}]({prefix}_cross/{slug}.md) (dream: {dream_id})' + + if os.path.isfile(shadow_path): + with open(shadow_path) as f: + lines = f.readlines() + else: + # Create a minimal shadow file with the canonical header + footer if + # the cross-reference predates per-file analysis. The specific symbols + # are unknown here, so the `## File-Level` section (which + # _canonical_header_lines emits) holds file-scope content. + os.makedirs(os.path.dirname(shadow_path), exist_ok=True) + lines = _canonical_header_lines(shadow_path) + [ + '## Cross-References\n', '\n', '_No cross-cutting discoveries yet._\n', + ] + + lines = _ensure_cross_references_section(lines) + + # Idempotency: only treat as "already present" when the line contains + # the actual markdown link target — `](<prefix>_cross/<slug>.md)`. + # Substring matching on `_cross/{slug}.md` false-positives on any + # discovery body that mentions the slug (e.g. an `Also involves:` ref + # like `\`_cross/db-lifecycle.md::section\``), silently swallowing the + # legitimate back-pointer add. + link_marker = f']({prefix}_cross/{slug}.md)' + if any(link_marker in line for line in lines): + with open(shadow_path, 'w') as f: + f.writelines(lines) + return False + + xref_idx = find_cross_references_heading(lines) + # Find end of Cross-References section + section_end = len(lines) + for i in range(xref_idx + 1, len(lines)): + if lines[i].startswith('## ') or lines[i].startswith('### '): + section_end = i + break + + # Replace the empty placeholder if present, else append. + placeholder_idx = -1 + for i in range(xref_idx + 1, section_end): + if '_No cross-cutting discoveries yet._' in lines[i]: + placeholder_idx = i + break + + if placeholder_idx >= 0: + lines[placeholder_idx] = f'{backpointer}\n' + else: + # Insert before the next heading (or at section_end) + insert_at = section_end + # Trim trailing blank lines inside the section + while insert_at > xref_idx + 1 and lines[insert_at - 1].strip() == '': + insert_at -= 1 + lines[insert_at:insert_at] = [f'{backpointer}\n'] + + with open(shadow_path, 'w') as f: + f.writelines(lines) + return True + + +def _merge_refs_into_cross_file(cross_path, new_refs): + """Union new refs into an existing _cross/<slug>.md **Refs**: section. + + When two dreams use the same cross-cutting slug, the later one must not + silently drop its refs: per-file back-pointers are still added pointing + at this cross file (below), so its **Refs**: block must list them or the + bidirectional-reference invariant breaks. Returns True if modified. + """ + try: + with open(cross_path) as f: + content = f.read() + except OSError: + return False + lines = content.split('\n') + refs_idx = None + for i, line in enumerate(lines): + if line.strip().lower().startswith('**refs**:'): + refs_idx = i + break + if refs_idx is None: + return False + block_end = refs_idx + 1 + existing = set() + while block_end < len(lines): + m = re.match(r'-\s*`([^`]+)`', lines[block_end].strip()) + if m: + existing.add(m.group(1)) + block_end += 1 + else: + break + to_add = [r for r in new_refs if r and r not in existing] + if not to_add: + return False + lines[block_end:block_end] = [f'- `{r}`' for r in to_add] + with open(cross_path, 'w') as f: + f.write('\n'.join(lines)) + return True + + +def merge_discoveries(repo_root, manifests, dry_run=False): + """Merge all discoveries from manifests into main's shadow files.""" + merged_count = 0 + skipped_count = 0 + + for branch, dream_id, manifest in manifests: + discoveries = manifest.get('discoveries', []) + for disc in discoveries: + # Normalize string discoveries to dicts + if isinstance(disc, str): + disc = {'anchor': '', 'text': disc} + anchor = disc.get('anchor', '') + if '::' not in anchor: + skipped_count += 1 + continue + + file_part, symbol = anchor.split('::', 1) + shadow_path = os.path.join(repo_root, '.shadow', file_part + '.md') + + if dry_run: + print(f" Would merge: {anchor} <- {disc.get('text', '')[:60]}") + merged_count += 1 + continue + + # Ensure shadow directory exists + os.makedirs(os.path.dirname(shadow_path), exist_ok=True) + + if merge_discovery_into_file(shadow_path, symbol, disc, dream_id): + merged_count += 1 + else: + skipped_count += 1 + + # Handle cross-cutting discoveries + cross_cutting = manifest.get('cross_cutting', []) + for cross in cross_cutting: + # Normalize string entries to dicts + if isinstance(cross, str): + cross = {'slug': re.sub(r'[^a-z0-9]+', '-', cross[:60].lower()).strip('-'), 'description': cross} + slug = cross.get('slug', '') + if not slug: + continue + + cross_path = os.path.join(repo_root, '.shadow', '_cross', f'{slug}.md') + refs = cross.get('refs', []) or [] + title = cross.get('title', slug) + + if dry_run: + print(f" Would create cross-cutting: _cross/{slug}.md") + for ref in refs: + if '::' in ref: + file_part = ref.split('::', 1)[0] + print(f" + back-pointer in .shadow/{file_part}.md") + merged_count += 1 + continue + + if not os.path.isfile(cross_path): + os.makedirs(os.path.dirname(cross_path), exist_ok=True) + refs_str = '\n'.join(f'- `{r}`' for r in refs) + content = ( + f"# {title}\n\n" + f"**Category**: {cross.get('category', 'behavior')}\n" + f"**Refs**:\n{refs_str}\n\n" + f"**Discovery**: {cross.get('text', '')}\n\n" + f"_({cross.get('status', 'verified')}, " + f"source: {cross.get('source', 'exploration')})_\n" + ) + with open(cross_path, 'w') as f: + f.write(content) + merged_count += 1 + else: + # Cross file already exists (e.g. a prior dream used the same + # slug). Union our refs into its **Refs**: block so it stays + # consistent with the back-pointers added below. + if _merge_refs_into_cross_file(cross_path, refs): + merged_count += 1 + else: + skipped_count += 1 + + # Maintain bidirectional invariant: add a back-pointer in each + # referenced per-file shadow's ## Cross-References section. + # We do this even when the cross-cutting file already exists, so + # that re-runs heal any missing back-pointers. + for ref in refs: + if '::' not in ref: + continue + file_part = ref.split('::', 1)[0] + try: + add_cross_reference_backpointer( + repo_root, file_part, slug, title, dream_id + ) + except OSError as e: + print(f" ⚠️ Could not write back-pointer for {file_part}: {e}") + + return merged_count, skipped_count + + +# --- Step 4: Mirror reports --- + +def mirror_reports(repo_root, manifests, dry_run=False): + """Copy report.md, manifest.json, patch.diff from branches to main.""" + mirrored = 0 + corrupted = [] + + for branch, dream_id, manifest in manifests: + dream_dir = os.path.join(repo_root, '.shadow', '_dreams', dream_id) + + if dry_run: + print(f" Would mirror: {dream_id}/") + mirrored += 1 + continue + + os.makedirs(dream_dir, exist_ok=True) + + # Read report and check for corruption + report = git_show(f'origin/{branch}', f'.shadow/_dreams/{dream_id}/report.md', cwd=repo_root) + if report: + # Verify report dream_id matches + m = re.match(r'^\ufeff?\s*---\r?\n(.*?)\r?\n---', report, re.S) + report_corrupt = False + if m: + dm = re.search(r'^dream_id:\s*["\']?(.+?)["\']?\s*$', m.group(1), re.M) + report_did = dm.group(1).strip() if dm else '' + if report_did and report_did != dream_id: + corrupted.append((dream_id, report_did)) + report_corrupt = True + # Write placeholder for the report only — the manifest + # and patch below are still mirrored unconditionally so a + # single bad frontmatter line never discards valid + # artifacts (discoveries are read from the manifest). + with open(os.path.join(dream_dir, 'report.md'), 'w') as f: + f.write(f"# Corrupted Report\n\nContained content from {report_did}.\n" + f"Original on branch: {branch}\n") + + if not report_corrupt: + with open(os.path.join(dream_dir, 'report.md'), 'w') as f: + f.write(report) + + # Mirror manifest + with open(os.path.join(dream_dir, 'manifest.json'), 'w') as f: + json.dump(manifest, f, indent=2) + + # Mirror patch + # + # Use `is not None` (not a truthy check): git_show returns None when + # the file is absent on the ref, and "" when the file exists but is + # 0 bytes. Truthy collapses both into "skip", which loses information + # — a legitimately empty patch.diff on the dream branch would never + # be mirrored to main, and verify_artifacts then reports the dream as + # `missing patch.diff` even though it exists on origin. Always + # mirror the file when the ref had one (even if empty); skip the + # write only when truly absent. + patch = git_show(f'origin/{branch}', f'.shadow/_dreams/{dream_id}/patch.diff', cwd=repo_root) + if patch is not None: + with open(os.path.join(dream_dir, 'patch.diff'), 'w') as f: + f.write(patch) + + mirrored += 1 + + return mirrored, corrupted + + +# --- Step 5: Update index --- + +def _resolve_tip_commit(repo_root, branch): + """Return the short SHA for origin/<branch>, or 'unknown' on failure. + + Avoids writing empty/garbage values into _index.md when the ref is + pruned, the network is down, or git is otherwise unhappy. Uses git's + own short-SHA length (git auto-extends past 7 when 7 would be + ambiguous) instead of truncating, so the stored value always resolves + uniquely. Validates that the result looks like a hex SHA. + """ + raw = git('rev-parse', '--short', f'origin/{branch}', + cwd=repo_root, check=False) + candidate = raw.strip().split('\n', 1)[0] if raw else '' + if candidate and re.fullmatch(r'[0-9a-fA-F]{7,40}', candidate): + return candidate + return 'unknown' + + +def update_index(repo_root, manifests, dry_run=False): + """Add entries to _dreams/_index.md for reconciled branches.""" + index_path = os.path.join(repo_root, '.shadow', '_dreams', '_index.md') + + if dry_run: + for branch, dream_id, manifest in manifests: + print(f" Would index: {dream_id}") + return + + # Bootstrap if missing (skipped on dry-run so the directory tree + # stays clean — dry-run must not mutate disk). + if not os.path.isfile(index_path): + os.makedirs(os.path.dirname(index_path), exist_ok=True) + with open(index_path, 'w') as f: + f.write('# Dream Experiment Archive\n\n' + '| dream_id | category | verdict | title | branch | parent | tip_commit |\n' + '|----------|----------|---------|-------|--------|--------|------------|\n') + + with open(index_path, 'a') as f: + for branch, dream_id, manifest in manifests: + tip = _resolve_tip_commit(repo_root, branch) + cat = re.sub(r'\s*\(.*\)\s*$', '', manifest.get('category', 'unknown').lower().strip()) + verdict = manifest.get('verdict', 'unknown').lower().strip() + parent = manifest.get('parent_branch', 'main').strip() + + # Get title from manifest or report heading + title = manifest.get('title', '') + if not title: + report = git_show(f'origin/{branch}', + f'.shadow/_dreams/{dream_id}/report.md', cwd=repo_root) + if report: + fm_end = report.find('---', report.find('---') + 3) + body = report[fm_end + 3:] if fm_end > 0 else report + m = re.search(r'^#\s+(.+)', body, re.M) + title = m.group(1).strip() if m else '' + if not title: + title = f'Dream {dream_id}' + title = title.replace('|', '-').replace('\n', ' ')[:120] + + f.write(f'| {dream_id} | {cat} | {verdict} | {title} | {branch} | {parent} | {tip} |\n') + + +# --- Shared: discovery counting --- + +def _count_discoveries(shadow_path): + """Count per-file discoveries in a shadow file. + + Bullets inside `## Cross-References` are back-pointer links to + `_cross/*.md`, not discoveries — exclude them. Heading lookahead is + case-insensitive so meditate/user-rewritten lowercase headings still + delimit the section correctly (matches `find_cross_references_heading`). + """ + if not os.path.isfile(shadow_path): + return 0 + count = 0 + with open(shadow_path) as sf: + in_xref = False + for line in sf: + stripped = line.rstrip() + if stripped.lower().startswith('## cross-references'): + in_xref = True + continue + if stripped.startswith('## ') or stripped.startswith('### '): + in_xref = False + continue + if not in_xref and line.startswith('- '): + count += 1 + return count + + +# --- Step 6: Update state.json --- + +def update_state(repo_root, manifests, dry_run=False): + """Update _meta/state.json with dream reconciliation metadata.""" + state_path = os.path.join(repo_root, '.shadow', '_meta', 'state.json') + + if not os.path.isfile(state_path): + if dry_run: + print(" Would create state.json") + return + os.makedirs(os.path.dirname(state_path), exist_ok=True) + state = { + 'version': 1, + 'initialized_at': datetime.now(timezone.utc).isoformat(), + 'total_files': 0, + 'total_symbols': 0, + 'total_discoveries': 0, + 'dream_cycles_completed': 0, + } + else: + with open(state_path) as f: + state = json.load(f) + + if dry_run: + print(" Would update state.json") + return + + # Recount discoveries + total_discoveries = 0 + total_files = 0 + total_symbols = 0 + shadow_dir = os.path.join(repo_root, '.shadow') + + for root, dirs, files in os.walk(shadow_dir): + # Prune `_*` internal directories (_meta, _cross, _dreams) in-place + # so os.walk never descends into them. Without this, large _dreams/ + # archives are walked on every reconcile for no benefit. Internal dirs + # only exist at the top level, so prune there only — deeper `_`-prefixed + # source dirs (e.g. src/_internal/) are real shadows and must be counted. + if root == shadow_dir: + dirs[:] = [d for d in dirs if not d.startswith('_')] + + for fname in files: + if not fname.endswith('.md'): + continue + filepath = os.path.join(root, fname) + rel_path = os.path.relpath(filepath, shadow_dir) + if rel_path.startswith('_'): + continue + + total_files += 1 + with open(filepath) as f: + # `## Cross-References` and `## File-Level` are structural + # sections, not symbols. Bullets inside `## Cross-References` + # are back-pointer links to `_cross/*.md`, not discoveries. + in_xref = False + for line in f: + stripped = line.rstrip() + if stripped.startswith('## ') or stripped.startswith('### '): + prefix_len = 4 if stripped.startswith('### ') else 3 + heading_text = stripped[prefix_len:].strip().strip('`').strip() + if heading_text.lower() == 'cross-references': + in_xref = True + continue + in_xref = False + if heading_text == 'File-Level': + continue + total_symbols += 1 + continue + if not in_xref and line.startswith('- '): + total_discoveries += 1 + + state['last_update_at'] = datetime.now(timezone.utc).isoformat() + state['last_update_type'] = 'dream' + state['total_files'] = total_files + state['total_symbols'] = total_symbols + state['total_discoveries'] = total_discoveries + state['dream_cycles_completed'] = state.get('dream_cycles_completed', 0) + 1 + + # Record last commit + try: + state['last_commit'] = git('rev-parse', 'HEAD', cwd=repo_root) + except RuntimeError: + pass + + with open(state_path, 'w') as f: + json.dump(state, f, indent=2) + + +# --- Step 7: Rebuild top-level _index.md --- + +# Container kinds whose heading text gets a `<kind> Name` prefix (per +# shadow-init.py's Symbol.heading_text). We strip the prefix to surface the +# bare class/interface name in the index table. +_HEADING_KIND_PREFIXES = ( + 'class ', 'interface ', 'enum ', 'trait ', + 'struct ', 'protocol ', 'module ', +) + + +def _shadow_symbol_names(shadow_path): + """Extract top-level symbol names from a shadow file. + + Returns names from `## `name`` headings, skipping the structural + `## Cross-References` and `## File-Level` sections. Nested `###` + headings (e.g. methods inside a class) are NOT counted here — the + top-level index row lists only top-level symbols (matching + shadow-init.py's `build_index`, which iterates symbols with no parent). + """ + names = [] + if not os.path.isfile(shadow_path): + return names + with open(shadow_path) as sf: + for line in sf: + stripped = line.rstrip() + if not stripped.startswith('## '): + continue + heading_text = stripped[3:].strip().strip('`').strip() + lower = heading_text.lower() + if lower == 'cross-references' or lower == 'file-level': + continue + for kw in _HEADING_KIND_PREFIXES: + if heading_text.startswith(kw): + heading_text = heading_text[len(kw):].strip() + break + if heading_text: + names.append(heading_text) + return names + + +def _shadow_language(shadow_path): + """Read the `**Language**:` field from a shadow file header. + + Returns 'Unknown' if the header line is missing (e.g. a shadow created + on-the-fly by `add_cross_reference_backpointer` without metadata). + Avoids re-implementing shadow-init's extension map here. + """ + if not os.path.isfile(shadow_path): + return 'Unknown' + with open(shadow_path) as sf: + for i, line in enumerate(sf): + if i > 10: + break + m = re.match(r'\*\*Language\*\*:\s*([^|]+?)\s*(\||$)', line.rstrip()) + if m: + return m.group(1).strip() or 'Unknown' + return 'Unknown' + + +def rebuild_top_index(repo_root, dry_run=False): + """Regenerate `.shadow/_index.md` from current per-file shadow state. + + `update_state` (Step 6) already refreshes state.json totals, but the + top-level `_index.md` table — per-file symbol/discovery counts the + viewer and hooks rely on — is otherwise frozen at init time and goes + stale after every dream reconcile. This step is the missing companion + to `update_state`: it walks `.shadow/` (excluding `_*` internal dirs), + re-counts per-file discoveries via `_count_discoveries`, and rewrites + the table. + + Bootstraps a fresh `_index.md` if the file doesn't exist. Preserves + the original `> Generated by shadow-frog-init on <date>` line as + `> Initially generated by shadow-frog-init on <date>` when found so the + init provenance survives reconciler rewrites. Honors `dry_run`. + """ + shadow_dir = os.path.join(repo_root, '.shadow') + index_path = os.path.join(shadow_dir, '_index.md') + + if dry_run: + print(" Would regenerate _index.md") + return + + if not os.path.isdir(shadow_dir): + print(" No .shadow/ directory — skipping _index.md") + return + + # Salvage the original init date from any pre-existing index so we don't + # lose the "first seen" provenance when reconciler rewrites the header. + original_init_date = None + if os.path.isfile(index_path): + try: + with open(index_path) as f: + for line in f: + m = re.match( + r'>\s*(?:Initially g|G)enerated by shadow-frog-init on (\S+)', + line, + ) + if m: + original_init_date = m.group(1).strip() + break + except OSError: + pass + + rows = [] # (rel_source_path, language, names, sym_count, disc_count) + total_symbols = 0 + total_discoveries = 0 + + for root, dirs, files in os.walk(shadow_dir): + # Prune `_*` internal directories so we never descend into `_meta`, + # `_cross`, `_dreams`, etc. These only exist at the top level, so prune + # there only — deeper `_`-prefixed source dirs (e.g. src/_internal/) are + # real mirrored shadows and belong in the index. Mutating `dirs` + # in-place is the documented `os.walk` way to skip subtrees. + if root == shadow_dir: + dirs[:] = sorted(d for d in dirs if not d.startswith('_')) + else: + dirs[:] = sorted(dirs) + + for fname in sorted(files): + if not fname.endswith('.md'): + continue + if fname.startswith('_'): + continue + shadow_path = os.path.join(root, fname) + rel_shadow = os.path.relpath(shadow_path, shadow_dir) + # The shadow path mirrors the source path with `.md` appended. + source_path = rel_shadow[:-3] + + language = _shadow_language(shadow_path) + names = _shadow_symbol_names(shadow_path) + sym_count = len(names) + disc_count = _count_discoveries(shadow_path) + + total_symbols += sym_count + total_discoveries += disc_count + rows.append((source_path, language, names, sym_count, disc_count)) + + rows.sort(key=lambda r: r[0]) + + cross_dir = os.path.join(shadow_dir, '_cross') + cross_count = 0 + if os.path.isdir(cross_dir): + try: + cross_count = sum( + 1 for f in os.listdir(cross_dir) + if f.endswith('.md') and not f.startswith('_') + ) + except OSError: + pass + + # Pull dream-cycle count from state.json (already updated by Step 6). + dream_cycles = 0 + state_path = os.path.join(shadow_dir, '_meta', 'state.json') + if os.path.isfile(state_path): + try: + with open(state_path) as f: + dream_cycles = int(json.load(f).get('dream_cycles_completed', 0) or 0) + except (json.JSONDecodeError, OSError, ValueError, TypeError): + dream_cycles = 0 + + today = datetime.now(timezone.utc).strftime('%Y-%m-%d') + total_files = len(rows) + + out = ['# Shadow Index', ''] + if original_init_date: + out.append(f'> Initially generated by shadow-frog-init on {original_init_date}') + out.append(f'> Last updated by shadow-frog-dream on {today}') + totals = ( + f'> Total files: {total_files} | Symbols: {total_symbols} ' + f'| Discoveries: {total_discoveries} | Cross-cutting: {cross_count}' + ) + if dream_cycles > 0: + totals += f' | Dream cycles: {dream_cycles}' + out.append(totals) + out.append('') + out.append('| File | Language | Symbols | Discoveries |') + out.append('|------|----------|---------|-------------|') + + for rel_path, language, names, sym_count, disc_count in rows: + if sym_count == 0: + sym_display = '0' + elif len(names) <= 3: + sym_display = f"{sym_count} ({', '.join(names)})" + else: + sym_display = f"{sym_count} ({', '.join(names[:3])}, ...)" + out.append(f'| {rel_path} | {language} | {sym_display} | {disc_count} |') + + out.append('') + + try: + with open(index_path, 'w') as f: + f.write('\n'.join(out)) + except OSError as e: + print(f" ⚠️ Could not write _index.md: {e}") + return + + print(f" Index: {total_files} files, {total_discoveries} discoveries") + + +# --- Step 8: Verify --- + +def verify_reconciliation(repo_root, manifests): + """Verify all artifacts are present on main. Returns list of failures. + + Index-membership uses the parsed `dream_id` column (via + `_read_indexed_dream_ids`) rather than `dream_id in f.read()`. A naive + substring check false-positives on shared prefixes (e.g. dream_id + `20260420-1400-foo` would appear "indexed" merely because the index + contains `20260420-14001-bar`), silently swallowing missing-index bugs. + """ + failures = [] + + index_path = os.path.join(repo_root, '.shadow', '_dreams', '_index.md') + index_exists = os.path.isfile(index_path) + indexed_ids = _read_indexed_dream_ids(repo_root) if index_exists else set() + + for branch, dream_id, manifest in manifests: + dream_dir = os.path.join(repo_root, '.shadow', '_dreams', dream_id) + + for required in ['report.md', 'manifest.json', 'patch.diff']: + if not os.path.isfile(os.path.join(dream_dir, required)): + failures.append(f"{dream_id}: missing {required}") + + if not index_exists: + failures.append(f"{dream_id}: _index.md does not exist") + elif dream_id not in indexed_ids: + failures.append(f"{dream_id}: missing from _index.md") + + return failures + + +# --- Step 9: Cleanup branches --- + +def cleanup_branches(repo_root, manifests, dream_ns, dry_run=False): + """Delete reconciled dream branches (local and remote). + + Only deletes a branch if: + - The reconciliation commit is already on origin/<default-branch> + (so the artifacts the branch carries are durably persisted) + - All 3 artifacts exist on main (report.md, manifest.json, patch.diff) + - The dream_id appears in _index.md + - No un-reconciled branches list this branch as parent + + Returns (deleted, kept) counts. + """ + index_path = os.path.join(repo_root, '.shadow', '_dreams', '_index.md') + indexed_ids = _read_indexed_dream_ids(repo_root) if os.path.isfile(index_path) else set() + + # Check if SHADOWFROG_KEEP_BRANCHES is set + if os.environ.get('SHADOWFROG_KEEP_BRANCHES', '').strip() in ('1', 'true', 'yes'): + print(" SHADOWFROG_KEEP_BRANCHES is set — skipping cleanup.") + return 0, len(manifests) + + # Safety check 0: refuse cleanup unless reconciliation commit is on + # origin/<default-branch>. Otherwise a single failed `git push` would + # destroy the only copy of the discoveries. + if not dry_run: + # Safety check 0a: refuse cleanup while .shadow/ has uncommitted + # changes. In the combined `reconcile --cleanup-branches` invocation + # the merge (Steps 3-9) writes discoveries into the WORKING TREE only + # — HEAD has not moved yet — so the ancestor check below passes + # trivially against the pre-reconciliation HEAD while the merged + # discoveries are still unpersisted. Deleting the dream branches here + # would destroy the only durable copy. A dirty .shadow/ is the + # signal that reconciliation output has not been committed + pushed. + shadow_status = subprocess.run( + ['git', 'status', '--porcelain=v1', '--', '.shadow/'], + capture_output=True, text=True, cwd=repo_root + ) + if shadow_status.stdout.strip(): + print(f" ❌ Refusing cleanup: .shadow/ has uncommitted changes.", file=sys.stderr) + print(f" Commit and push the reconciliation first:", file=sys.stderr) + print(f" git add .shadow/ && git commit && git push", file=sys.stderr) + print(f" then re-run with --cleanup-branches.", file=sys.stderr) + return 0, len(manifests) + + try: + default_branch = git( + 'symbolic-ref', '--short', 'refs/remotes/origin/HEAD', + cwd=repo_root, check=False + ) + default_branch = default_branch.replace('origin/', '').strip() or 'main' + except RuntimeError: + default_branch = 'main' + + head_sha = git('rev-parse', 'HEAD', cwd=repo_root, check=False) + ancestor_check = subprocess.run( + ['git', 'merge-base', '--is-ancestor', + head_sha, f'origin/{default_branch}'], + capture_output=True, text=True, cwd=repo_root + ) + if ancestor_check.returncode != 0: + print(f" ❌ Refusing cleanup: HEAD ({head_sha[:7]}) is NOT an", file=sys.stderr) + print(f" ancestor of origin/{default_branch}.", file=sys.stderr) + print(f" Run `git push` first so the discoveries are durable,", file=sys.stderr) + print(f" then re-run with --cleanup-branches.", file=sys.stderr) + return 0, len(manifests) + + # Find all remaining dream branches (to check for descendants) + all_branches_raw = git('branch', '-r', '--format=%(refname:short)', + cwd=repo_root, check=False) + prefix = f'dream/{dream_ns}/' + all_remote_branches = set() + for b in all_branches_raw.split('\n'): + b = b.strip() + if not b: + continue + short = b[len('origin/'):] if b.startswith('origin/') else b + if short.startswith(prefix): + all_remote_branches.add(short) + + deleted = 0 + kept = 0 + + for branch, dream_id, manifest in manifests: + dream_dir = os.path.join(repo_root, '.shadow', '_dreams', dream_id) + + # Safety check 1: all artifacts on main + artifacts_ok = all( + os.path.isfile(os.path.join(dream_dir, f)) + for f in ('report.md', 'manifest.json', 'patch.diff') + ) + if not artifacts_ok: + print(f" ⚠️ KEEPING {branch} — artifacts not on main") + kept += 1 + continue + + # Safety check 2: in index. Set-membership (NOT substring) — a raw + # `dream_id in index_content` falsely passes when our dream_id is a + # prefix of any indexed ID, which would delete an un-reconciled branch. + # Same prefix-collision shape as `verify_reconciliation`. + if dream_id not in indexed_ids: + print(f" ⚠️ KEEPING {branch} — not in _index.md") + kept += 1 + continue + + # Safety check 3: no un-reconciled descendants + has_descendants = False + for other_branch in all_remote_branches: + if other_branch == branch: + continue + other_id = other_branch[len(prefix):] if other_branch.startswith(prefix) else other_branch + # Check if this other branch is NOT in the index (un-reconciled) + # AND lists our branch as parent. Set-membership for the same + # prefix-collision reason as Safety check 2 above. + if other_id not in indexed_ids: + # Check manifest for parent reference + other_manifest_raw = git_show( + f'origin/{other_branch}', + f'.shadow/_dreams/{other_id}/manifest.json', + cwd=repo_root + ) + if other_manifest_raw: + try: + other_manifest = json.loads(other_manifest_raw) + if other_manifest.get('parent_branch', '') == branch: + has_descendants = True + break + except json.JSONDecodeError: + pass + + if has_descendants: + print(f" ⚠️ KEEPING {branch} — has un-reconciled descendants") + kept += 1 + continue + + # All checks passed — delete + if dry_run: + print(f" Would delete: {branch}") + deleted += 1 + continue + + # Delete remote first (network op that can fail) + result = subprocess.run( + ['git', 'push', 'origin', '--delete', branch], + capture_output=True, text=True, cwd=repo_root + ) + if result.returncode == 0: + print(f" 🗑 Deleted remote: {branch}") + else: + # Remote might not exist (local-only branch) + if 'remote ref does not exist' not in result.stderr: + print(f" ⚠️ Failed to delete remote {branch}: {result.stderr.strip()}") + + # Delete local + result = subprocess.run( + ['git', 'branch', '-D', branch], + capture_output=True, text=True, cwd=repo_root + ) + if result.returncode == 0: + print(f" 🗑 Deleted local: {branch}") + # Also delete the remote-tracking ref + subprocess.run( + ['git', 'branch', '-dr', f'origin/{branch}'], + capture_output=True, text=True, cwd=repo_root + ) + + # Best-effort worktree GC — the directory at + # `${DREAM_WORKTREE_BASE:-/tmp/shadowfrog-dreams}/<ns>/dream-<slug>` + # is now orphaned (its branch is gone). Removing it here closes the + # leak documented in bug-worktree-leak.md. + # + # `branch` is the branch we just deleted; passing it lets the GC + # refuse to remove a path that another dream (sharing the same + # slug) has reclaimed for its own live worktree. + _gc_worktree_after_merge(repo_root, dream_ns, dream_id, branch) + + deleted += 1 + + return deleted, kept + + +# Compiled here so the error message is consistent with `dream-setup.sh`. +# DREAM_ID format: YYYYMMDD-HHMMSSZ-<slug>. The leading timestamp is +# fixed-width (8 digits + '-' + 6 digits + 'Z' + '-' = 17 chars), but we +# anchor on the regex to be robust against drift. +_DREAM_ID_SPLIT_RE = re.compile(r'^(\d{8}-\d{6}Z)-(.+)$') + + +def _slug_from_dream_id(dream_id): + """Return the slug portion of a dream_id, or None if it doesn't match + the canonical `YYYYMMDD-HHMMSSZ-<slug>` shape. + + Critical: do NOT use `dream_id.partition('-')[2]` — dream_ids contain + multiple `-` (the date itself has one), so partition() returns the + rest of the timestamp, NOT the slug. + """ + m = _DREAM_ID_SPLIT_RE.match(dream_id or '') + return m.group(2) if m else None + + +def _registered_worktree_branch(repo_root, candidate_path): + """Return the branch name (without `refs/heads/` prefix) that git has + registered at `candidate_path`, or `None` if no worktree is registered + at that path (or git can't tell). Detached-HEAD worktrees return `None`. + + Parses `git worktree list --porcelain` output: + worktree /abs/path + HEAD <sha> + branch refs/heads/<name> + or: + worktree /abs/path + HEAD <sha> + detached + + Paths are compared after `realpath` so macOS `/tmp` ↔ `/private/tmp` and + other symlinked-base setups don't break the match. + """ + try: + result = subprocess.run( + ['git', 'worktree', 'list', '--porcelain'], + capture_output=True, text=True, cwd=repo_root, timeout=10, + ) + except (OSError, subprocess.SubprocessError): + return None + if result.returncode != 0: + return None + + try: + target = os.path.realpath(candidate_path) + except (OSError, ValueError): + return None + + def _matches(p): + try: + return os.path.realpath(p) == target + except (OSError, ValueError): + return False + + cur_path = None + cur_branch = None + for line in result.stdout.splitlines(): + if line.startswith('worktree '): + # Flush previous entry if it matched. + if cur_path is not None and _matches(cur_path): + return cur_branch + cur_path = line[len('worktree '):] + cur_branch = None + elif line.startswith('branch '): + ref = line[len('branch '):] + cur_branch = ref[len('refs/heads/'):] if ref.startswith('refs/heads/') else ref + if cur_path is not None and _matches(cur_path): + return cur_branch + return None + + +def _gc_worktree_after_merge(repo_root, dream_ns, dream_id, deleted_branch=None): + """Remove the dream worktree directory after its branch has been + deleted. Safety-gated by `_worktree_safety.safe_worktree_path` — will + NEVER `rm -rf` a path outside `$DREAM_WORKTREE_BASE/<ns>/dream-<slug>`. + + Cross-deletion guard: worktree paths are keyed on slug only (see + `dream-setup.sh`: `WORKTREE_DIR=<base>/<ns>/dream-<slug>`), but + `dream_id` includes a timestamp. So two dreams that re-use the same + slug at different times share a worktree path. If the path we're + about to GC is currently registered to a DIFFERENT branch — i.e. a + later dream has reclaimed it — we must NOT touch it. Pass + `deleted_branch` to enable this check. + + All failures are swallowed: the branch delete already succeeded, so a + leaked worktree (the pre-fix steady state) is strictly less bad than + an aborted cleanup_branches() loop. + """ + try: + slug = _slug_from_dream_id(dream_id) + if not slug: + return # Can't derive worktree path — bail silently. + base = os.environ.get('DREAM_WORKTREE_BASE', '/tmp/shadowfrog-dreams') + candidate = os.path.join(base, dream_ns, f'dream-{slug}') + + # Import lazily so this module remains importable for tests that + # don't exercise the worktree-GC path even if the helper is moved. + sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + try: + from _worktree_safety import safe_worktree_path, UnsafePath + finally: + # Only pop our own insertion (defensive against re-import). + if sys.path and sys.path[0] == os.path.dirname(os.path.abspath(__file__)): + sys.path.pop(0) + + try: + resolved = safe_worktree_path(candidate, base) + except UnsafePath as exc: + print(f" ⚠️ Skipping worktree GC for {dream_id}: {exc}") + return + + # Cross-deletion guard: refuse to touch a path another dream owns. + # Only consult git when we know which branch we expected (i.e., + # `deleted_branch` was passed by `cleanup_branches`). Callers that + # pre-date this guard (tests, future ad-hoc invocations) default to + # the original behavior — still gated by the safety check above. + if deleted_branch: + registered = _registered_worktree_branch(repo_root, str(resolved)) + if registered is not None and registered != deleted_branch: + print( + f" ↳ Skipping worktree GC for {dream_id}: " + f"path {resolved} now belongs to {registered}" + ) + return + + # Polite path first: let git update its own bookkeeping. + worktree_removed = False + result = subprocess.run( + ['git', 'worktree', 'remove', str(resolved), '--force'], + capture_output=True, text=True, cwd=repo_root, + ) + if result.returncode == 0: + worktree_removed = True + print(f" 🗑 Removed worktree: {resolved}") + # Fallback: directory may still be on disk (git failed, dead + # gitdir pointer, etc.). The safety gate already proved the path + # is `<base>/<ns>/dream-<slug>` so the rm is bounded. + if not worktree_removed and (resolved.exists() or resolved.is_symlink()): + try: + shutil.rmtree(str(resolved), ignore_errors=False) + print(f" 🗑 Removed worktree (fallback rm -rf): {resolved}") + except OSError as exc: + # Worst case: leak the directory but don't break cleanup. + print(f" ⚠️ Worktree rm failed for {resolved}: {exc}") + return + # Clean up git's stale-worktree bookkeeping. + subprocess.run( + ['git', 'worktree', 'prune'], + capture_output=True, text=True, cwd=repo_root, + ) + except Exception as exc: # noqa: BLE001 — GC must never crash cleanup. + print(f" ⚠️ Worktree GC raised for {dream_id}: {exc}") + + +# --- Main orchestration --- + +def main(): + # Parse arguments + repo_root = None + dry_run = False + verify_only = False + cleanup = False + namespace_override = None + + args = sys.argv[1:] + i = 0 + while i < len(args): + if args[i] in ('--help', '-h'): + print(__doc__) + sys.exit(0) + elif args[i] == '--dry-run': + dry_run = True + elif args[i] == '--verify-only': + verify_only = True + elif args[i] == '--cleanup-branches': + cleanup = True + elif args[i] == '--namespace': + i += 1 + if i >= len(args): + print("ERROR: --namespace requires a value", file=sys.stderr) + sys.exit(1) + namespace_override = args[i] + elif not args[i].startswith('-'): + repo_root = args[i] + else: + print(f"Unknown argument: {args[i]}", file=sys.stderr) + print("Run with --help for usage.", file=sys.stderr) + sys.exit(1) + i += 1 + + if not repo_root: + repo_root = subprocess.run( + ['git', 'rev-parse', '--show-toplevel'], + capture_output=True, text=True + ).stdout.strip() + + if not repo_root or not os.path.isdir(repo_root): + print("ERROR: Not in a git repository", file=sys.stderr) + sys.exit(1) + + # Resolve namespace + dream_ns = namespace_override or os.environ.get('DREAM_NAMESPACE', '') + if not dream_ns: + task_info = os.path.join(repo_root, 'TASK_INFO.json') + if os.path.isfile(task_info): + try: + dream_ns = json.load(open(task_info)).get('dream_namespace', '') + except (json.JSONDecodeError, OSError): + pass + if not dream_ns: + env_file = os.path.join(repo_root, '.env') + if os.path.isfile(env_file): + with open(env_file) as f: + for line in f: + if line.startswith('DREAM_NAMESPACE='): + dream_ns = line.split('=', 1)[1].strip() + break + if not dream_ns: + dream_ns = os.path.basename(repo_root) + + print(f"=== Dream Reconciliation ===") + print(f"Repo: {repo_root}") + print(f"Namespace: {dream_ns}") + if dry_run: + print("Mode: DRY RUN") + print() + + # Verify-only mode: re-check everything already in _index.md (this is what + # users actually want — "did my reconciliation produce the right files?"). + # The previous implementation called discover_branches() which by design + # returns ONLY un-reconciled branches, so verify-only could never see + # anything to verify. + if verify_only: + indexed = _read_indexed_branches(repo_root) + if not indexed: + print("No reconciled dreams found in _index.md.") + sys.exit(0) + manifests = [] + for branch, dream_id in indexed: + # Reconstruct minimal manifest from local mirrored copy + local_manifest = os.path.join( + repo_root, '.shadow', '_dreams', dream_id, 'manifest.json' + ) + if os.path.isfile(local_manifest): + try: + with open(local_manifest) as f: + m = json.load(f) + manifests.append((branch, dream_id, m)) + except (json.JSONDecodeError, OSError): + manifests.append((branch, dream_id, {})) + else: + manifests.append((branch, dream_id, {})) + failures = verify_reconciliation(repo_root, manifests) + if failures: + print("Verification FAILED:") + for f in failures: + print(f" ❌ {f}") + sys.exit(1) + print(f"✓ All {len(manifests)} indexed dreams verified.") + sys.exit(0) + + # Step 1: Discover branches + print("Step 1: Discovering branches...") + branches = discover_branches(repo_root, dream_ns) + if not branches: + print(" No new branches to reconcile.") + # If --cleanup-branches was requested, fall through to cleanup using + # the branches already in the index. This is the canonical post-push + # flow: reconcile → push → re-run with --cleanup-branches. + if cleanup: + indexed = _read_indexed_branches(repo_root) + if not indexed: + print(" Nothing in _index.md to clean up either.") + sys.exit(0) + # Synthesize minimal manifests from the local mirrored copies so + # cleanup_branches can do its safety checks. + cleanup_manifests = [] + for branch, dream_id in indexed: + local_manifest = os.path.join( + repo_root, '.shadow', '_dreams', dream_id, 'manifest.json' + ) + m = {} + if os.path.isfile(local_manifest): + try: + with open(local_manifest) as f: + m = json.load(f) + except (json.JSONDecodeError, OSError): + pass + cleanup_manifests.append((branch, dream_id, m)) + print() + print("Step 9: Cleaning up reconciled branches...") + deleted, kept = cleanup_branches( + repo_root, cleanup_manifests, dream_ns, dry_run=dry_run + ) + print(f" Deleted: {deleted}, Kept: {kept}") + sys.exit(0) + print(f" Found {len(branches)} new branch(es):") + for branch, dream_id in branches: + print(f" {branch}") + print() + + # Step 2: Load manifests + print("Step 2: Loading manifests...") + manifests, skipped = load_manifests(repo_root, branches) + print(f" Valid: {len(manifests)}, Skipped: {len(skipped)}") + for branch, dream_id, reason in skipped: + print(f" SKIP {dream_id}: {reason}") + print() + + if not manifests: + print("No valid manifests to reconcile.") + sys.exit(0) + + # Step 3: Merge discoveries + print("Step 3: Merging discoveries...") + merged, dup_skipped = merge_discoveries(repo_root, manifests, dry_run=dry_run) + print(f" Merged: {merged}, Duplicates skipped: {dup_skipped}") + print() + + # Step 4: Mirror reports + print("Step 4: Mirroring reports...") + mirrored, corrupted = mirror_reports(repo_root, manifests, dry_run=dry_run) + print(f" Mirrored: {mirrored}") + if corrupted: + print(f" Corrupted: {len(corrupted)}") + for did, wrong_did in corrupted: + print(f" ⚠️ {did} contained report from {wrong_did}") + print() + + # Step 5: Update index + print("Step 5: Updating index...") + update_index(repo_root, manifests, dry_run=dry_run) + print(f" Added {len(manifests)} entries") + print() + + # Step 6: Update state + print("Step 6: Updating state.json...") + update_state(repo_root, manifests, dry_run=dry_run) + print() + + # Step 7: Rebuild top-level _index.md (must run AFTER update_state so + # the dream_cycles_completed count it reads is current). + print("Step 7: Rebuilding top-level _index.md...") + rebuild_top_index(repo_root, dry_run=dry_run) + print() + + # Step 8: Verify + if not dry_run: + print("Step 8: Verifying...") + failures = verify_reconciliation(repo_root, manifests) + if failures: + print(" VERIFICATION FAILED:") + for f in failures: + print(f" ❌ {f}") + print() + print("Re-run reconciliation for failed dreams.") + sys.exit(1) + else: + print(f" ✓ All {len(manifests)} dreams verified.") + + print() + print(f"=== Reconciliation {'would complete' if dry_run else 'complete'} ===") + print(f" Dreams reconciled: {len(manifests)}") + print(f" Discoveries merged: {merged}") + if not dry_run: + print() + print("Next: git add .shadow/ && git commit && git push") + + # Step 9: Cleanup branches (optional, after user commits and pushes) + if cleanup and manifests: + print() + print("Step 9: Cleaning up reconciled branches...") + if not dry_run: + print(" ⚠️ Run this AFTER 'git push' succeeds on main.") + print(" Checking artifacts on main...") + deleted, kept = cleanup_branches( + repo_root, manifests, dream_ns, dry_run=dry_run + ) + print(f" Deleted: {deleted}, Kept: {kept}") + + +if __name__ == '__main__': + main() diff --git a/skills/shadow-frog-dream/dream-setup.sh b/skills/shadow-frog-dream/dream-setup.sh new file mode 100755 index 0000000..d68c16c --- /dev/null +++ b/skills/shadow-frog-dream/dream-setup.sh @@ -0,0 +1,347 @@ +#!/usr/bin/env bash +# Dream experiment setup — creates worktree and exports environment. +# +# Usage: +# eval "$(dream-setup.sh --slug t01-csv-fuzzer)" || { echo "setup failed" >&2; exit 1; } +# eval "$(dream-setup.sh --slug t03-extend --base-branch dream/ns/20250612-143012Z-csv-fuzzer)" || exit 1 +# +# IMPORTANT: ALWAYS check the exit status of `eval` — on failure this script +# prints to stderr and exits non-zero, which `eval "$(...)"` cannot detect on +# its own. Without `|| exit 1` the agent silently proceeds with empty env vars. +# +# Why `eval` is safe HERE (do not cargo-cult it elsewhere): this script never +# echoes untrusted input back. All inputs (slug/namespace) are validated +# against [A-Za-z0-9_-][A-Za-z0-9._-]* before use, and every exported value is +# shell-escaped with `printf %q`, so the emitted text is a fixed set of safe +# `export` lines. Only `eval` output you produced under these guarantees. +# +# This script: +# 1. Validates inputs (slug/namespace) against [A-Za-z0-9_-][A-Za-z0-9._-]* +# (non-`.` first char rejects bare `.`/`..`) to prevent shell-metachar +# injection through the eval contract +# 2. Computes DREAM_ID, BRANCH_NAME, WORKTREE_DIR, BASE_COMMIT +# 3. Creates the worktree (idempotent — cleans existing if found) +# 4. Prints export statements (values shell-escaped via printf %q) +# +# Flags: +# --slug NAME Task slug (required, e.g., "t01-csv-fuzzer") +# --base-branch REF Branch to base from (default: default branch = fresh) +# --namespace NS Override DREAM_NAMESPACE (default: env or repo basename) +# --repo-root DIR Override repo root (default: git rev-parse) +# --print-env Print export statements (default behavior) +# --print-json Print JSON instead of shell exports +# --dry-run Compute values without creating worktree +# --help, -h Show this help message +# +# Design decisions: +# - Idempotent: re-running with same slug cleans and recreates +# - External path: worktrees always in /tmp/shadowfrog-dreams/<NS>/ +# - Validates worktree is NOT inside project directory +# - Detects default branch (main/master) automatically +# - Resolves DREAM_NAMESPACE from env > TASK_INFO.json > .env > repo name + +set -euo pipefail + +# --- Argument parsing --- +SLUG="" +BASE_BRANCH="" +NAMESPACE_OVERRIDE="" +REPO_ROOT_OVERRIDE="" +OUTPUT_MODE="env" +DRY_RUN=false + +show_help() { + sed -n '2,/^$/p' "$0" | sed 's/^# \{0,1\}//' + exit 0 +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --slug) SLUG="$2"; shift 2 ;; + --base-branch) BASE_BRANCH="$2"; shift 2 ;; + --namespace) NAMESPACE_OVERRIDE="$2"; shift 2 ;; + --repo-root) REPO_ROOT_OVERRIDE="$2"; shift 2 ;; + --print-json) OUTPUT_MODE="json"; shift ;; + --print-env) OUTPUT_MODE="env"; shift ;; + --dry-run) DRY_RUN=true; shift ;; + --help|-h) show_help ;; + *) echo "ERROR: Unknown argument: $1" >&2; exit 1 ;; + esac +done + +if [[ -z "$SLUG" ]]; then + echo "ERROR: --slug is required" >&2 + echo "Usage: dream-setup.sh --slug t01-name [--base-branch BRANCH]" >&2 + exit 1 +fi + +# --- Input validation: prevent shell injection via the `eval` contract --- +# The script's output is fed to `eval`, so any unescaped shell metachars in +# slug/namespace would execute as code. Restrict to filesystem-safe chars, +# AND require a non-`.` first char to reject bare `.`/`..` as a slug or ns. +SAFE_RE='^[A-Za-z0-9_-][A-Za-z0-9._-]*$' +if ! [[ "$SLUG" =~ $SAFE_RE ]]; then + echo "ERROR: --slug must match $SAFE_RE (got: $SLUG)" >&2 + echo " Use kebab-case alphanumerics like 't01-csv-fuzzer'." >&2 + exit 1 +fi +if [[ -n "$NAMESPACE_OVERRIDE" ]] && ! [[ "$NAMESPACE_OVERRIDE" =~ $SAFE_RE ]]; then + echo "ERROR: --namespace must match $SAFE_RE (got: $NAMESPACE_OVERRIDE)" >&2 + exit 1 +fi + +# --- Resolve repo root --- +if [[ -n "$REPO_ROOT_OVERRIDE" ]]; then + REPO_ROOT="$REPO_ROOT_OVERRIDE" +else + REPO_ROOT=$(git rev-parse --show-toplevel 2>/dev/null) || { + echo "ERROR: Not in a git repository" >&2; exit 1 + } +fi +cd "$REPO_ROOT" + +# --- Guard: .shadow/ must be tracked by git (not gitignored) --- +# The dream workflow moves .shadow/ content through git: artifacts are +# committed onto the dream branch, pushed, then read back by the reconciler +# via `git show origin/<branch> .shadow/...`. If .shadow/ is gitignored, +# `git add -A` silently skips those files, nothing is pushed, and the +# reconciler finds no manifest — every discovery is lost without warning. +# Fail fast with a clear message instead. (shadow-frog-init asks the user +# whether to commit or gitignore .shadow/; the dream skill requires committed.) +# Probe a NEW child path rather than `.shadow` itself: when `.shadow/` is +# gitignored but already tracked, `git check-ignore .shadow` reports "not +# ignored" (tracked content wins), yet `git add -A` still silently drops any +# NEW files created under it (e.g. _dreams/<id>/manifest.json) — the exact +# data-loss case this guard exists to prevent. A child path under .shadow/ +# reflects the gitignore rule regardless of tracking state. +if git check-ignore -q .shadow/_dreams/__shadowfrog_probe__/manifest.json 2>/dev/null; then + echo "ERROR: .shadow/ is gitignored — shadow-frog-dream requires it to be tracked by git." >&2 + echo " Dream experiments commit .shadow/ artifacts onto a branch, push them, and" >&2 + echo " reconcile reads them back from the remote. A gitignored .shadow/ would be" >&2 + echo " silently dropped at commit time, losing every discovery." >&2 + echo " Fix: remove the '.shadow/' entry from .gitignore and commit .shadow/," >&2 + echo " or run shadow-frog-update (which works in local-only mode) instead of dream." >&2 + exit 1 +fi + +# --- Detect default branch --- +DEFAULT_BRANCH=$(git symbolic-ref refs/remotes/origin/HEAD 2>/dev/null \ + | sed 's|refs/remotes/origin/||') +if [[ -z "$DEFAULT_BRANCH" ]]; then + if git show-ref --verify refs/remotes/origin/main >/dev/null 2>&1; then + DEFAULT_BRANCH="main" + elif git show-ref --verify refs/remotes/origin/master >/dev/null 2>&1; then + DEFAULT_BRANCH="master" + else + echo "ERROR: Cannot detect default branch. Fix: git remote set-head origin <branch>" >&2 + exit 1 + fi +fi + +# --- Resolve namespace --- +if [[ -n "$NAMESPACE_OVERRIDE" ]]; then + DREAM_NS="$NAMESPACE_OVERRIDE" +elif [[ -n "${DREAM_NAMESPACE:-}" ]]; then + DREAM_NS="$DREAM_NAMESPACE" +elif [[ -f TASK_INFO.json ]]; then + DREAM_NS=$(python3 -c "import json; print(json.load(open('TASK_INFO.json')).get('dream_namespace',''))" 2>/dev/null || echo "") +elif [[ -f .env ]]; then + DREAM_NS=$(grep '^DREAM_NAMESPACE=' .env 2>/dev/null | head -1 | cut -d'=' -f2- | sed -E 's/^[[:space:]]*["'\'']?//; s/["'\'']?[[:space:]]*$//' || echo "") +fi +DREAM_NS="${DREAM_NS:-$(basename "$REPO_ROOT")}" + +# Validate resolved namespace too (could come from TASK_INFO/.env/basename) +if ! [[ "$DREAM_NS" =~ $SAFE_RE ]]; then + echo "ERROR: Resolved DREAM_NS contains unsafe characters: $DREAM_NS" >&2 + echo " Allowed: $SAFE_RE" >&2 + echo " Override with --namespace or set DREAM_NAMESPACE." >&2 + exit 1 +fi + +# --- Compute identifiers --- +DREAM_ID="$(date -u +%Y%m%d-%H%M%SZ)-${SLUG}" +BRANCH_NAME="dream/${DREAM_NS}/${DREAM_ID}" + +# --- Compute worktree path (ALWAYS in /tmp, NEVER in project) --- +WORKTREE_BASE="${DREAM_WORKTREE_BASE:-/tmp/shadowfrog-dreams}/${DREAM_NS}" +WORKTREE_DIR="${WORKTREE_BASE}/dream-${SLUG}" + +# Validate worktree is external to project +case "$WORKTREE_DIR" in + "$REPO_ROOT"|"$REPO_ROOT"/*) + echo "ERROR: Worktree would be inside project: $WORKTREE_DIR" >&2 + echo "MUST use external path (default: /tmp/shadowfrog-dreams/)" >&2 + exit 1 + ;; +esac + +# --- Resolve base reference --- +if [[ -z "$BASE_BRANCH" ]]; then + BASE_REF="origin/$DEFAULT_BRANCH" + PARENT_BRANCH="$DEFAULT_BRANCH" +else + BASE_REF="origin/$BASE_BRANCH" + PARENT_BRANCH="$BASE_BRANCH" + # Verify the base branch exists on remote + if ! git show-ref --verify "refs/remotes/$BASE_REF" >/dev/null 2>&1; then + echo "ERROR: Base branch not found: $BASE_REF" >&2 + exit 1 + fi +fi + +# --- Create worktree (unless dry-run) --- +BASE_COMMIT="" +if [[ "$DRY_RUN" == "false" ]]; then + mkdir -p "$WORKTREE_BASE" + + # --- Periodic auto-GC (Bug A fix from bug-cleanup-gaps.md) --- + # `dream-gc.sh` is documented as "run periodically" but had no caller in + # the skill flow, so long-running fleets accumulated orphans forever + # (machine reboots, OOM-killed agents, `$REPO_ROOT` unset, races, …). + # Trigger it from here, throttled by a per-namespace tombstone file so + # the cost amortizes across many dreams. ALL output is redirected to + # stderr or /dev/null to preserve the `eval "$(...)"` contract. + if [[ "${DREAM_GC_AUTO:-1}" != "0" ]]; then + SCRIPT_DIR_SH="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + GC_SCRIPT="$SCRIPT_DIR_SH/dream-gc.sh" + TOMBSTONE="$WORKTREE_BASE/.last-gc" + GC_INTERVAL_MIN="${DREAM_GC_INTERVAL_MIN:-60}" + GC_AGE_MIN="${DREAM_GC_AGE_MIN:-60}" + + # Validate the env-supplied integers so a hostile value can't reach + # `find -mmin` or `dream-gc.sh --min-age-min` as injected args. + # (dream-gc.sh itself also re-validates, but defense in depth.) + SAFE_INT='^[0-9]+$' + if ! [[ "$GC_INTERVAL_MIN" =~ $SAFE_INT ]] || ! [[ "$GC_AGE_MIN" =~ $SAFE_INT ]]; then + echo "WARN: DREAM_GC_INTERVAL_MIN / DREAM_GC_AGE_MIN must be non-negative integers — skipping auto-GC" >&2 + elif [[ -x "$GC_SCRIPT" ]] || [[ -f "$GC_SCRIPT" ]]; then + should_run=false + if [[ ! -f "$TOMBSTONE" ]]; then + should_run=true + elif [[ "$GC_INTERVAL_MIN" -eq 0 ]]; then + # Interval 0 ⇒ "always run". `-mmin +0` would NOT match a + # tombstone touched in the last ~1 min, so we'd silently + # skip-throttle the very first invocation after touch. Bypass + # the find entirely. + should_run=true + elif [[ -n "$(find "$TOMBSTONE" -mmin "+$GC_INTERVAL_MIN" 2>/dev/null)" ]]; then + should_run=true + fi + if [[ "$should_run" == true ]]; then + # Touch BEFORE running, so a parallel `dream-setup.sh` for + # the same ns sees a fresh tombstone and skips. Worst case + # of a tombstone-vs-gc race is one extra GC pass; never a + # missed cleanup. + touch "$TOMBSTONE" 2>/dev/null || true + bash "$GC_SCRIPT" \ + --repo-root "$REPO_ROOT" \ + --quiet \ + --min-age-min "$GC_AGE_MIN" \ + >&2 || true + fi + fi + fi + + # Clean existing worktree (idempotent). Try git first, then fall back + # to a safety-gated rm -rf for stale worktrees git can't see. Matches + # the dream-cleanup.sh path so a leak healed during a retry instead of + # cascading into "branch already exists" errors at git worktree add. + if [[ -d "$WORKTREE_DIR" ]]; then + SAFETY_SH="$(dirname "${BASH_SOURCE[0]}")/_worktree_safety.py" + if ! git worktree remove "$WORKTREE_DIR" --force 2>/dev/null; then + # Safety gate: only rm if path matches `<base>/<ns>/dream-<slug>`. + # If the safety module is missing, REFUSE — never fall through + # to an un-gated rm just because the gate is unloadable. + if [[ ! -f "$SAFETY_SH" ]]; then + echo "ERROR: safety module not found, refusing pre-clean rm: $SAFETY_SH" >&2 + exit 1 + fi + if python3 "$SAFETY_SH" "$WORKTREE_DIR" "${DREAM_WORKTREE_BASE:-/tmp/shadowfrog-dreams}" >/dev/null 2>&1; then + rm -rf -- "$WORKTREE_DIR" + fi + fi + git worktree prune + fi + + # Create worktree with new branch. + # IMPORTANT: redirect BOTH stdout and stderr — modern git prints + # "branch '...' set up to track..." and "HEAD is now at..." to stdout, + # which would corrupt the `eval "$(...)"` contract used by callers. + git worktree add "$WORKTREE_DIR" -b "$BRANCH_NAME" "$BASE_REF" >/dev/null 2>&1 || { + git branch -D "$BRANCH_NAME" >/dev/null 2>&1 || true + git worktree prune >/dev/null 2>&1 + git worktree add "$WORKTREE_DIR" -b "$BRANCH_NAME" "$BASE_REF" >/dev/null 2>&1 || { + echo "ERROR: git worktree add failed for $WORKTREE_DIR (branch $BRANCH_NAME, base $BASE_REF)" >&2 + exit 1 + } + } + + BASE_COMMIT=$(git -C "$WORKTREE_DIR" rev-parse HEAD) +else + BASE_COMMIT=$(git rev-parse "$BASE_REF" 2>/dev/null || echo "DRY_RUN") +fi + +# --- Detect RUN_PREFIX --- +RUN_PREFIX="" +if [[ -f uv.lock ]]; then + RUN_PREFIX="uv run" +elif [[ -f package-lock.json ]]; then + RUN_PREFIX="npx" +elif [[ -f yarn.lock ]]; then + RUN_PREFIX="npx" +fi + +# --- Output (values shell-escaped via printf %q for safe `eval`) --- +emit_export() { + # printf %q produces a string that bash/zsh/sh can parse back losslessly. + printf 'export %s=%q\n' "$1" "$2" +} + +if [[ "$OUTPUT_MODE" == "json" ]]; then + # JSON output — use python for safe escaping (handles quotes, backslashes, + # control chars). Values are passed via the environment (NOT interpolated + # into the Python source) so a repo path containing ", \, or $ can't break + # the script. Falls back with a clear error if python is missing. + SF_REPO_ROOT="$REPO_ROOT" \ + SF_DEFAULT_BRANCH="$DEFAULT_BRANCH" \ + SF_DREAM_NS="$DREAM_NS" \ + SF_DREAM_ID="$DREAM_ID" \ + SF_BRANCH_NAME="$BRANCH_NAME" \ + SF_PARENT_BRANCH="$PARENT_BRANCH" \ + SF_WORKTREE_DIR="$WORKTREE_DIR" \ + SF_WORKTREE_BASE="$WORKTREE_BASE" \ + SF_BASE_COMMIT="$BASE_COMMIT" \ + SF_RUN_PREFIX="$RUN_PREFIX" \ + SF_SLUG="$SLUG" \ + python3 - <<'PYEOF' || { +import json, os +print(json.dumps({ + "repo_root": os.environ["SF_REPO_ROOT"], + "default_branch": os.environ["SF_DEFAULT_BRANCH"], + "dream_ns": os.environ["SF_DREAM_NS"], + "dream_id": os.environ["SF_DREAM_ID"], + "branch_name": os.environ["SF_BRANCH_NAME"], + "parent_branch": os.environ["SF_PARENT_BRANCH"], + "worktree_dir": os.environ["SF_WORKTREE_DIR"], + "worktree_base": os.environ["SF_WORKTREE_BASE"], + "base_commit": os.environ["SF_BASE_COMMIT"], + "run_prefix": os.environ["SF_RUN_PREFIX"], + "slug": os.environ["SF_SLUG"], +}, indent=2)) +PYEOF + echo "ERROR: python3 required for --print-json" >&2 + exit 1 + } +else + emit_export REPO_ROOT "$REPO_ROOT" + emit_export DEFAULT_BRANCH "$DEFAULT_BRANCH" + emit_export DREAM_NS "$DREAM_NS" + emit_export DREAM_ID "$DREAM_ID" + emit_export BRANCH_NAME "$BRANCH_NAME" + emit_export PARENT_BRANCH "$PARENT_BRANCH" + emit_export WORKTREE_DIR "$WORKTREE_DIR" + emit_export WORKTREE_BASE "$WORKTREE_BASE" + emit_export BASE_COMMIT "$BASE_COMMIT" + emit_export RUN_PREFIX "$RUN_PREFIX" + emit_export SLUG "$SLUG" +fi diff --git a/skills/shadow-frog-dream/dream-validate.py b/skills/shadow-frog-dream/dream-validate.py new file mode 100644 index 0000000..9c3c51a --- /dev/null +++ b/skills/shadow-frog-dream/dream-validate.py @@ -0,0 +1,355 @@ +#!/usr/bin/env python3 +"""Validate dream artifacts before commit. + +Usage: python3 dream-validate.py DREAM_ID [WORKTREE_DIR] + python3 dream-validate.py --help + +Checks: + 1. No flat files (subdirectory format required) + 2. Dream subdirectory exists + 3. Required files exist (report.md, manifest.json, patch.diff) + 4. patch.diff is non-empty (completion criterion #1) + 5. Report frontmatter dream_id matches + 6. Manifest dream_id matches + 7. Required manifest fields present + 8. Category and verdict validation + 9. Discovery `op` is `add` (update/refute not yet supported by reconciler) + 10. Discoveries in manifest are mirrored into per-file shadows on the + dream branch (the reconciler ports manifest entries into main, but + the documented workflow also requires updating branch shadows so + human PR reviewers can read the discoveries in context) + 11. Non-blocking label-triage warnings (bug/security/performance signal + phrases in discovery text without matching labels) + +Exits 0 on success, 1 on ANY validation failure. +This is the hard gate — agents MUST NOT commit/push if this fails. +""" + +import json +import os +import re +import subprocess +import sys + + +def main(): + args = sys.argv[1:] + + if not args or args[0] in ('--help', '-h'): + print(__doc__) + sys.exit(0 if args else 1) + + dream_id = args[0] + worktree = args[1] if len(args) > 1 else os.getcwd() + + errors = [] + warnings = [] + dream_dir = os.path.join(worktree, '.shadow', '_dreams', dream_id) + + # 1. Check for flat files (wrong format) + for ext in ['.md', '.manifest.json', '.patch.diff', '.diff']: + flat = os.path.join(worktree, '.shadow', '_dreams', dream_id + ext) + if os.path.isfile(flat): + errors.append(f"Flat file found: .shadow/_dreams/{dream_id}{ext}") + errors.append(f"MUST use subdirectory: .shadow/_dreams/{dream_id}/") + + # 2. Check subdirectory exists + if not os.path.isdir(dream_dir): + errors.append(f"Missing dream subdirectory .shadow/_dreams/{dream_id}/") + errors.append(f"Create with: mkdir -p .shadow/_dreams/{dream_id}") + for e in errors: + print(f"ERROR: {e}") + sys.exit(1) + + # 3. Check required files + for required in ['report.md', 'manifest.json', 'patch.diff']: + path = os.path.join(dream_dir, required) + if not os.path.isfile(path): + errors.append(f"Missing .shadow/_dreams/{dream_id}/{required}") + + if errors: + for e in errors: + print(f"ERROR: {e}") + sys.exit(1) + + # 4. Enforce non-empty patch.diff (completion criterion #1) + patch_path = os.path.join(dream_dir, 'patch.diff') + patch_bytes = os.path.getsize(patch_path) + if patch_bytes == 0: + errors.append( + f"patch.diff is empty — completion criterion #1 requires " + f"code was written or modified. Either implement something, " + f"or delete this dream and choose a different task." + ) + else: + # Non-empty is not enough: whitespace-only or a stray newline must + # not count as "code was written". Require a real unified-diff marker. + with open(patch_path, encoding='utf-8', errors='replace') as pf: + patch_text = pf.read() + if not re.search(r'(?m)^(diff --git |--- |\+\+\+ |@@ )', patch_text): + errors.append( + "patch.diff has content but no unified-diff markers " + "(`diff --git`, `---`, `+++`, or `@@` hunk headers). It does " + "not look like a real code diff — regenerate it with " + "`git diff <base_commit>..HEAD -- . ':!.shadow'` or delete " + "this dream." + ) + + # 5. Validate report dream_id + report_path = os.path.join(dream_dir, 'report.md') + with open(report_path) as f: + content = f.read() + m = re.match(r'^\ufeff?\s*---\r?\n(.*?)\r?\n---', content, re.S) + if m: + dm = re.search(r'^dream_id:\s*["\']?(.+?)["\']?\s*$', m.group(1), re.M) + report_did = dm.group(1) if dm else '' + else: + report_did = '' + + if report_did != dream_id: + errors.append( + f"report.md dream_id mismatch: '{report_did}' (expected '{dream_id}')" + ) + errors.append("REWRITE the report with the correct dream_id.") + + # 6. Validate manifest + manifest_path = os.path.join(dream_dir, 'manifest.json') + try: + with open(manifest_path) as f: + manifest = json.load(f) + except json.JSONDecodeError as e: + errors.append(f"manifest.json is invalid JSON: {e}") + for e in errors: + print(f"ERROR: {e}") + sys.exit(1) + + manifest_did = manifest.get('dream_id', '') + if manifest_did != dream_id: + errors.append( + f"manifest.json dream_id mismatch: '{manifest_did}' (expected '{dream_id}')" + ) + errors.append("REWRITE the manifest with the correct dream_id.") + + # 7. Check required fields + required_fields = [ + 'dream_id', 'branch', 'parent_branch', 'category', 'verdict', 'title' + ] + missing = [f for f in required_fields if not manifest.get(f)] + if missing: + errors.append(f"manifest.json missing required fields: {missing}") + + # 8. Validate category and verdict values + valid_cats = { + 'investigation', 'bug hunting', 'feature design', + 'refactoring', 'optimization', 'security audit' + } + cat = manifest.get('category', '').lower() + if cat and cat not in valid_cats: + errors.append(f'Invalid category "{manifest.get("category")}". Must be one of: {valid_cats}') + + valid_verdicts = {'useful', 'dead_end'} + verdict = manifest.get('verdict', '').lower() + if verdict and verdict not in valid_verdicts: + errors.append(f'Invalid verdict "{manifest.get("verdict")}". Must be one of: {valid_verdicts}') + + # 9. Validate discovery `op` values — reconciler currently only + # implements `add`. update/refute are reserved for future use; allowing + # them through silently corrupts main's shadow because the reconciler + # appends them as new discoveries instead of mutating the original. + discoveries = manifest.get('discoveries', []) or [] + # Normalize bare-string discoveries to dicts, mirroring the reconciler + # (dream-reconcile.py merge_discoveries) so validate accepts exactly the + # manifests the reconciler does instead of crashing on str entries. + discoveries = [ + {'text': d} if isinstance(d, str) else d + for d in discoveries + ] + for i, disc in enumerate(discoveries): + if not isinstance(disc, dict): + errors.append( + f'discoveries[{i}] must be a string or object, got ' + f'{type(disc).__name__}.' + ) + continue + op = (disc.get('op') or 'add').lower() + if op != 'add': + errors.append( + f'discoveries[{i}].op = "{op}" — only "add" is supported ' + f'by the reconciler today. Drop the op field (defaults to ' + f'"add"), or split this into a meditate session.' + ) + + # 10. Discoveries must be mirrored into per-file shadows on the dream + # branch. The reconciler reads manifest entries directly when merging + # into main, so the discoveries themselves are NOT lost — but the + # documented dream workflow also requires updating per-file shadows + # so human PR reviewers can read discoveries in context alongside the + # branch's code changes. A manifest with discoveries but zero + # modified .shadow/*.md files outside _dreams/ means the branch is + # out of sync with its own manifest. This is a hard error: it + # signals the agent skipped the mirroring step. + if discoveries: + bm_match = re.search( + r'^base_commit:\s*["\']?([0-9a-fA-F]{7,40})["\']?\s*$', + m.group(1), re.M, + ) if m else None + if not bm_match: + errors.append( + "manifest declares discoveries but report.md frontmatter " + "is missing `base_commit` — cannot verify shadow files were " + "updated. Add base_commit (full SHA emitted by dream-setup.sh)." + ) + else: + base = bm_match.group(1) + modified = set() + git_failed = False + try: + # --relative forces output paths relative to CWD, so + # `.shadow/foo.md` checks work whether the worktree is + # the repo root (normal dream case) or a subdirectory + # (testing this validator on an example). + d = subprocess.run( + ['git', '-C', worktree, 'diff', '--name-only', + '--relative', f'{base}..HEAD', '--', '.shadow/'], + capture_output=True, text=True, timeout=10, + ) + if d.returncode == 0: + if d.stdout: + modified.update(d.stdout.strip().splitlines()) + else: + # Non-zero almost always means base_commit is not a + # resolvable ref in this worktree — we cannot compute the + # mirror diff, so don't emit the misleading "no shadows + # modified" hard error below. + git_failed = True + # Include uncommitted (staged + unstaged) — the final + # commit may not have happened at validate time. + # `--untracked-files=all` forces git to enumerate every + # untracked file individually; without it, a fresh + # `.shadow/src/` subtree is rolled up into a single + # `?? .shadow/` line and the mirror check false-fails. + s = subprocess.run( + ['git', '-C', worktree, 'status', '--porcelain=v1', + '--untracked-files=all', '--', '.shadow/'], + capture_output=True, text=True, timeout=10, + ) + if s.returncode == 0: + for line in s.stdout.strip().splitlines(): + if len(line) > 3: + # "XY path" or "R old -> new" — take the + # final path segment. + modified.add(line[3:].split(' -> ')[-1]) + else: + git_failed = True + except (subprocess.TimeoutExpired, FileNotFoundError, OSError): + warnings.append( + "could not run `git diff` against base_commit — " + "skipping the discovery-mirror check. Verify " + "manually that .shadow/*.md files were updated." + ) + else: + non_dream = [ + f for f in modified + if f.startswith('.shadow/') + and not f.startswith('.shadow/_dreams/') + and f.endswith('.md') + ] + if not non_dream and git_failed: + warnings.append( + f"could not resolve base_commit {base[:8]} in this " + f"worktree (git diff/status returned an error), so the " + f"discovery-mirror check was skipped. Verify manually " + f"that .shadow/*.md files were updated, or confirm " + f"base_commit is correct." + ) + elif not non_dream: + errors.append( + f"manifest declares {len(discoveries)} discoveries " + f"but NO .shadow/*.md files outside _dreams/ were " + f"modified vs base_commit {base[:8]}. The reconciler " + f"merges manifest entries into main directly (so " + f"discoveries are not lost at merge time), but the " + f"branch shadows must also be updated so PR reviewers " + f"can read the discoveries in context. Mirror each " + f"discovery into the corresponding per-file shadow " + f"(e.g. .shadow/src/foo.py.md) or into .shadow/_cross/ " + f"before validating." + ) + + # 11. Label triage (non-blocking) — nudge the agent to re-check label + # assignment when discovery text contains common signal phrases but no + # label is set. Keywords are intentionally narrow to limit false positives + # (e.g. "slow path" alone is too noisy). The agent makes the final call. + LABEL_SIGNALS = { + 'bug': [ + r'\bsilent(?:ly)? (?:fail|return|ignore|drop|truncate|swallow)', + r'\boff[- ]by[- ]one\b', + r'\bincorrect(?:ly)? (?:return|compute|round|order)', + r'\brace condition\b', + r'\bnot thread[- ]safe\b', + r'\bvalidate[- ]then[- ](?:use|calculate|apply)', + r'\bmasks? (?:errors?|exceptions?|failures?)\b', + r'\bcrash(?:es)? on\b', + r'\bnever (?:fires?|runs?|reaches?)\b', + ], + 'security': [ + r'\binjection\b', + r'\bpath traversal\b', + r'\b\.\./\b', + r'\bunsanitized\b', + r'\bunescape(?:d)?\b', + r'\bauth(?:n|z)? bypass\b', + r'\bmissing auth(?:n|z| check)\b', + r'\b(?:logs?|leaks?) (?:secret|password|token|api[- ]?key)', + r'\beval\(', + r'\bos\.system\(', + r'\bshell=True\b', + ], + 'performance': [ + r'\bO\(N(?:\^|\*\*)?2\)', + r'\bquadratic\b', + r'\b(?:slow|takes?) \d+(?:\.\d+)? *(?:ms|s|seconds?|minutes?)', + r'\bre[- ]?reads?\b.*\b(?:every|each) (?:call|request|iteration)', + r'\bbottleneck\b', + r'\bN\+1\b', + ], + } + sig_patterns = { + lbl: [re.compile(p, re.I) for p in pats] + for lbl, pats in LABEL_SIGNALS.items() + } + for i, disc in enumerate(discoveries): + if not isinstance(disc, dict): + continue + text = (disc.get('text') or '') + if not text: + continue + existing = {l.lower() for l in (disc.get('labels') or [])} + for lbl, patterns in sig_patterns.items(): + if lbl in existing: + continue + for pat in patterns: + if pat.search(text): + warnings.append( + f"discoveries[{i}] text contains '{lbl}' signal " + f"(/{pat.pattern}/) but no '{lbl}' label is set. " + f"Re-check whether this discovery is actionable; " + f"if so, add it to manifest labels and the in-file " + f"discovery metadata." + ) + break + + # Output + for w in warnings: + print(f"WARNING: {w}") + + if errors: + for e in errors: + print(f"ERROR: {e}") + sys.exit(1) + + print(f"✓ Validation passed for {dream_id}") + + +if __name__ == '__main__': + main() diff --git a/skills/shadow-frog-init/SKILL.md b/skills/shadow-frog-init/SKILL.md new file mode 100644 index 0000000..6298466 --- /dev/null +++ b/skills/shadow-frog-init/SKILL.md @@ -0,0 +1,267 @@ +--- +name: shadow-frog-init +description: >- + Initialize a shadow knowledge base for any codebase. Creates a .shadow/ + directory that mirrors the source tree with markdown files for AI-discovered + insights. Run this once per repo before using other shadow-frog skills. + Refuses to overwrite an existing .shadow/ unless --reset is passed. +scripts: + - shadow-init.py +--- + +# ShadowFrog Init + +Creates `.shadow/` directory with symbol-organized shadow files for every +source file. Run once per repo. If `.shadow/` exists, ask user to reset or skip. + +## Primary: Python Helper Script + +The companion script `shadow-init.py` lives in the same directory as +this SKILL.md file. To find and run it: + +```bash +# Project install (Copilot CLI): +python3 .github/skills/shadow-frog-init/shadow-init.py [options] +# Or for Claude Code: +python3 .claude/skills/shadow-frog-init/shadow-init.py [options] +``` + +**IMPORTANT: Run from the repo/worktree root directory.** The script +auto-detects the root via `git rev-parse --show-toplevel`, which returns +the correct root for regular repos AND worktrees. If auto-detection fails +(common when python is routed through Docker or the `.git` file points to +an inaccessible path), pass `--root` explicitly: + +```bash +# If auto-detect fails, pass the root explicitly: +python3 .github/skills/shadow-frog-init/shadow-init.py --root "$(pwd)" + +# In Docker wrapper scenarios (eval harness), git may not work inside +# the container. Use --root to bypass git detection: +python3 .github/skills/shadow-frog-init/shadow-init.py --root /testbed +``` + +### Options + +| Flag | Effect | +|------|--------| +| `--root DIR` | Repository root (default: auto-detect via git) | +| `--reset` | Delete existing .shadow/ and recreate | +| `--dry-run` | Show what would be created without writing | + +### What it does + +1. Discovers source files via `git ls-files` +2. Filters through `.shadow/.shadowignore` (gitignore syntax) +3. Extracts symbols from each file (classes, functions, methods) +4. Creates per-file shadow `.md` files with symbol headings +5. Creates `_index.md`, `_prefs.md`, `_meta/state.json`, `.shadowignore` +6. Reports: files, symbols, languages detected + +### After running + +Tell the user: +- "Edit `.shadow/.shadowignore` to exclude files that shouldn't be shadowed" +- "Run `/shadow-frog-dream` for autonomous exploration, or `/shadow-frog-update` after your next changes" + +Then decide the version-control mode (do not skip this — it is not handled +by the script). Ask the user: "Should `.shadow/` be **committed** (shared +with your team via git) or **gitignored** (local to your machine only)?" +Make the trade-off explicit before they choose — see +[Step 9: Handle .gitignore](#9-handle-gitignore) for the full committed vs +gitignored comparison. Key caveat: a gitignored `.shadow/` disables +`shadow-frog-dream` (dreams move `.shadow/` through git). If gitignored, add +`.shadow/` to `.gitignore`. + +## Fallback: Manual Init + +If the Python script fails (wrong Python version, missing file, etc.), +follow these steps manually: + +### 1. Check preconditions + +```bash +git rev-parse --is-inside-work-tree # must be a git repo +test -d .shadow && echo "exists" # if exists, ask user: reset or skip +``` + +### 2. Discover source files + +```bash +git ls-files --cached --others --exclude-standard +``` + +Include patterns (auto-detect from repo contents): +`*.py`, `*.js`, `*.ts`, `*.tsx`, `*.jsx`, `*.java`, `*.go`, `*.rs`, `*.rb`, +`*.cpp`, `*.c`, `*.h`, `*.cs`, `*.swift`, `*.kt`, `*.scala`, `*.php`, +`*.sh`, `*.bash`, `*.zsh`, `*.yaml`, `*.yml`, `*.toml`, `*.json`, +`Makefile`, `Dockerfile`, `docker-compose*.yml` + +Default excludes (always applied): `node_modules/`, `vendor/`, `venv/`, +`.venv/`, `__pycache__/`, `*.min.js`, `*.min.css`, `*.map`, `*.lock`, +`dist/`, `build/`, `target/`, `out/`, `.shadow/`, binary files + +After discovering files, filter them through `.shadow/.shadowignore` +(if it exists). The ignore file uses `.gitignore` syntax. + +### 3. Create directories + +```bash +mkdir -p .shadow/_cross .shadow/_meta .shadow/_dreams +# For each source file, create parent dirs: mkdir -p .shadow/<dir>/ +``` + +### 4. Create `.shadow/.shadowignore` + +Uses `.gitignore` syntax. Seed with sensible defaults: + +```gitignore +# Directories +node_modules/ +vendor/ +venv/ +.venv/ +__pycache__/ +dist/ +build/ +target/ +out/ + +# Generated / minified +*.min.js +*.min.css +*.map +*.lock + +# Binary +*.png +*.jpg +*.gif +*.ico +*.woff +*.woff2 +*.ttf +*.eot +*.pdf +*.zip +*.tar.gz + +# The shadow itself +.shadow/ + +# ShadowFrog's own install artifacts (project install copies these here) +.github/skills/shadow-frog*/ +.github/hooks/scripts/shadow-frog-* +.claude/skills/shadow-frog*/ +.claude/hooks/scripts/shadow-frog-* +``` + +Tell the user: "Edit `.shadow/.shadowignore` to exclude files or +folders that shouldn't be shadowed (e.g., vendored code, generated +files, tool configs)." + +### 5. Create `_prefs.md` + +```markdown +# Preferences + +_No preferences recorded yet._ +``` + +This file stores project-wide user preferences and conventions that are +not tied to any specific file or symbol. It is populated by +`/shadow-frog-update` when the user shares general directives. + +### 6. Create `_meta/state.json` + +```json +{ + "version": 1, + "initialized_at": "<ISO timestamp>", + "last_update_at": "<ISO timestamp>", + "last_commit": "<full 40-char HEAD SHA>", + "last_update_type": "init|auto|manual|dream|meditate", + "total_files": 0, + "total_symbols": 0, + "total_discoveries": 0, + "dream_cycles_completed": 0 +} +``` + +### 7. Generate per-file shadows + +For each source file, extract symbols and create a shadow with this structure: + +```markdown +# Shadow: <path/to/file.py> + +**Language**: <lang> | **Lines**: <N> | **Last modified**: <date> + +## File-Level + +_No discoveries yet._ + +## `class <ClassName>` + +### `<ClassName.method>` + +_No discoveries yet._ + +## `<function_name>` + +_No discoveries yet._ + +## Cross-References + +_No cross-cutting discoveries yet._ +``` + +Rules: +- Every class, function, method gets a `##`/`###` heading +- Symbol name is the stable anchor — no line numbers in headings +- `## Cross-References` section is always last +- Extract symbols using language-appropriate static analysis (AST, tree-sitter, regex) +- If static analysis is not feasible, create `## File-Level` only; other skills fill in symbols later + +### 8. Generate `_index.md` + +```markdown +# Shadow Index + +> Generated by shadow-frog-init on <date> +> Total files: N | Symbols: M | Discoveries: 0 | Cross-cutting: 0 + +| File | Language | Symbols | Discoveries | +|------|----------|---------|-------------| +| src/auth.py | Python | 5 (UserAuth, authenticate_user, ...) | 0 | +``` + +### 9. Handle .gitignore + +Ask the user: "Should `.shadow/` be **committed** (shared with your team via +git) or **gitignored** (local to your machine only)?" + +Before they decide, make the trade-off explicit: + +**Committed (shared) — full functionality:** +- The whole team shares one knowledge base; discoveries compound across people. +- `shadow-frog-dream` works — autonomous experiments commit `.shadow/` + artifacts onto dream branches, push them, and reconcile merges them back. +- `shadow-frog-meditate` and the viewer work normally. + +**Gitignored (local only) — reduced functionality:** +- Your shadow stays private to your machine and never leaves the repo. +- `shadow-frog-init`, `shadow-frog-update`, `shadow-frog-meditate`, and the + viewer all still work (they operate on the local filesystem). +- **`shadow-frog-dream` will NOT work.** Dreams move `.shadow/` through git + (commit → push → reconcile from the remote); a gitignored `.shadow/` is + silently skipped by `git add`, so discoveries never reach the remote and + are lost. `dream-setup.sh` detects this and refuses to start with a clear + error rather than failing silently. + +- If gitignored: add `.shadow/` to `.gitignore`. + +### 10. Report + +Print: files discovered, languages detected, total symbols. +Suggest: `/shadow-frog-update` for deeper analysis, `/shadow-frog-dream` for autonomous exploration. diff --git a/skills/shadow-frog-init/shadow-init.py b/skills/shadow-frog-init/shadow-init.py new file mode 100755 index 0000000..9945185 --- /dev/null +++ b/skills/shadow-frog-init/shadow-init.py @@ -0,0 +1,1513 @@ +#!/usr/bin/env python3 +""" +shadow-init.py — Initialize a .shadow/ knowledge base for any codebase. + +Part of the ShadowFrog skills suite. Discovers source files via git, +extracts symbols with language-aware regex, and creates per-file shadow +markdown stubs plus scaffolding (_index.md, _prefs.md, state.json, etc.). + +Usage: + shadow-init.py [--root DIR] [--reset] [--dry-run] + +Exit codes: + 0 Success (possibly with warnings on stderr) + 1 Fatal error (not a git repo, permissions, etc.) +""" + +import argparse +import json +import os +import re +import shutil +import subprocess +import sys +import traceback +from collections import defaultdict +from datetime import datetime, timezone +from pathlib import Path + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +_warning_count = 0 +_error_count = 0 +_diagnostics = [] # collect (level, msg) for end-of-run summary + + +def warn(msg): + """Print a warning to stderr. Agents read these to adjust strategy.""" + global _warning_count + _warning_count += 1 + _diagnostics.append(("warning", msg)) + print(f"[shadow-init warning] {msg}", file=sys.stderr) + + +def error(msg): + """Print an error to stderr.""" + global _error_count + _error_count += 1 + _diagnostics.append(("error", msg)) + print(f"[shadow-init error] {msg}", file=sys.stderr) + + +def run_git(args, cwd=None): + """Run a git command and return stdout. Returns None on failure. + + Strips GIT_DIR / GIT_WORK_TREE from the environment so git auto-detects + from .git (file or directory) in the cwd. This is critical in worktrees + where inherited env vars would point git to the wrong repository. + """ + try: + env = os.environ.copy() + env.pop("GIT_DIR", None) + env.pop("GIT_WORK_TREE", None) + result = subprocess.run( + ["git"] + args, + capture_output=True, text=True, timeout=30, cwd=cwd, env=env, + ) + if result.returncode != 0: + warn(f"git {' '.join(args)} failed: {result.stderr.strip()}") + return None + return result.stdout + except FileNotFoundError: + error("git is not installed or not on PATH") + return None + except subprocess.TimeoutExpired: + warn(f"git {' '.join(args)} timed out after 30s") + return None + except Exception as e: + warn(f"git {' '.join(args)} error: {e}") + return None + + +def find_repo_root(): + """Auto-detect the git repository root. Works in regular repos, worktrees, + and sub-worktrees. Returns None with diagnostics on failure.""" + # First, check if we're inside a git repo at all + check = run_git(["rev-parse", "--is-inside-work-tree"]) + if check is None or check.strip() != "true": + # Not a git repo — maybe the agent's CWD is wrong + warn("Not inside a git work tree. " + "If in a worktree, make sure the .git file exists and points to a valid gitdir.") + return None + + out = run_git(["rev-parse", "--show-toplevel"]) + if out is None: + # show-toplevel failed — common in worktrees where .git is a file + # pointing to a non-existent or inaccessible gitdir path + git_path = Path(".git") + if git_path.is_file(): + gitdir_line = git_path.read_text().strip() + warn(f"git rev-parse --show-toplevel failed. " + f".git is a file (worktree): {gitdir_line}. " + "The gitdir path may be inaccessible (e.g., running inside Docker " + "with a host-side .git reference). Use --root DIR to specify the root.") + else: + warn("git rev-parse --show-toplevel failed. Use --root DIR to specify the root.") + return None + + root = out.strip() + if not root: + warn("git rev-parse --show-toplevel returned empty. Use --root DIR to specify the root.") + return None + + # Validate the detected root actually exists and has git + root_path = Path(root) + if not root_path.is_dir(): + warn(f"Detected root {root} is not a directory. Use --root DIR to override.") + return None + + # Log whether this is a worktree + git_obj = root_path / ".git" + if git_obj.is_file(): + print(f"[shadow-init] Detected worktree root: {root}", file=sys.stderr) + else: + print(f"[shadow-init] Detected repo root: {root}", file=sys.stderr) + + return root + + +# --------------------------------------------------------------------------- +# File discovery +# --------------------------------------------------------------------------- + +SOURCE_EXTENSIONS = { + # Python + ".py", + # JavaScript / TypeScript + ".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs", + # Java / Kotlin / Scala + ".java", ".kt", ".kts", ".scala", + # Go + ".go", + # Rust + ".rs", + # Ruby + ".rb", + # C / C++ + ".c", ".h", ".cpp", ".hpp", ".cc", ".hh", ".cxx", ".hxx", + # C# + ".cs", + # PHP + ".php", + # Shell + ".sh", ".bash", ".zsh", + # Swift + ".swift", + # Config / data + ".yaml", ".yml", ".toml", ".json", +} + +# Basenames (no extension) that are source files +SOURCE_BASENAMES = {"Makefile", "Dockerfile", "Containerfile", "Rakefile", "Gemfile"} + +EXCLUDE_DIRS = { + "node_modules", "vendor", "venv", ".venv", "__pycache__", + "dist", "build", "target", "out", ".shadow", +} + +EXCLUDE_PATTERNS_SUFFIX = [".min.js", ".min.css", ".map", ".lock"] + +# ShadowFrog's own install artifacts. A project install copies the skills and +# hook scripts into the target repo's .github/, where they are untracked source +# files that `git ls-files --others` surfaces — so init would otherwise shadow +# ShadowFrog's own internals on a user's first run. These prefixes are matched +# against the POSIX-normalized relative path. +SHADOWFROG_ARTIFACT_PREFIXES = ( + ".github/skills/shadow-frog", + ".github/hooks/scripts/shadow-frog", + ".claude/skills/shadow-frog", + ".claude/hooks/scripts/shadow-frog", +) + + +def _is_excluded_path(rel_path): + """Check if a path should be excluded based on directory or suffix rules.""" + norm = Path(rel_path).as_posix() + for prefix in SHADOWFROG_ARTIFACT_PREFIXES: + if norm.startswith(prefix): + return True + parts = Path(rel_path).parts + for part in parts: + if part in EXCLUDE_DIRS: + return True + for suffix in EXCLUDE_PATTERNS_SUFFIX: + if rel_path.endswith(suffix): + return True + return False + + +def _is_source_file(rel_path): + """Check if a file matches known source extensions or basenames.""" + p = Path(rel_path) + if p.name in SOURCE_BASENAMES: + return True + return p.suffix.lower() in SOURCE_EXTENSIONS + + +def _walk_files(repo_root): + """Fallback file discovery via os.walk when git is unavailable.""" + root = Path(repo_root) + result = [] + for dirpath, dirnames, filenames in os.walk(root): + # Prune excluded directories and `.git` only (not all dot-dirs): + # the git path (discover_files) includes hidden source dirs like + # `.github/`, so the fallback must too, or shadows differ by path. + # The later _is_source_file / _is_excluded_path filters still apply. + dirnames[:] = [ + d for d in dirnames + if d not in EXCLUDE_DIRS and d != ".git" + ] + for fname in filenames: + full = Path(dirpath) / fname + try: + rel = str(full.relative_to(root)) + except ValueError: + continue + result.append(rel) + return result + + +def _load_shadowignore(shadow_dir): + """Load .shadowignore and return a matcher function. + + Uses pathspec if available, falls back to fnmatch. + Returns a callable(rel_path) -> bool (True = ignored). + """ + ignore_file = shadow_dir / ".shadowignore" + if not ignore_file.exists(): + return lambda _: False + + try: + lines = ignore_file.read_text(encoding="utf-8").splitlines() + except Exception as e: + warn(f"Could not read .shadowignore: {e}") + return lambda _: False + + patterns = [] + for line in lines: + stripped = line.strip() + if not stripped or stripped.startswith("#"): + continue + patterns.append(stripped) + + if not patterns: + return lambda _: False + + # Try pathspec first (proper gitignore semantics) + try: + import pathspec + + # Newer pathspec deprecates the "gitwildmatch" factory in favor of + # "gitignore"; fall back to "gitwildmatch" for older versions. + try: + spec = pathspec.PathSpec.from_lines("gitignore", patterns) + except (ValueError, KeyError, LookupError): + spec = pathspec.PathSpec.from_lines("gitwildmatch", patterns) + return lambda path: spec.match_file(path) + except ImportError: + pass + except Exception as e: + warn(f"pathspec library loaded but failed to parse .shadowignore: {e}. Falling back to fnmatch.") + + # Fallback: fnmatch-based matching + import fnmatch + + def _match(rel_path): + for pat in patterns: + # Directory pattern: match any path component + if pat.endswith("/"): + dir_pat = pat.rstrip("/") + for part in Path(rel_path).parts: + if fnmatch.fnmatch(part, dir_pat): + return True + else: + if fnmatch.fnmatch(rel_path, pat): + return True + if fnmatch.fnmatch(Path(rel_path).name, pat): + return True + return False + + return _match + + +def discover_files(repo_root, shadow_dir): + """Return sorted list of relative paths to source files. + + Tries git ls-files first. Falls back to filesystem walk if git is + unavailable (e.g., inside a Docker container where .git references + are broken). + """ + out = run_git( + ["ls-files", "--cached", "--others", "--exclude-standard"], + cwd=repo_root, + ) + if out is None: + warn("git ls-files failed. Falling back to filesystem walk. " + "This may include files that would normally be gitignored.") + all_files = _walk_files(repo_root) + else: + all_files = [f for f in out.splitlines() if f.strip()] + + # Apply built-in filters + filtered = [] + for f in all_files: + if _is_excluded_path(f): + continue + if not _is_source_file(f): + continue + filtered.append(f) + + # Apply .shadowignore + is_ignored = _load_shadowignore(shadow_dir) + result = [] + for f in filtered: + try: + if is_ignored(f): + continue + except Exception as e: + warn(f"Shadowignore matcher failed on '{f}': {e}. Including file anyway.") + result.append(f) + + return sorted(set(result)) + + +# --------------------------------------------------------------------------- +# Language detection +# --------------------------------------------------------------------------- + +EXTENSION_TO_LANG = { + ".py": "Python", + ".js": "JavaScript", ".jsx": "JavaScript", + ".mjs": "JavaScript", ".cjs": "JavaScript", + ".ts": "TypeScript", ".tsx": "TypeScript", + ".java": "Java", + ".kt": "Kotlin", ".kts": "Kotlin", + ".scala": "Scala", + ".go": "Go", + ".rs": "Rust", + ".rb": "Ruby", + ".c": "C", ".h": "C", + ".cpp": "C++", ".hpp": "C++", ".cc": "C++", + ".hh": "C++", ".cxx": "C++", ".hxx": "C++", + ".cs": "C#", + ".php": "PHP", + ".sh": "Shell", ".bash": "Shell", ".zsh": "Shell", + ".swift": "Swift", + ".yaml": "YAML", ".yml": "YAML", + ".toml": "TOML", + ".json": "JSON", +} + +BASENAME_TO_LANG = { + "Makefile": "Makefile", + "Dockerfile": "Dockerfile", + "Containerfile": "Dockerfile", + "Rakefile": "Ruby", + "Gemfile": "Ruby", +} + + +def detect_language(rel_path): + p = Path(rel_path) + if p.name in BASENAME_TO_LANG: + return BASENAME_TO_LANG[p.name] + return EXTENSION_TO_LANG.get(p.suffix.lower(), "Unknown") + + +# --------------------------------------------------------------------------- +# Symbol extraction +# --------------------------------------------------------------------------- + +class Symbol: + """Represents an extracted symbol.""" + __slots__ = ("name", "kind", "parent") + + def __init__(self, name, kind, parent=None): + self.name = name # e.g. "UserAuth" or "validate" + self.kind = kind # "class", "function", "method", "interface", etc. + self.parent = parent # parent class name or None + + @property + def display_name(self): + if self.parent: + return f"{self.parent}.{self.name}" + return self.name + + @property + def heading_text(self): + """Text for the markdown heading (backtick-wrapped).""" + if self.kind == "class": + return f"class {self.name}" + if self.kind == "interface": + return f"interface {self.name}" + if self.kind == "enum": + return f"enum {self.name}" + if self.kind == "trait": + return f"trait {self.name}" + if self.kind == "struct": + return f"struct {self.name}" + if self.kind == "protocol": + return f"protocol {self.name}" + if self.kind == "module": + return f"module {self.name}" + if self.parent: + return f"{self.parent}.{self.name}" + return self.name + + @property + def is_container(self): + return self.kind in ("class", "interface", "enum", "trait", + "struct", "protocol", "module") + + +# --- Per-language regex extractors --- + +def _extract_python(source): + symbols = [] + current_class = None + class_indent = -1 + + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.lstrip() + indent = len(line) - len(stripped) + + # Track class scope by indentation + if current_class and indent <= class_indent and stripped: + current_class = None + class_indent = -1 + + m = re.match(r'^class\s+([A-Za-z_]\w*)', stripped) + if m: + current_class = m.group(1) + class_indent = indent + symbols.append(Symbol(current_class, "class")) + continue + + m = re.match(r'^(?:async\s+)?def\s+([A-Za-z_]\w*)', stripped) + if m: + name = m.group(1) + if current_class and indent > class_indent: + symbols.append(Symbol(name, "method", parent=current_class)) + else: + symbols.append(Symbol(name, "function")) + if current_class and indent <= class_indent: + current_class = None + class_indent = -1 + continue + + # Module-level ALL_CAPS constants/globals (indent 0, not inside a + # class). These often carry real behavioral weight (shared mutable + # caches, sentinels, tunables). ALL_CAPS keeps this low-noise vs. + # capturing every lowercase temp assignment. + if current_class is None and indent == 0: + m = re.match( + r'^([A-Z_][A-Z0-9_]*)\s*(?::\s*[^=]+?)?\s*=(?!=)', + stripped, + ) + if m: + symbols.append(Symbol(m.group(1), "constant")) + continue + except Exception as e: + warn(f"Python extractor: error at line {line_num}: {e}") + continue + + return symbols + + +def _extract_javascript(source): + symbols = [] + current_class = None + class_brace_depth = 0 + brace_depth = 0 + + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.strip() + + # Track braces for class scope + open_braces = stripped.count("{") + close_braces = stripped.count("}") + + # Class declaration + m = re.match(r'^(?:export\s+(?:default\s+)?)?class\s+([A-Za-z_$]\w*)', stripped) + if m: + current_class = m.group(1) + class_brace_depth = brace_depth + symbols.append(Symbol(current_class, "class")) + brace_depth += open_braces - close_braces + continue + + # Standalone / export function + m = re.match(r'^(?:export\s+(?:default\s+)?)?(?:async\s+)?function\s*\*?\s*([A-Za-z_$]\w*)', stripped) + if m: + name = m.group(1) + if current_class and brace_depth > class_brace_depth: + symbols.append(Symbol(name, "method", parent=current_class)) + else: + symbols.append(Symbol(name, "function")) + brace_depth += open_braces - close_braces + continue + + # Method inside class (name(...) { or async name(...) {) + if current_class and brace_depth > class_brace_depth: + m = re.match(r'^(?:async\s+)?(?:static\s+)?(?:get\s+|set\s+)?([A-Za-z_$]\w*)\s*\(', stripped) + if m and m.group(1) not in ("if", "for", "while", "switch", "catch", "return", "new"): + symbols.append(Symbol(m.group(1), "method", parent=current_class)) + brace_depth += open_braces - close_braces + continue + + # const/let/var name = (arrow function or value) + m = re.match(r'^(?:export\s+(?:default\s+)?)?(?:const|let|var)\s+([A-Za-z_$]\w*)\s*=', stripped) + if m: + symbols.append(Symbol(m.group(1), "function")) + brace_depth += open_braces - close_braces + continue + + brace_depth += open_braces - close_braces + if brace_depth < 0: + brace_depth = 0 + if current_class and brace_depth <= class_brace_depth: + current_class = None + class_brace_depth = 0 + except Exception as e: + warn(f"JS/TS extractor: error at line {line_num}: {e}") + continue + + return symbols + + +def _extract_java_like(source): + """Java, Kotlin, C#, Scala.""" + symbols = [] + current_class = None + class_brace_depth = 0 + brace_depth = 0 + + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.strip() + open_braces = stripped.count("{") + close_braces = stripped.count("}") + + # Class / interface / enum + m = re.match( + r'^(?:(?:public|private|protected|internal|abstract|sealed|static|final|open|data)\s+)*' + r'(class|interface|enum)\s+([A-Za-z_]\w*)', + stripped, + ) + if m: + kind = m.group(1) + name = m.group(2) + current_class = name + class_brace_depth = brace_depth + symbols.append(Symbol(name, kind)) + brace_depth += open_braces - close_braces + continue + + # Method signatures (access modifier + return type + name) + if current_class and brace_depth > class_brace_depth: + m = re.match( + r'^(?:(?:public|private|protected|internal|abstract|static|final|override|open|suspend|virtual|async)\s+)*' + r'(?:(?:fun|void|int|long|float|double|boolean|char|byte|short|string|String|var|val|Task|IActionResult|' + r'[A-Z]\w*(?:<[^>]*>)?)\s+)' + r'([A-Za-z_]\w*)\s*(?:<[^>]*>)?\s*\(', + stripped, + ) + if m and m.group(1) not in ("if", "for", "while", "switch", "catch", "return", "new"): + symbols.append(Symbol(m.group(1), "method", parent=current_class)) + brace_depth += open_braces - close_braces + continue + + # Kotlin fun keyword + m = re.match( + r'^(?:(?:public|private|protected|internal|abstract|override|open|suspend)\s+)*' + r'fun\s+(?:<[^>]*>\s*)?([A-Za-z_]\w*)', + stripped, + ) + if m: + symbols.append(Symbol(m.group(1), "method", parent=current_class)) + brace_depth += open_braces - close_braces + continue + + brace_depth += open_braces - close_braces + if brace_depth < 0: + brace_depth = 0 + if current_class and brace_depth <= class_brace_depth: + current_class = None + class_brace_depth = 0 + except Exception as e: + warn(f"Java/Kotlin/C# extractor: error at line {line_num}: {e}") + continue + + return symbols + + +def _extract_go(source): + symbols = [] + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.strip() + + # type Foo struct/interface + m = re.match(r'^type\s+([A-Za-z_]\w*)\s+(struct|interface)\b', stripped) + if m: + kind = "struct" if m.group(2) == "struct" else "interface" + symbols.append(Symbol(m.group(1), kind)) + continue + + # func (receiver) Method(...) — method + m = re.match(r'^func\s+\(\s*\w+\s+\*?([A-Za-z_]\w*)\s*\)\s*([A-Za-z_]\w*)', stripped) + if m: + symbols.append(Symbol(m.group(2), "method", parent=m.group(1))) + continue + + # func Foo(...) — top-level function + m = re.match(r'^func\s+([A-Za-z_]\w*)', stripped) + if m: + symbols.append(Symbol(m.group(1), "function")) + continue + except Exception as e: + warn(f"Go extractor: error at line {line_num}: {e}") + continue + + return symbols + + +def _extract_rust(source): + symbols = [] + current_impl = None + impl_brace_depth = 0 + brace_depth = 0 + + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.strip() + open_braces = stripped.count("{") + close_braces = stripped.count("}") + + # struct / enum / trait + m = re.match(r'^(?:pub(?:\(crate\))?\s+)?(struct|enum|trait)\s+([A-Za-z_]\w*)', stripped) + if m: + symbols.append(Symbol(m.group(2), m.group(1))) + brace_depth += open_braces - close_braces + continue + + # impl Block + m = re.match(r'^impl(?:<[^>]*>)?\s+(?:[A-Za-z_]\w*\s+for\s+)?([A-Za-z_]\w*)', stripped) + if m: + current_impl = m.group(1) + impl_brace_depth = brace_depth + brace_depth += open_braces - close_braces + continue + + # fn + m = re.match(r'^(?:pub(?:\(crate\))?\s+)?(?:async\s+)?(?:unsafe\s+)?(?:const\s+)?fn\s+([A-Za-z_]\w*)', stripped) + if m: + name = m.group(1) + if current_impl and brace_depth > impl_brace_depth: + symbols.append(Symbol(name, "method", parent=current_impl)) + else: + symbols.append(Symbol(name, "function")) + brace_depth += open_braces - close_braces + continue + + brace_depth += open_braces - close_braces + if brace_depth < 0: + brace_depth = 0 + if current_impl and brace_depth <= impl_brace_depth: + current_impl = None + impl_brace_depth = 0 + except Exception as e: + warn(f"Rust extractor: error at line {line_num}: {e}") + continue + + return symbols + + +def _extract_ruby(source): + symbols = [] + current_class = None + class_depth = 0 + depth = 0 # keyword-level nesting (class/module/def/do ... end) + + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.strip() + + # Approximate depth tracking via keywords + opens = len(re.findall(r'\b(class|module|def|do|begin|if|unless|case|while|until|for)\b', stripped)) + closes = len(re.findall(r'\bend\b', stripped)) + + m = re.match(r'^(class|module)\s+([A-Za-z_]\w*)', stripped) + if m: + kind = m.group(1) + name = m.group(2) + if kind == "class": + current_class = name + class_depth = depth + symbols.append(Symbol(name, "class")) + else: + symbols.append(Symbol(name, "module")) + depth += opens - closes + continue + + m = re.match(r'^def\s+(self\.)?([A-Za-z_]\w*[!?=]?)', stripped) + if m: + name = m.group(2) + if current_class and depth > class_depth: + symbols.append(Symbol(name, "method", parent=current_class)) + else: + symbols.append(Symbol(name, "function")) + depth += opens - closes + continue + + depth += opens - closes + if depth < 0: + depth = 0 + if current_class and depth <= class_depth: + current_class = None + class_depth = 0 + except Exception as e: + warn(f"Ruby extractor: error at line {line_num}: {e}") + continue + + return symbols + + +def _extract_c_cpp(source): + symbols = [] + current_class = None + class_brace_depth = 0 + brace_depth = 0 + + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.strip() + if stripped.startswith("//") or stripped.startswith("#"): + continue + + open_braces = stripped.count("{") + close_braces = stripped.count("}") + + # class / struct + m = re.match(r'^(?:template\s*<[^>]*>\s*)?(?:class|struct)\s+([A-Za-z_]\w*)', stripped) + if m: + current_class = m.group(1) + class_brace_depth = brace_depth + symbols.append(Symbol(current_class, "class")) + brace_depth += open_braces - close_braces + continue + + # Function/method: return_type name(...) + m = re.match( + r'^(?:(?:static|virtual|inline|extern|const|unsigned|signed|volatile)\s+)*' + r'(?:[A-Za-z_]\w*(?:::[A-Za-z_]\w*)*[\s*&]*\s+)' + r'(?:([A-Za-z_]\w*)::)?([A-Za-z_]\w*)\s*\(', + stripped, + ) + if m: + scope = m.group(1) + name = m.group(2) + if name in ("if", "for", "while", "switch", "catch", "return", "sizeof", "typeof"): + brace_depth += open_braces - close_braces + continue + if scope: + symbols.append(Symbol(name, "method", parent=scope)) + elif current_class and brace_depth > class_brace_depth: + symbols.append(Symbol(name, "method", parent=current_class)) + else: + symbols.append(Symbol(name, "function")) + brace_depth += open_braces - close_braces + continue + + brace_depth += open_braces - close_braces + if brace_depth < 0: + brace_depth = 0 + if current_class and brace_depth <= class_brace_depth: + current_class = None + class_brace_depth = 0 + except Exception as e: + warn(f"C/C++ extractor: error at line {line_num}: {e}") + continue + + return symbols + + +def _extract_php(source): + symbols = [] + current_class = None + class_brace_depth = 0 + brace_depth = 0 + + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.strip() + open_braces = stripped.count("{") + close_braces = stripped.count("}") + + m = re.match( + r'^(?:(?:abstract|final)\s+)?(?:class|interface|trait)\s+([A-Za-z_]\w*)', + stripped, + ) + if m: + current_class = m.group(1) + class_brace_depth = brace_depth + symbols.append(Symbol(current_class, "class")) + brace_depth += open_braces - close_braces + continue + + m = re.match( + r'^(?:(?:public|private|protected|static|abstract|final)\s+)*function\s+([A-Za-z_]\w*)', + stripped, + ) + if m: + name = m.group(1) + if current_class and brace_depth > class_brace_depth: + symbols.append(Symbol(name, "method", parent=current_class)) + else: + symbols.append(Symbol(name, "function")) + brace_depth += open_braces - close_braces + continue + + brace_depth += open_braces - close_braces + if brace_depth < 0: + brace_depth = 0 + if current_class and brace_depth <= class_brace_depth: + current_class = None + class_brace_depth = 0 + except Exception as e: + warn(f"PHP extractor: error at line {line_num}: {e}") + continue + + return symbols + + +def _extract_shell(source): + symbols = [] + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.strip() + + # function foo { ... } + m = re.match(r'^function\s+([A-Za-z_]\w*)', stripped) + if m: + symbols.append(Symbol(m.group(1), "function")) + continue + + # foo() { ... } + m = re.match(r'^([A-Za-z_]\w*)\s*\(\s*\)', stripped) + if m: + symbols.append(Symbol(m.group(1), "function")) + continue + except Exception as e: + warn(f"Shell extractor: error at line {line_num}: {e}") + continue + + return symbols + + +def _extract_swift(source): + symbols = [] + current_class = None + class_brace_depth = 0 + brace_depth = 0 + + for line_num, line in enumerate(source.splitlines(), 1): + try: + stripped = line.strip() + open_braces = stripped.count("{") + close_braces = stripped.count("}") + + # class / struct / enum / protocol + m = re.match( + r'^(?:(?:public|private|fileprivate|internal|open|final)\s+)*' + r'(class|struct|enum|protocol)\s+([A-Za-z_]\w*)', + stripped, + ) + if m: + kind = m.group(1) + name = m.group(2) + current_class = name + class_brace_depth = brace_depth + symbols.append(Symbol(name, kind)) + brace_depth += open_braces - close_braces + continue + + # func + m = re.match( + r'^(?:(?:public|private|fileprivate|internal|open|static|override|class|mutating)\s+)*' + r'func\s+([A-Za-z_]\w*)', + stripped, + ) + if m: + name = m.group(1) + if current_class and brace_depth > class_brace_depth: + symbols.append(Symbol(name, "method", parent=current_class)) + else: + symbols.append(Symbol(name, "function")) + brace_depth += open_braces - close_braces + continue + + brace_depth += open_braces - close_braces + if brace_depth < 0: + brace_depth = 0 + if current_class and brace_depth <= class_brace_depth: + current_class = None + class_brace_depth = 0 + except Exception as e: + warn(f"Swift extractor: error at line {line_num}: {e}") + continue + + return symbols + + +# Dispatcher +EXTRACTORS = { + "Python": _extract_python, + "JavaScript": _extract_javascript, + "TypeScript": _extract_javascript, + "Java": _extract_java_like, + "Kotlin": _extract_java_like, + "C#": _extract_java_like, + "Scala": _extract_java_like, + "Go": _extract_go, + "Rust": _extract_rust, + "Ruby": _extract_ruby, + "C": _extract_c_cpp, + "C++": _extract_c_cpp, + "PHP": _extract_php, + "Shell": _extract_shell, + "Swift": _extract_swift, +} + + +def extract_symbols(source, language, rel_path="<unknown>"): + """Extract symbols from source code. Returns list of Symbol objects. + + Never raises — returns empty list on any failure, with diagnostic warnings. + """ + extractor = EXTRACTORS.get(language) + if extractor is None: + return [] + try: + symbols = extractor(source) + # Validate symbols — catch malformed Symbol objects + for s in symbols: + if not isinstance(s.name, str) or not s.name: + warn(f"Symbol extractor for {language} produced empty name in {rel_path}. Skipping symbol.") + symbols = [s for s in symbols if isinstance(s.name, str) and s.name] + break + return symbols + except re.error as e: + warn(f"Regex error during {language} symbol extraction for {rel_path}: {e}. " + "This may indicate unusual syntax in the source file. Returning no symbols.") + return [] + except Exception as e: + warn(f"Symbol extraction failed for {rel_path} ({language}): {type(e).__name__}: {e}. " + "The shadow file will still be created with a File-Level section only.") + return [] + + +# --------------------------------------------------------------------------- +# Shadow file generation +# --------------------------------------------------------------------------- + +def get_last_modified(rel_path, repo_root): + """Get the last commit date for a file via git log. + + Returns a date string like '2025-01-15', or 'unknown' on any failure. + """ + try: + out = run_git(["log", "-1", "--format=%ci", "--", rel_path], cwd=repo_root) + if out and out.strip(): + # Format: "2025-01-15 10:00:00 -0500" → take date part + parts = out.strip().split(" ") + if parts and len(parts[0]) >= 8: + return parts[0] + warn(f"Unexpected git log date format for {rel_path}: '{out.strip()}'") + return "unknown" + except Exception as e: + warn(f"Could not get last modified date for {rel_path}: {e}") + return "unknown" + + +def count_lines(source): + return len(source.splitlines()) + + +def build_shadow_content(rel_path, language, line_count, last_modified, symbols): + """Build the markdown content for a per-file shadow. + + Never raises — returns a minimal valid shadow on any failure. + """ + try: + lines = [] + lines.append(f"# Shadow: {rel_path}") + lines.append("") + lines.append(f"**Language**: {language} | **Lines**: {line_count} | **Last modified**: {last_modified}") + lines.append("") + lines.append("## File-Level") + lines.append("") + lines.append("_No discoveries yet._") + + # Group symbols: containers with their children + i = 0 + while i < len(symbols): + try: + sym = symbols[i] + if sym.parent is None: + lines.append("") + lines.append(f"## `{sym.heading_text}`") + if sym.is_container: + # Collect child symbols belonging to this container + children = [] + j = i + 1 + while j < len(symbols) and symbols[j].parent == sym.name: + children.append(symbols[j]) + j += 1 + if not children: + lines.append("") + lines.append("_No discoveries yet._") + else: + for child in children: + lines.append("") + lines.append(f"### `{child.heading_text}`") + lines.append("") + lines.append("_No discoveries yet._") + i = j + continue + else: + lines.append("") + lines.append("_No discoveries yet._") + else: + # Orphan nested symbol (parent wasn't found as container) + lines.append("") + lines.append(f"### `{sym.heading_text}`") + lines.append("") + lines.append("_No discoveries yet._") + except Exception as e: + warn(f"Error formatting symbol #{i} in {rel_path}: {e}. Skipping this symbol.") + i += 1 + + lines.append("") + lines.append("## Cross-References") + lines.append("") + lines.append("_No cross-cutting discoveries yet._") + lines.append("") + + return "\n".join(lines) + + except Exception as e: + # Fallback: return a minimal valid shadow with no symbols + warn(f"Failed to build shadow content for {rel_path}: {type(e).__name__}: {e}. " + "Creating minimal shadow with File-Level section only.") + return ( + f"# Shadow: {rel_path}\n\n" + f"**Language**: {language} | **Lines**: {line_count} | **Last modified**: {last_modified}\n\n" + f"## File-Level\n\n" + f"_No discoveries yet._\n\n" + f"## Cross-References\n\n" + f"_No cross-cutting discoveries yet._\n" + ) + + +# --------------------------------------------------------------------------- +# Scaffolding +# --------------------------------------------------------------------------- + +SHADOWIGNORE_CONTENT = """\ +# Directories +node_modules/ +vendor/ +venv/ +.venv/ +__pycache__/ +dist/ +build/ +target/ +out/ + +# Generated / minified +*.min.js +*.min.css +*.map +*.lock + +# Binary +*.png +*.jpg +*.gif +*.ico +*.woff +*.woff2 +*.ttf +*.eot +*.pdf +*.zip +*.tar.gz + +# The shadow itself +.shadow/ + +# ShadowFrog's own install artifacts (project install copies these here) +.github/skills/shadow-frog*/ +.github/hooks/scripts/shadow-frog-* +.claude/skills/shadow-frog*/ +.claude/hooks/scripts/shadow-frog-* +""" + +PREFS_CONTENT = """\ +# Preferences + +_No preferences recorded yet._ +""" + + +def build_state_json(total_files, total_symbols, last_commit): + """Build state.json dict. last_commit should be a 40-char SHA or "none" + (sentinel for non-git or empty repos). Downstream hooks rely on "none" + to distinguish missing-commit from empty-string.""" + try: + now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + except Exception: + now = "unknown" + return { + "version": 1, + "initialized_at": now, + "last_update_at": now, + "last_commit": last_commit or "none", + "last_update_type": "init", + "total_files": total_files, + "total_symbols": total_symbols, + "total_discoveries": 0, + "dream_cycles_completed": 0, + } + + +def build_index(file_records, total_symbols, total_discoveries=0, cross_cutting=0): + """Build _index.md content. + + file_records: list of (rel_path, language, symbols_list) + Never raises — returns a valid index even if some rows fail. + """ + try: + now = datetime.now(timezone.utc).strftime("%Y-%m-%d") + except Exception: + now = "unknown" + total_files = len(file_records) + + lines = [] + lines.append("# Shadow Index") + lines.append("") + lines.append( + f"> Generated by shadow-frog-init on {now}" + ) + lines.append( + f"> Total files: {total_files} | Symbols: {total_symbols} " + f"| Discoveries: {total_discoveries} | Cross-cutting: {cross_cutting}" + ) + lines.append("") + lines.append("| File | Language | Symbols | Discoveries |") + lines.append("|------|----------|---------|-------------|") + + for rel_path, language, symbols in file_records: + try: + sym_count = len(symbols) + if sym_count == 0: + sym_display = "0" + else: + names = [] + for s in symbols: + if s.parent is None: + names.append(s.name) + if not names: + names = [s.display_name for s in symbols] + if len(names) <= 3: + sym_display = f"{sym_count} ({', '.join(names)})" + else: + sym_display = f"{sym_count} ({', '.join(names[:3])}, ...)" + + lines.append(f"| {rel_path} | {language} | {sym_display} | 0 |") + except Exception as e: + warn(f"Failed to build index row for {rel_path}: {e}. Using fallback row.") + lines.append(f"| {rel_path} | {language} | ? | 0 |") + + lines.append("") + return "\n".join(lines) + + +# --------------------------------------------------------------------------- +# Main logic +# --------------------------------------------------------------------------- + +def init_shadow(repo_root, reset=False, dry_run=False): + """Initialize the .shadow/ knowledge base. + + Returns True on success (even with non-fatal warnings), False on fatal errors. + Designed to never crash — all sections are independently protected. + """ + root = Path(repo_root).resolve() + shadow_dir = root / ".shadow" + + # Handle existing shadow + if shadow_dir.exists(): + if not reset: + error( + f".shadow/ already exists at {shadow_dir}. " + "Use --reset to delete and recreate, or remove it manually." + ) + return False + if dry_run: + print(f"[dry-run] Would delete {shadow_dir}") + else: + try: + shutil.rmtree(shadow_dir) + except PermissionError as e: + error(f"Permission denied removing .shadow/: {e}. " + "Check file ownership and permissions.") + return False + except Exception as e: + error(f"Failed to remove existing .shadow/: {type(e).__name__}: {e}") + return False + + # Create base directories + dirs_to_create = [ + shadow_dir, + shadow_dir / "_meta", + shadow_dir / "_cross", + shadow_dir / "_dreams", + ] + if dry_run: + for d in dirs_to_create: + print(f"[dry-run] Would create directory {d.relative_to(root)}") + else: + for d in dirs_to_create: + try: + d.mkdir(parents=True, exist_ok=True) + except PermissionError as e: + error(f"Permission denied creating {d}: {e}. " + "Check write permissions on the repository root.") + return False + except Exception as e: + error(f"Failed to create directory {d}: {type(e).__name__}: {e}") + return False + + # Write .shadowignore first (needed for file discovery filtering) + shadowignore_path = shadow_dir / ".shadowignore" + if dry_run: + print(f"[dry-run] Would create {shadowignore_path.relative_to(root)}") + else: + try: + shadowignore_path.write_text(SHADOWIGNORE_CONTENT, encoding="utf-8") + except Exception as e: + error(f"Failed to write .shadowignore: {e}. " + "Cannot proceed without this file.") + return False + + # Create empty _dreams/_index.md scaffold with the 7-column schema + # the reconciler/lineage parser expect. The extra columns (branch, + # parent, tip_commit) stay blank for the no-dream initial state but + # the header alignment is what dream-reconcile.py / dream-lineage.py + # validate against. + dreams_index = shadow_dir / "_dreams" / "_index.md" + if dry_run: + print(f"[dry-run] Would create {dreams_index.relative_to(root)}") + else: + try: + dreams_index.write_text( + "# Dream Experiment Archive\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n", + encoding="utf-8" + ) + except Exception as e: + warn(f"Failed to write _dreams/_index.md: {e}. " + "Dream archive will be created on first dream run.") + + # Discover files + try: + source_files = discover_files(str(root), shadow_dir) + except Exception as e: + error(f"File discovery crashed: {type(e).__name__}: {e}. " + "Is this a valid git repository?") + source_files = [] + if not source_files: + warn("No source files found. The shadow will be empty. " + "Check that the repository has committed or tracked files, " + "and that .shadowignore isn't excluding everything.") + + # Get latest commit + last_commit_out = run_git(["rev-parse", "HEAD"], cwd=str(root)) + last_commit = last_commit_out.strip() if last_commit_out else "none" + + # Process each file + file_records = [] # (rel_path, language, symbols) + total_symbols = 0 + language_counts = defaultdict(int) + skipped_files = [] + + for rel_path in source_files: + try: + abs_path = root / rel_path + language = detect_language(rel_path) + language_counts[language] += 1 + + # Read file content + try: + source = abs_path.read_text(encoding="utf-8", errors="replace") + except PermissionError: + warn(f"Permission denied reading {rel_path}. Skipping.") + skipped_files.append((rel_path, "permission denied")) + continue + except OSError as e: + warn(f"OS error reading {rel_path}: {e}. " + "File may have been deleted since discovery. Skipping.") + skipped_files.append((rel_path, str(e))) + continue + + line_count = count_lines(source) + symbols = extract_symbols(source, language, rel_path) + total_symbols += len(symbols) + + last_modified = get_last_modified(rel_path, str(root)) + + shadow_content = build_shadow_content( + rel_path, language, line_count, last_modified, symbols, + ) + + # Determine shadow file path + shadow_file = shadow_dir / (rel_path + ".md") + + if dry_run: + print(f"[dry-run] Would create {shadow_file.relative_to(root)} " + f"({language}, {len(symbols)} symbols)") + else: + try: + shadow_file.parent.mkdir(parents=True, exist_ok=True) + shadow_file.write_text(shadow_content, encoding="utf-8") + except PermissionError: + warn(f"Permission denied writing shadow for {rel_path}. Skipping.") + skipped_files.append((rel_path, "write permission denied")) + continue + except OSError as e: + warn(f"OS error writing shadow for {rel_path}: {e}. " + "Path may be too long or contain invalid characters. Skipping.") + skipped_files.append((rel_path, str(e))) + continue + + file_records.append((rel_path, language, symbols)) + + except Exception as e: + warn(f"Unexpected error processing {rel_path}: {type(e).__name__}: {e}. " + "Skipping this file. Other files will still be processed.") + skipped_files.append((rel_path, f"unexpected: {e}")) + continue + + # Write _prefs.md + prefs_path = shadow_dir / "_prefs.md" + if dry_run: + print(f"[dry-run] Would create {prefs_path.relative_to(root)}") + else: + try: + prefs_path.write_text(PREFS_CONTENT, encoding="utf-8") + except Exception as e: + warn(f"Failed to write _prefs.md: {e}. " + "You can create this file manually: '# Preferences\\n\\n_No preferences recorded yet._'") + + # Write _index.md + try: + index_content = build_index(file_records, total_symbols) + except Exception as e: + warn(f"Failed to build index content: {type(e).__name__}: {e}. Writing minimal index.") + index_content = "# Shadow Index\n\n> Index generation failed. Run shadow-frog-update to rebuild.\n" + index_path = shadow_dir / "_index.md" + if dry_run: + print(f"[dry-run] Would create {index_path.relative_to(root)}") + else: + try: + index_path.write_text(index_content, encoding="utf-8") + except Exception as e: + warn(f"Failed to write _index.md: {e}") + + # Write state.json + try: + state = build_state_json(len(file_records), total_symbols, last_commit) + except Exception as e: + warn(f"Failed to build state.json content: {e}. Writing minimal state.") + state = {"version": 1, "last_update_type": "init", + "total_files": len(file_records), "total_symbols": total_symbols, + "total_discoveries": 0, "dream_cycles_completed": 0} + state_path = shadow_dir / "_meta" / "state.json" + if dry_run: + print(f"[dry-run] Would create {state_path.relative_to(root)}") + print(f"[dry-run] state.json: {json.dumps(state, indent=2)}") + else: + try: + state_path.write_text( + json.dumps(state, indent=2) + "\n", encoding="utf-8", + ) + except Exception as e: + warn(f"Failed to write state.json: {e}. " + "The shadow is usable but state tracking will be incomplete.") + + # --- Summary --- + total_files = len(file_records) + try: + lang_summary = ", ".join( + f"{lang} ({count})" + for lang, count in sorted(language_counts.items(), key=lambda x: -x[1]) + ) + except Exception: + lang_summary = f"{len(language_counts)} languages" + + print("") + if dry_run: + print("[dry-run] Shadow would be initialized.") + else: + print("Shadow initialized.") + print(f" Files: {total_files}") + print(f" Symbols: {total_symbols}") + print(f" Languages: {lang_summary or 'none'}") + + if skipped_files: + print(f" Skipped: {len(skipped_files)} files (see warnings above)") + # List skipped files for agent diagnosis + for path, reason in skipped_files: + print(f" - {path}: {reason}", file=sys.stderr) + + if _warning_count or _error_count: + print(f" Diagnostics: {_warning_count} warnings, {_error_count} errors") + + return True + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + +def main(): + parser = argparse.ArgumentParser( + description="Initialize a .shadow/ knowledge base for any codebase.", + prog="shadow-init.py", + ) + parser.add_argument( + "--root", + metavar="DIR", + default=None, + help="Repository root (default: auto-detect via git)", + ) + parser.add_argument( + "--reset", + action="store_true", + help="Delete existing .shadow/ and recreate", + ) + parser.add_argument( + "--dry-run", + action="store_true", + help="Show what would be created without writing", + ) + + args = parser.parse_args() + + try: + # Determine repo root + repo_root = args.root + if repo_root is None: + repo_root = find_repo_root() + if repo_root is None: + error("Could not detect git repository root. " + "Make sure you are inside a git repository, or use --root DIR to specify.") + sys.exit(1) + + repo_root = str(Path(repo_root).resolve()) + if not Path(repo_root).is_dir(): + error(f"Root directory does not exist: {repo_root}") + sys.exit(1) + + success = init_shadow(repo_root, reset=args.reset, dry_run=args.dry_run) + sys.exit(0 if success else 1) + + except KeyboardInterrupt: + error("Interrupted by user.") + sys.exit(130) + except Exception as e: + # Top-level catch-all — should never fire, but if it does, + # give the agent maximum diagnostic context + error(f"Unexpected top-level crash: {type(e).__name__}: {e}") + print("\n--- Full traceback (for agent diagnosis) ---", file=sys.stderr) + traceback.print_exc(file=sys.stderr) + print("--- End traceback ---\n", file=sys.stderr) + print( + "[shadow-init guidance] This is a bug in shadow-init.py. " + "As a workaround, you can initialize .shadow/ manually by following " + "the instructions in shadow-frog-init/SKILL.md under 'Fallback: Manual Init'.", + file=sys.stderr, + ) + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/skills/shadow-frog-meditate/SKILL.md b/skills/shadow-frog-meditate/SKILL.md new file mode 100644 index 0000000..0a6e81a --- /dev/null +++ b/skills/shadow-frog-meditate/SKILL.md @@ -0,0 +1,328 @@ +--- +name: shadow-frog-meditate +description: >- + Clean and consolidate the shadow knowledge base. Scans all shadow files + for duplicate discoveries (same claim, different wording), near-duplicates + (one extends another), and conflicting entries (contradicting claims). + Merges duplicates, resolves conflicts by investigating the code, and + asks the user only when resolution is unclear. Invoke periodically to + keep the shadow focused and free of noise. +scripts: + - meditate-repair.py +--- + +# ShadowFrog Meditate + +Shadow hygiene — deduplicate, merge, and resolve conflicts across the +entire `.shadow/` knowledge base. Prerequisite: `.shadow/` exists with +discoveries. + +## Why Meditate? + +Over time, shadows accumulate noise: +- **Duplicates**: the same insight written differently by different sessions +- **Near-duplicates**: one discovery is a subset of another +- **Conflicts**: two discoveries contradict each other (code may have changed, + or one was wrong) +- **Cross-scope duplicates**: a per-file discovery and a `_cross/` entry + saying the same thing + +This noise confuses downstream agents and dilutes signal. Meditate cleans +it up. + +## Phase 1: Scan + +Use parallel subagents to scan the shadow. Each subagent handles a batch +of shadow files. + +### Scope Optimization + +Not every file needs scanning. To reduce cost: +- **Skip files with 0-1 discoveries** — they can't have internal duplicates +- **Focus on files modified since last meditate** — check `_meta/state.json` + `last_update_at` against file modification times +- **Always scan files with 5+ discoveries** — highest duplicate risk + +For the first meditate after a large dream run, most files will need +scanning. For incremental meditation after small updates, this can +reduce scope by 80%+. + +### Per-File Scan + +For each per-file shadow (e.g., `src/auth.py.md`): + +1. Read all discoveries under each `## symbol` heading +2. For each pair of discoveries under the **same symbol**, classify: + - **Duplicate**: same behavioral claim, different wording + - **Near-duplicate**: one discovery is a subset/refinement of the other + - **Conflict**: the two discoveries make contradicting claims + - **Distinct**: genuinely different insights — no action needed +3. Record each finding as a structured action (see below) + +### Scan Output Format + +Subagents must output findings as **one JSON object per line** so the +orchestrator can auto-apply resolutions. This is critical for automation — +prose recommendations require manual interpretation. + +``` +{"action": "merge", "file": "src/auth.py.md", "symbol": "authenticate_user", "keep": "- silently returns None on expired tokens...", "remove": "- returns None when token expires...", "merged": "- authenticate_user() silently returns None on expired tokens instead of raising. 3 of 7 callers don't check.\n _(verified, source: exploration)_", "reason": "duplicate: same claim, different wording"} +{"action": "merge", "file": "src/db.py.md", "symbol": "connect", "keep": "- connection pool exhaustion...", "remove": "- pool runs out...", "merged": "...", "reason": "near-duplicate: first extends second"} +{"action": "conflict", "file": "src/auth.py.md", "symbol": "validate_token", "entry_a": "- raises ValueError...", "entry_b": "- returns False...", "resolution": "verified_a", "reason": "code inspection: line 42 raises ValueError"} +{"action": "conflict", "file": "src/cache.py.md", "symbol": "invalidate", "entry_a": "...", "entry_b": "...", "resolution": "escalate", "reason": "both claims have evidence, needs user input"} +{"action": "move_to_cross", "file": "src/auth.py.md", "symbol": "validate_token", "entry": "- all validators share...", "cross_slug": "shared-validation-pattern", "reason": "cross-scope: involves 4 files"} +``` + +Fields: +- `action`: `merge` | `conflict` | `move_to_cross` | `move_from_cross` +- `file`: shadow file path relative to `.shadow/` +- `symbol`: the `##`/`###` heading the discovery lives under +- `keep`: the discovery text to keep (for merge) +- `remove`: the discovery text to delete (for merge) +- `merged`: the final merged text (for merge) +- `resolution`: `verified_a` | `verified_b` | `escalate` (for conflict) +- `reason`: human-readable explanation + +The orchestrator collects all lines, applies `merge` and `conflict` +actions automatically, and presents `escalate` items to the user. + +### Cross-Scope Scan + +After per-file scanning: + +1. Collect all per-file discoveries into a flat list +2. For each `_cross/*.md` discovery, check if any per-file discovery + makes the same or overlapping claim +3. For each `_prefs.md` preference, check if any per-file discovery + or `_cross/` entry duplicates it +4. Record cross-scope findings the same way + +### Scanning Guidelines + +- Compare claims semantically, not just textually. "Returns None on + expired tokens" and "Silently returns None when token expires" are + duplicates. +- Two discoveries about the same function but covering different + behaviors are **distinct**, not duplicates. E.g., "returns None on + expired tokens" vs "uses constant-time comparison" — these are + unrelated observations about the same function. +- Pay attention to `Also involves:` — two discoveries with overlapping + `Also involves:` refs are more likely related. + +## Phase 2: Resolve + +Process each finding by type. + +### Duplicates → Merge + +Combine into a single discovery: +- Keep the **richer** wording (more detail, more context) +- Keep the **stronger** trust: `source: user` > `source: interaction` > `source: exploration` +- Keep the **stronger** status: `verified` > `uncertain` > `refuted` +- Merge `Also involves:` refs (union of both) +- Delete the weaker entry + +Example: +``` +BEFORE (two entries under same symbol): +- authenticate_user() returns None on expired tokens. + _(verified, source: exploration)_ +- When the token is expired, authenticate_user silently returns None + instead of raising. 3 of 7 callers don't check. + _(verified, source: exploration)_ + +AFTER (merged): +- authenticate_user() silently returns None on expired tokens instead + of raising. 3 of 7 callers don't check the return value. + _(verified, source: exploration)_ +``` + +### Near-Duplicates → Absorb + +The broader discovery absorbs the narrower one: +- Expand the broader entry to include any extra detail from the narrower +- Delete the narrower entry +- Preserve the stronger trust/status between the two + +### Conflicts → Investigate + +When two discoveries contradict each other: + +1. Read the actual source code at the `file::symbol` location +2. Trace the logic to determine which claim is correct +3. If needed, write and run a short test script to verify +4. Mark the correct claim `verified`, the incorrect one `refuted` +5. If the incorrect claim was once true but code changed, update it + to reflect the current behavior and mark it `verified` + +If investigation takes more than a few minutes without resolution: +- Keep both discoveries +- Add a note: `(conflict unresolved — needs user input)` +- Ask the user to clarify at the end of the meditate session + +### Cross-Scope Duplicates → Place Correctly + +When the same discovery exists in both a per-file shadow and `_cross/`: + +- If it involves 3+ files → keep in `_cross/`, remove from per-file +- If it involves 1-2 files → keep in per-file, remove from `_cross/` +- Update cross-references in both directions after moving + +When a per-file discovery duplicates a `_prefs.md` entry: + +- If it's truly project-wide (not tied to a specific symbol) → keep + in `_prefs.md`, remove from per-file +- If it's specific to that symbol but happens to match a pref → keep + both (they serve different purposes) + +## Phase 3: Report + +After all resolutions, print a summary: + +``` +Meditate Summary +================ + Files scanned: 42 + Duplicates merged: 7 + Near-dupes absorbed: 3 + Conflicts resolved: 2 + Conflicts escalated: 1 + Cross-scope fixed: 2 + Total entries removed: 12 + +Escalated (needs your input): + src/auth.py::validate_token + - "raises ValueError on invalid format" vs "returns False on invalid format" + Both claims have evidence. Which behavior is correct? +``` + +## Parallelism + +For large shadows (20+ files), use parallel subagents: + +1. Partition shadow files into batches of ~10 files each +2. Launch one subagent per batch for Phase 1 (scan) +3. Each subagent outputs JSON-per-line findings (see Scan Output Format) +4. Orchestrator collects all JSON lines from all subagents +5. Auto-apply `merge` actions: use `edit` tool with `remove` as `old_str`, + replace `keep` with `merged` +6. Auto-apply `conflict` actions where `resolution` is `verified_a` or + `verified_b` — mark the loser `refuted` +7. Collect `escalate` items for user review +8. Run Phase 3 (report) + +For smaller shadows, run everything in a single pass — the scan output +format is still useful for traceability. + +## Dream Archive Hygiene (`_dreams/`) + +Meditate performs consistency checks and field repair on `_dreams/`. It +does NOT delete or rewrite report prose (those are historical records), +but it DOES fix missing/wrong metadata in the index. + +### Structural Checks + +1. **Index consistency** — verify every folder in `_dreams/` has a row in + `_dreams/_index.md`, and every row in the index has a matching folder. + Fix mismatches (add missing rows, remove orphaned rows). + +2. **Report completeness** — each `_dreams/<id>/` should contain at minimum + a `report.md`. Flag any empty directories. + +3. **Stale patches** — if `base_commit` in a report's frontmatter is more + than 100 commits behind current HEAD, add a note to the index: + `⚠️ patch may not apply cleanly`. Check with: + ```bash + git rev-list <base_commit>..HEAD --count 2>/dev/null + ``` + +4. **Cross-reference integrity** — if a per-file discovery has a + `Dream report: _dreams/<id>/` reference, verify that dream folder + exists. Remove dangling references. + +### Index Field Repair (run after Structural Checks) + +Dream subagents sometimes write incomplete index rows (e.g., `unknown` +category/verdict, generic titles). `meditate-repair.py` (below) +auto-resolves these by reading each experiment's `report.md` frontmatter, +`manifest.json`, and verdict-section signals. + +**Do NOT hand-edit 50+ rows.** Use the script. The agent's job is to +surface what the script CAN'T auto-fix: + +- **Corrupted reports** — when `report.md`'s `dream_id` (frontmatter or + body) doesn't match the folder name, the report was copy-pasted from + another experiment. The script prints these to stderr and skips them. + Do NOT auto-fix corruption — the content is wrong, not just the ID. + Log them in the meditate summary for user review. +- **Ambiguous parent branches** — when no `manifest.json` exists for an + experiment, the script falls back to slug heuristics (`-extend`, + `-fix`, `-deeper`, `-improve`, `-integration`, `-cleanup`, `-metrics`, + `-remaining` strongly suggest compounding from a sibling). If the + heuristic match is not high-confidence, flag for user review rather + than guessing. + +### Applying Index Repairs + +```bash +SKILL_DIR="" +for DIR in .github/skills/shadow-frog-meditate \ + .claude/skills/shadow-frog-meditate; do + [ -d "$DIR" ] && SKILL_DIR="$DIR" && break +done + +if [ -n "$SKILL_DIR" ] && [ -x "$SKILL_DIR/meditate-repair.py" ]; then + python3 "$SKILL_DIR/meditate-repair.py" +else + echo "meditate-repair.py not found; falling back to manual scan." >&2 +fi +``` + +What it does: +- Backs up `.shadow/_dreams/_index.md` to `.bak` first +- Detects corrupted reports (frontmatter `dream_id` != folder name) and + prints them to stderr — these rows are skipped, you resolve manually +- For every other row with `unknown`/empty category/verdict/title, + resolves the canonical value from the report's frontmatter, manifest, + or verdict section signals +- Title repair replaces generic forms (raw slug, "Dream Report: <slug>", + "Dream t##: <slug>") with the first `# H1` or `## Summary` line + +Verdict detection order is `manifest > VERDICT_SECTION signals > +whole-body signals`. Dead-end signals are checked BEFORE useful signals +so `not useful` doesn't match `useful`. + +Idempotent — safe to rerun until output reports `0 repaired`. + +Do NOT delete dream reports during meditate — only the user decides +what to keep or discard (via Phase 7 review or manual cleanup). + +## Rules + +- **Never delete a `source: user` discovery** without asking — user + knowledge is the highest trust. If it conflicts with `source: + exploration`, investigate thoroughly before concluding the user was wrong. +- **Preserve `Also involves:` refs** — when merging, take the union. +- **Update `_index.md`** after removing entries (discovery counts change). +- **Update `_meta/state.json`** — set `last_update_type: "meditate"`. +- **Don't touch `_prefs.md` placement** unless a pref is clearly + duplicated verbatim in a per-file shadow. + +## Format Compliance + +**Every merged or rewritten discovery must exactly follow the canonical +format in `/shadow-frog`** (Discovery Format, Cross-Cutting, Preferences). +Re-read both the original entries and the spec before writing — a +malformed discovery is worse than a duplicate; it breaks the viewer +parser and downstream agents. + +Meditate-specific rules: +- When merging, take the **union** of `labels: [...]` from both entries. +- When merging, preserve every `Also involves: file::symbol` from both + entries (union, not intersection). +- `Dream report: _dreams/<id>/` references must survive the merge — + re-attach to the merged entry if either original had one. +- Status (`verified`/`uncertain`/`refuted`) is taken from the stronger + source: `user` ≻ `interaction` ≻ `verified` exploration ≻ `uncertain`. +- Discovery text stays **behavioral**, not a code summary — preserve the + more behavioral wording when entries differ in style. diff --git a/skills/shadow-frog-meditate/meditate-repair.py b/skills/shadow-frog-meditate/meditate-repair.py new file mode 100755 index 0000000..3b69c7f --- /dev/null +++ b/skills/shadow-frog-meditate/meditate-repair.py @@ -0,0 +1,371 @@ +#!/usr/bin/env python3 +"""Repair _dreams/_index.md rows from their own report.md + manifest.json. + +Used by shadow-frog-meditate when the index has been corrupted or has +'unknown' category/verdict/title cells. Safe to rerun — only rows with +the targeted defects are touched. + +Usage: + python3 meditate-repair.py [--shadow-dir DIR] + +Defaults to ./.shadow/. Writes a .bak backup of the index before +modifying. + +Logic: +1. Detect corrupted reports (frontmatter dream_id doesn't match folder + name). These rows are flagged in stderr and skipped — the user must + resolve manually. +2. For each non-corrupted row with 'unknown' or empty + category/verdict/title, look up the canonical value from the report's + frontmatter, manifest's verdict field, or verdict section signals. +3. Verdict detection order: manifest > VERDICT_SECTION signals > + whole-body signals. Dead-end signals are checked BEFORE useful + signals so 'not useful' doesn't match 'useful'. +4. Title repair: replace generic titles (raw slug, "Dream Report: <slug>", + "Dream t##: <slug>") with the first H1 or first ## Summary line. +5. Parent linkage from manifest: for rows where parent is 'main', check + manifest.json and report.md frontmatter for a different parent_branch. + Validate the parent exists in the index (match by slug if timestamps + differ). +6. Parent linkage from slug heuristics: for rows still parented to 'main' + with no manifest info, infer from compounding suffixes (-extend, -fix, + -deeper, -improve, -integration, -cleanup, -metrics, -remaining). +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import sys + + +DREAM_ID_RE = re.compile(r'^(\d{8}-\d{6}Z)-(.+)$') +COMPOUNDING_SUFFIXES = re.compile( + r'-(extend|fix|deeper|improve|integration|cleanup|metrics|remaining)$' +) + +DEAD_SIGNALS = re.compile( + r'\bdead.end\b|\bdead_end\b|no improvement|\bnot useful\b' + r'|\u274c|\bfailed to\b|\binconclusive\b', + re.I, +) +USEFUL_SIGNALS = re.compile( + r'\buseful\b|verdict:\s*useful|\btests pass\b|\ball pass\b' + r'|\bpassed in\b|\b0 failed\b|\u2705|\d+ (tests )?passed' + r'|\bconfirmed\b|\bverified\b', + re.I, +) +GENERIC_TITLE = re.compile( + r'^(Dream Report:\s*|Dream t\d+:\s*)?[a-z0-9-]+$', + re.I, +) +VERDICT_SECTION = re.compile(r'## Verdict[^\n]*\n(.*?)(?=\n## |\Z)', re.S) + + +def detect_corrupted(dreams_dir: str) -> set[str]: + """Find dream folders whose report.md frontmatter ID doesn't match.""" + corrupted: set[str] = set() + for d in os.listdir(dreams_dir): + dpath = os.path.join(dreams_dir, d) + if not os.path.isdir(dpath) or d.startswith('_'): + continue + report = os.path.join(dpath, 'report.md') + if not os.path.exists(report): + continue + with open(report) as rf: + rcontent = rf.read() + fm_match = re.search(r'dream_id:\s*(\S+)', rcontent) + body_match = re.search(r'\*\*Dream ID\*\*:\s*(\S+)', rcontent) + found_id: str | None = None + if fm_match: + found_id = fm_match.group(1).strip().strip('"').strip("'") + elif body_match: + found_id = body_match.group(1).strip().strip('"').strip("'") + if found_id and found_id != d: + corrupted.add(d) + print( + f'WARNING CORRUPTED: {d} contains report from {found_id}', + file=sys.stderr, + ) + return corrupted + + +def lookup_verdict(dream_dir: str, content: str) -> str: + """Resolve verdict via manifest, verdict section, then whole-body scan.""" + manifest = os.path.join(dream_dir, 'manifest.json') + if os.path.exists(manifest): + try: + with open(manifest) as mf: + mdata = json.load(mf) + v = (mdata.get('verdict') or '').lower().strip() + if v and v != 'unknown': + return v + except (OSError, json.JSONDecodeError): + pass + vsec = VERDICT_SECTION.search(content) + scan_text = vsec.group(1) if vsec else content + if DEAD_SIGNALS.search(scan_text): + return 'dead_end' + if USEFUL_SIGNALS.search(scan_text): + return 'useful' + return '' + + +def repair_row(parts: list[str], dreams_dir: str) -> tuple[list[str], bool] | None: + """Repair a single index row in-place. Returns (updated parts, changed) or None.""" + did = parts[1] + dream_dir = os.path.join(dreams_dir, did) + report = os.path.join(dream_dir, 'report.md') + if not os.path.exists(report): + return None + with open(report) as rf: + content = rf.read() + + original = list(parts) + + cat_match = re.search(r'\*\*Category\*\*:\s*(.+)', content) + if not cat_match: + # Category is stored in YAML frontmatter (`category: <value>`), + # not as a `**Category**:` body marker. + cat_match = re.search(r'^category:\s*(.+)', content, re.M) + if cat_match and parts[2].strip().lower() in ('unknown', ''): + parts[2] = cat_match.group(1).strip().strip('"\'').lower() + parts[2] = re.sub(r'\s*\(.*\)\s*$', '', parts[2]) + parts[2] = parts[2].lower().strip() + + if parts[3].strip().lower() in ('unknown', ''): + v = lookup_verdict(dream_dir, content) + if v: + parts[3] = v + + if GENERIC_TITLE.match(parts[4].strip()): + heading = re.search(r'^#\s+(.+)', content, re.M) + if heading and not GENERIC_TITLE.match(heading.group(1).strip()): + parts[4] = heading.group(1).strip()[:80] + else: + summary = re.search(r'## Summary\s*\n+(.+)', content) + if summary: + parts[4] = summary.group(1).strip()[:80] + + did_change = parts != original + return parts, did_change + + +def parse_dream_id(did: str) -> tuple[str, str] | None: + """Split dream_id into (timestamp, slug) or None if malformed.""" + m = DREAM_ID_RE.match(did) + if m: + return m.group(1), m.group(2) + return None + + +def resolve_parent_in_index( + parent_id: str, all_dream_ids: list[str] +) -> str | None: + """Resolve a parent dream_id against the index, matching by slug if needed.""" + if parent_id in all_dream_ids: + return parent_id + parsed = parse_dream_id(parent_id) + if not parsed: + return None + _, parent_slug = parsed + for did in all_dream_ids: + p = parse_dream_id(did) + if p and p[1] == parent_slug: + return did + return None + + +def repair_parent( + parts: list[str], dreams_dir: str, all_dream_ids: list[str], + branch_by_dream_id: dict[str, str], +) -> bool: + """Repair the parent cell (index 6) if it is 'main' and better info exists. + + The parent column is canonically a BRANCH NAME (matching what the + reconciler writes from `manifest.parent_branch` and what dream-lineage.py + keys its graph on), NOT a dream_id. So once we resolve the parent to an + indexed row, we write that row's branch — not its dream_id. + + Returns True if the parent was changed. + """ + did = parts[1] + parent = parts[6].strip() + if parent != 'main': + return False + + dream_dir = os.path.join(dreams_dir, did) + resolved_did: str | None = None + + # Step 10: manifest.json lookup + manifest_path = os.path.join(dream_dir, 'manifest.json') + parent_branch_raw: str | None = None + if os.path.exists(manifest_path): + try: + with open(manifest_path) as mf: + mdata = json.load(mf) + pb = mdata.get('parent_branch', '').strip() + if pb and pb != 'main': + parent_branch_raw = pb + except (OSError, json.JSONDecodeError): + pass + + # Fallback: report.md frontmatter + if parent_branch_raw is None: + report_path = os.path.join(dream_dir, 'report.md') + if os.path.exists(report_path): + with open(report_path) as rf: + rcontent = rf.read() + fm_match = re.search(r'parent_branch:\s*["\']?([^"\'\n]+)', rcontent) + if fm_match: + pb = fm_match.group(1).strip() + if pb and pb != 'main': + parent_branch_raw = pb + + if parent_branch_raw: + # Extract dream_id from branch path (last /-separated segment) + candidate_id = parent_branch_raw.split('/')[-1] + resolved_did = resolve_parent_in_index(candidate_id, all_dream_ids) + + # Step 11: slug heuristic (only if step 10 didn't resolve anything) + if resolved_did is None: + parsed = parse_dream_id(did) + if not parsed: + return False + _, slug = parsed + suffix_match = COMPOUNDING_SUFFIXES.search(slug) + if not suffix_match: + return False + base_slug = slug[:suffix_match.start()] + + # Find candidates: dream_ids whose slug starts with base_slug + candidates: list[str] = [] + for other_did in all_dream_ids: + if other_did == did: + continue + other_parsed = parse_dream_id(other_did) + if not other_parsed: + continue + _, other_slug = other_parsed + # Candidate if other slug starts with base_slug (prefix match) + if other_slug.startswith(base_slug): + candidates.append(other_did) + + if not candidates: + return False + + # Sort by timestamp ascending, take earliest + candidates.sort() + + if len(candidates) == 1: + resolved_did = candidates[0] + else: + # Multiple candidates — check if they all share the exact base_slug + # (i.e., only differ by their own suffix). If so, take earliest. + exact_matches = [ + c for c in candidates + if parse_dream_id(c) and parse_dream_id(c)[1] == base_slug + ] + if len(exact_matches) == 1: + resolved_did = exact_matches[0] + else: + # Ambiguous + print( + f'WARNING AMBIGUOUS: {did} could compound any of: ' + + ', '.join(candidates), + file=sys.stderr, + ) + return False + + if resolved_did is None: + return False + + # Translate the resolved parent dream_id to its branch name (the canonical + # parent-column form). The index row's branch is authoritative — it is + # what dream-lineage.py looks up — even when the manifest's parent_branch + # carried a slightly different timestamp. + parent_branch = branch_by_dream_id.get(resolved_did) + if not parent_branch: + return False + parts[6] = parent_branch + print( + f'INFO PARENT: {did} parent main -> {parent_branch}', + file=sys.stderr, + ) + return True + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument('--shadow-dir', default='.shadow', + help='Shadow root (default: .shadow)') + args = ap.parse_args() + + dreams_dir = os.path.join(args.shadow_dir, '_dreams') + index_path = os.path.join(dreams_dir, '_index.md') + if not os.path.exists(index_path): + print(f'ERROR: {index_path} not found', file=sys.stderr) + return 1 + + with open(index_path) as f: + lines = f.readlines() + + backup = index_path + '.bak' + with open(backup, 'w') as bf: + bf.writelines(lines) + + corrupted = detect_corrupted(dreams_dir) + repaired = 0 + parent_repairs = 0 + + # Build list of all dream_ids + a dream_id->branch map for parent + # resolution. The parent column is written as a branch name, so we + # translate a resolved parent dream_id back to its row's branch. + all_dream_ids: list[str] = [] + branch_by_dream_id: dict[str, str] = {} + for line in lines: + parts = [p.strip() for p in line.split('|')] + if len(parts) < 8: + continue + did = parts[1] + if not did or did.startswith('-') or did == 'dream_id': + continue + all_dream_ids.append(did) + branch_by_dream_id[did] = parts[5] + + for i, line in enumerate(lines): + parts = [p.strip() for p in line.split('|')] + if len(parts) < 8: + continue + did = parts[1] + if not did or did.startswith('-') or did == 'dream_id': + continue + if did in corrupted: + continue + result = repair_row(parts, dreams_dir) + if result is None: + continue + new_parts, changed = result + parent_changed = repair_parent( + new_parts, dreams_dir, all_dream_ids, branch_by_dream_id + ) + if parent_changed: + parent_repairs += 1 + if changed or parent_changed: + lines[i] = '| ' + ' | '.join(new_parts[1:-1]) + ' |\n' + repaired += 1 + + with open(index_path, 'w') as f: + f.writelines(lines) + + print( + f'Repaired {repaired} rows ({parent_repairs} parent links); ' + f'backup at {backup}; {len(corrupted)} corrupted dream(s) skipped.' + ) + return 0 + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/skills/shadow-frog-update/SKILL.md b/skills/shadow-frog-update/SKILL.md new file mode 100644 index 0000000..a9764f3 --- /dev/null +++ b/skills/shadow-frog-update/SKILL.md @@ -0,0 +1,201 @@ +--- +name: shadow-frog-update +description: >- + Update the shadow knowledge base after code changes and from conversational + insights. Detects what changed via git diff, refreshes per-file shadows, + captures knowledge shared by the user during the session, and updates + cross-cutting discoveries. Invoke manually with /shadow-frog-update; the + preToolUse hook will remind the agent when the shadow is behind HEAD. +--- + +# ShadowFrog Update + +Updates `.shadow/` from two sources: code changes (git diff) and conversational +knowledge (what the user said during the session). Prerequisite: `.shadow/` exists. + +## Triggers + +1. Hook reminder: `preToolUse` injects a staleness warning when + `.shadow/_meta/state.json#last_commit` differs from HEAD. The hook + only reminds — it does NOT auto-run update. +2. Manual: user invokes `/shadow-frog-update` + +## Phase 1: Detect Changes + +```bash +# Read last_commit defensively: it may be missing, or the literal "none" +# when init ran without git (e.g. inside a container). `git diff none HEAD` +# would abort with "fatal: bad revision 'none'", and a missing key would +# make $LAST_COMMIT empty so `git diff HEAD` silently reports the wrong set. +LAST_COMMIT=$(python3 -c "import json,sys; print(json.load(sys.stdin).get('last_commit','none'))" < .shadow/_meta/state.json 2>/dev/null || echo "none") +if git rev-parse --verify "$LAST_COMMIT" >/dev/null 2>&1; then + git diff --name-only "$LAST_COMMIT" HEAD # committed changes since last update +else + echo "WARNING: state.json has no usable last_commit — falling back to a full re-scan." +fi +git diff --name-only HEAD # uncommitted changes +git diff --name-only --cached # staged changes +``` + +Categorize: modified, added, deleted, renamed. + +## Phase 2: Update Per-File Shadows (Symbol-Level) + +For each changed file, update its shadow at the symbol level: + +- **Added symbols** → add new `##` section +- **Removed symbols** → mark section as `REMOVED`, keep discoveries for history +- **Renamed symbols** → update heading, preserve discoveries +- **Modified symbols** → check if discoveries still hold + +Lightweight update (auto/hook): re-extract symbols, update headings, flag stale. +Deep update (manual/dream): read diffs, generate new discoveries, verify existing ones. + +## Phase 3: Capture Conversational Knowledge + +When the user shares knowledge during the session, write it immediately. +Do not batch for later. + +Signals to capture: + +| Signal | Example | Category | +|--------|---------|----------| +| Warning | "Don't change the retry logic, it's subtle" | warning | +| Design intent | "We use this pattern because the API is unreliable" | intent | +| History | "We tried caching here but it caused stale reads" | history | +| Gotcha | "This looks wrong but matches the tax authority spec" | warning | +| Deprecation | "This module is being replaced by v2/" | intent | +| Contract | "The 30s timeout matches our SLA" | contract | +| Convention | "Always use the helper in utils.py, not raw SQL" | convention | + +Write as: +```markdown +- <user's words, as close to verbatim as possible> + _(verified, source: user)_ +``` + +For knowledge emerging from collaborative work (debugging, refactoring, test failures): +```markdown +- <what was discovered and how> + _(verified, source: interaction)_ +``` + +Anchor to the specific `file::symbol`. `source: user` and `source: interaction` +are always `verified`. + +### Auto-Placement + +Users will not specify where to store their knowledge. You must find the +correct location. Procedure: + +1. Parse the user's statement for code references — file names, function names, + class names, module names, variable names, error messages, CLI flags. +2. If the knowledge is a **project-wide preference or convention** with no code + references (e.g., "always use snake_case", "no backward compatibility", + "prefer small PRs") → write to `_prefs.md`. +3. If explicit references found → look up those `file::symbol` paths in + `_index.md` and the corresponding shadow files. +4. If no explicit references → use context: + - What file is the user currently viewing or editing? + - What files were recently modified in this session? + - Search shadow files: `grep -rl "<keyword>" .shadow/ --include="*.md"` +5. If multiple candidate locations → pick the most specific symbol that the + knowledge applies to. Prefer a single `file::symbol` over file-level. +6. If the knowledge spans 3+ files → create a `_cross/<slug>.md` + entry and add back-pointers to each involved file's `## Cross-References`. + For 2-file discoveries, use per-file entries with `Also involves:` instead. +7. If no matching location exists (e.g., the user mentions a concept not yet in + the shadow) → place at the file-level `## File-Level` section of the most + relevant file, or create a new `_cross/` entry for repo-wide knowledge. + +Never ask the user "where should I put this?" — always resolve placement yourself. + +## Phase 4: Extract Session Insights + +At session end or manual trigger, review the session for: +1. Files modified and why +2. Patterns revealed by the changes +3. Unrecorded conversational knowledge (user statements not yet shadowed) +4. Cross-cutting discoveries (create in `_cross/<slug>.md` if 3+ files involved) + +## Phase 5: Handle Structural Changes + +Added files: +1. Check `.shadow/.shadowignore` — skip if the file matches an ignore pattern +2. Create `.shadow/<path>/<file>.md` with symbol-organized template +3. Add `## Cross-References` section +4. Add to `_index.md` + +Deleted files: +1. Add `ORPHANED` marker to shadow header +2. Keep shadow (discoveries explain history) +3. Mark `[REMOVED]` on any `_cross/` refs pointing to this file +4. Update `_index.md` + +Renamed files: +1. Move `.shadow/<old>.md` to `.shadow/<new>.md` +2. Update all `_cross/` `**Refs**:` entries (old path → new path) +3. Update all `Also involves:` in other per-file shadows +4. Update `_index.md` +5. Preserve all discoveries + +## Phase 6: Verify and Dedup + +Follow the dedup and writing rules in `/shadow-frog` — read before +write, merge or update existing entries, fix bad format in place. + +**Verify exploration discoveries** using the observe-based or do-based +methods in `/shadow-frog` § Verification. `source: user` and +`source: interaction` → always `verified`; only re-verify if the +underlying code changes. + +## Phase 7: Verify Reference Integrity + +Check the five core invariants (full 7-invariant set in `/shadow-frog`): +- Every `_cross/<slug>.md` ref has a back-pointer in per-file `## Cross-References` +- Every `## Cross-References` entry has a corresponding `_cross/<slug>.md` +- No duplicate cross-cutting filenames +- All `Also involves:` use `file::symbol` notation +- No duplicate discoveries (same behavioral claim at same symbol) + +Repair any violations before proceeding. + +## Phase 8: Update Metadata + +Preserve `dream_cycles_completed` from the existing state — only `dream-reconcile.py` increments it. + +```json +{ + "version": 1, + "initialized_at": "<preserved>", + "last_update_at": "<now ISO>", + "last_commit": "<full 40-char HEAD SHA>", + "last_update_type": "init|auto|manual|dream|meditate", + "total_files": N, + "total_symbols": N, + "total_discoveries": N, + "dream_cycles_completed": <preserved> +} +``` + +Refresh `_index.md` with current counts. + +## Discovery Writing Rules + +See `/shadow-frog` § Discovery Format for the verbatim per-file, +cross-cutting, and preference formats. Rules to keep in mind during +update sessions: + +- Be behavioral: "silently returns None on expired tokens" not "handles token expiration" +- `source: user` and `source: interaction` → always `verified`, use user's own words +- `source: exploration` → mark `uncertain` unless verified by code reading or tests +- If 3+ files involved → create in `_cross/<slug>.md` instead, add back-pointers +- If project-wide preference with no file reference → write to `_prefs.md` +- Slug naming: kebab-case derived from title (e.g., "Token expiry config split" → `token-expiry-config-split.md`) + +## Staleness Rules + +- Symbol modified → check if discovery still holds +- Symbol renamed → move discoveries to new heading +- Symbol removed → mark section `REMOVED`, keep discoveries +- `source: user` discoveries → only mark stale if symbol completely removed diff --git a/skills/shadow-frog-viewer/SKILL.md b/skills/shadow-frog-viewer/SKILL.md new file mode 100644 index 0000000..478f653 --- /dev/null +++ b/skills/shadow-frog-viewer/SKILL.md @@ -0,0 +1,162 @@ +--- +name: shadow-frog-viewer +description: >- + Browse and query the shadow knowledge base: overview, search for files + or symbols or text, view preferences, or see recent discoveries. + Invoke when the user wants to see what's in the shadow, get an + overview, or find specific knowledge. +scripts: + - shadow-viewer.py + - dream-lineage.py +--- + +# ShadowFrog Viewer + +Query and browse `.shadow/` content. Prerequisite: `.shadow/` exists. + +## Primary: Python Helper Script + +The companion script `shadow-viewer.py` lives in the same directory as +this SKILL.md file. To find and run it: + +```bash +# Project install (Copilot CLI): +python3 .github/skills/shadow-frog-viewer/shadow-viewer.py [options] +# Or for Claude Code: +python3 .claude/skills/shadow-frog-viewer/shadow-viewer.py [options] +``` + +### Available Views + +| Command | What it shows | +|---------|--------------| +| `--summary` | Overview: counts, source/status/label breakdown, per-file table, cross-cutting titles (default) | +| `--search QUERY` | Universal search — matches file names, symbol names, and discovery text. Includes cross-cutting and preferences | +| `--prefs` | Project-wide preferences | +| `--recent [N]` | N most recent discoveries with full content (default: 10) | +| `--labels LABEL` | Discoveries filtered by label (e.g., `bug`, `security`, `bug,performance`) | +| `--top FILE` | Top actionable discoveries for FILE — concise output (default: 3 entries, ~600 chars) suitable for the preToolUse hook. Includes both per-file shadow entries and any `_cross/` discoveries that reference FILE. Verified discoveries rank first. | +| `--check-invariants` | Audit structural integrity — bidirectional cross-references, label/source/category enum compliance, heading format, no-orphan-back-pointer. Exits 0 if clean, 1 with one violation per line. Run after dream reconciliation or before commit. | + +No arguments defaults to `--summary`. + +### Options + +| Flag | Effect | +|------|--------| +| `--shadow-dir DIR` | Override .shadow/ location (default: auto-detect from CWD) | +| `--top-labels LABELS` | Comma-separated label filter for `--top` (default: `bug,security`). Empty string disables label filtering. | +| `--top-limit N` | Max discoveries to show in `--top` (default: 3) | +| `--top-max-chars N` | Hard cap on `--top` total output length (default: 600). Use 0 for no cap. | + +### Examples + +```bash +# Quick overview +python3 shadow-viewer.py + +# Everything about auth (files, symbols, text, cross-cutting) +python3 shadow-viewer.py --search auth + +# Find token-related knowledge +python3 shadow-viewer.py --search "token expiry" + +# 5 most recent discoveries +python3 shadow-viewer.py --recent 5 + +# All known bugs +python3 shadow-viewer.py --labels bug + +# Security and performance issues +python3 shadow-viewer.py --labels security,performance + +# Top actionable discoveries for a single file (used by the preToolUse hook) +python3 shadow-viewer.py --top src/auth.py + +# Same, but broaden label filter and show up to 5 entries +python3 shadow-viewer.py --top src/auth.py --top-labels bug,security,performance --top-limit 5 + +# Team preferences +python3 shadow-viewer.py --prefs + +# Structural audit (run before commits or after dream reconciliation) +python3 shadow-viewer.py --check-invariants +``` + +## Dream Lineage Visualization + +The companion script `dream-lineage.py` generates an interactive HTML +visualization of the dream experiment tree. It reads `_dreams/_index.md` +and experiment reports to produce a self-contained HTML file. + +```bash +# (Claude Code users: replace .github/skills with .claude/skills below) + +# Generate dream-lineage.html in the current directory +python3 .github/skills/shadow-frog-viewer/dream-lineage.py + +# Custom output path +python3 .github/skills/shadow-frog-viewer/dream-lineage.py -o my-lineage.html + +# Explicit shadow directory +python3 .github/skills/shadow-frog-viewer/dream-lineage.py --shadow-dir /path/to/.shadow +``` + +The HTML file has three tabs: +- **🌳 Chains** — compounding chains as tree cards, sorted by depth +- **📋 Fresh** — non-compounding experiments grouped by category +- **🗂️ Full Tree** — compact view of the entire lineage in one tree + +Each node shows the experiment's category icon, name, verdict, test count, +and discovery count. Click "▶ Show report" to expand the full experiment +report inline. + +## Fallback: Shell One-Liners + +If the Python script fails to execute (wrong Python version, missing +file, permission error, etc.), fall back to these shell commands: + +### Summary + +Note: shell discovery counts are approximate (may include cross-reference links). + +```bash +echo "Files: $(find .shadow -name '*.md' -not -path '*/_cross/*' -not -path '*/_meta/*' -not -path '*/_dreams/*' -not -name '_index.md' -not -name '_prefs.md' | wc -l | tr -d ' ')" +echo "Discoveries: $(find .shadow -name '*.md' -not -path '*/_cross/*' -not -path '*/_meta/*' -not -path '*/_dreams/*' -not -name '_index.md' -not -name '_prefs.md' -exec grep -c '^\- ' {} \; 2>/dev/null | awk '{s+=$1} END {print s+0}')" +echo "Cross-cutting: $(ls .shadow/_cross/*.md 2>/dev/null | wc -l | tr -d ' ')" +python3 -c "import json; d=json.load(open('.shadow/_meta/state.json')); print(f'Last update: {d[\"last_update_at\"]} ({d[\"last_update_type\"]})')" +``` + +### Search + +```bash +grep -rn "QUERY" .shadow/ --include="*.md" +``` + +### Preferences + +```bash +cat .shadow/_prefs.md +``` + +### Recent + +```bash +# macOS +find .shadow -name '*.md' -not -path '*/_meta/*' -exec stat -f '%m %N' {} \; | sort -rn | head -10 | while read ts f; do echo "$(date -r "$ts" '+%Y-%m-%d %H:%M') $f"; done + +# Linux +find .shadow -name '*.md' -not -path '*/_meta/*' -printf '%T@ %p\n' | sort -rn | head -10 | while read ts f; do echo "$(date -d @"${ts%%.*}" '+%Y-%m-%d %H:%M') $f"; done +``` + +## Responding to the User + +After running a view, present the results clearly: +- For `--summary`: show the output directly, highlight anything notable +- For `--search`: summarize key findings, group by relevance +- For `--recent`: present the discoveries conversationally +- For `--top`: typically called by the preToolUse hook before a file is + edited; output is intentionally short and pre-formatted. If invoked + manually, present as-is. +- If the shadow is empty or has no discoveries, suggest running + `/shadow-frog-dream` to populate it diff --git a/skills/shadow-frog-viewer/dream-lineage.py b/skills/shadow-frog-viewer/dream-lineage.py new file mode 100644 index 0000000..e3f46a6 --- /dev/null +++ b/skills/shadow-frog-viewer/dream-lineage.py @@ -0,0 +1,737 @@ +#!/usr/bin/env python3 +"""Generate an interactive HTML visualization of dream experiment lineage. + +Reads .shadow/_dreams/_index.md and experiment reports to produce a +self-contained HTML file with: + - Stats dashboard (totals, categories, depth) + - Compounding chains tab (tree cards with expandable reports) + - Fresh experiments tab (grid grouped by category) + - Full tree tab (compact overview of entire lineage) + +Usage: + python3 dream-lineage.py # writes dream-lineage.html + python3 dream-lineage.py -o custom-name.html # custom output path + python3 dream-lineage.py --shadow-dir /path/to/.shadow +""" + +import argparse +import html as htmlmod +import json +import os +import re +import sys +from collections import defaultdict + + +# --------------------------------------------------------------------------- +# Argument parsing +# --------------------------------------------------------------------------- + +def parse_args(): + p = argparse.ArgumentParser(description="Dream lineage HTML visualizer") + p.add_argument("-o", "--output", default="dream-lineage.html", + help="Output HTML file path (default: dream-lineage.html)") + p.add_argument("--shadow-dir", default=None, + help="Path to .shadow/ directory (default: auto-detect)") + return p.parse_args() + + +# --------------------------------------------------------------------------- +# Data loading +# --------------------------------------------------------------------------- + +CAT_COLORS = { + "investigation": "#4CAF50", "bug hunting": "#F44336", + "feature design": "#2196F3", "refactoring": "#FF9800", + "optimization": "#9C27B0", "security audit": "#E91E63", + "unknown": "#607D8B", +} +VERDICT_MAP = { + "useful": "✅", "confirmed": "✅", "dead_end": "❌", "unknown": "—", +} + + +def find_shadow_dir(hint=None): + if hint and os.path.isdir(hint): + return hint + for candidate in [".shadow", os.path.join(os.getcwd(), ".shadow")]: + if os.path.isdir(candidate): + return candidate + print("ERROR: .shadow/ directory not found. Use --shadow-dir.", file=sys.stderr) + sys.exit(1) + + +def load_index(shadow_dir): + """Parse _dreams/_index.md into structured data.""" + index_path = os.path.join(shadow_dir, "_dreams", "_index.md") + if not os.path.exists(index_path): + print(f"ERROR: {index_path} not found.", file=sys.stderr) + sys.exit(1) + + children = defaultdict(list) + meta = {} + branch_by_slug = {} + + # First pass: collect all branches and build slug index + with open(index_path) as f: + for line in f: + parts = [p.strip() for p in line.split("|")] + if len(parts) < 8: + continue + did, cat, verdict, title, branch, parent, tip = parts[1:8] + if not did or did.startswith("-") or did == "dream_id": + continue + short = did.split("Z-")[-1] if "Z-" in did else did + meta[branch] = { + "short": short, "cat": cat, "verdict": verdict, + "title": title.strip(), "did": did, "tip": tip, + } + slug_match = re.search(r"t\d+-", branch) + if slug_match: + branch_by_slug[branch[slug_match.start():]] = branch + + # Second pass: resolve parent references (handle timestamp mismatches) + with open(index_path) as f: + for line in f: + parts = [p.strip() for p in line.split("|")] + if len(parts) < 8: + continue + did, cat, verdict, title, branch, parent, tip = parts[1:8] + if not did or did.startswith("-") or did == "dream_id": + continue + resolved = parent + if parent != "main" and parent not in meta: + m = re.search(r"t\d+-", parent) + if m and parent[m.start():] in branch_by_slug: + resolved = branch_by_slug[parent[m.start():]] + if branch not in children[resolved]: + children[resolved].append(branch) + + # Third pass: check manifest.json and report body for better parent info + dreams_dir = os.path.join(shadow_dir, "_dreams") + for branch in list(children.get("main", [])): + info = meta.get(branch, {}) + did = info.get("did", "") + if not did: + continue + mp = "" + # Try manifest.json first + manifest_path = os.path.join(dreams_dir, did, "manifest.json") + if os.path.exists(manifest_path): + try: + with open(manifest_path) as f: + mdata = json.load(f) + mp = mdata.get("parent_branch", "") + except Exception: + pass + # Try report body for parent references + if not mp or mp == "main": + report_path = os.path.join(dreams_dir, did, "report.md") + if os.path.exists(report_path): + try: + with open(report_path) as f: + head = f.read(2000) + # Check builds_on in frontmatter + m = re.search(r"builds_on:\s*\[?\s*[\"']?([^\]\"'\n,]+)", head) + if m: + mp = m.group(1).strip().strip("\"'") + except Exception: + pass + if not mp or mp == "main": + continue + # Resolve via slug matching + resolved = mp + if mp not in meta: + m = re.search(r"t\d+-", mp) + if m and mp[m.start():] in branch_by_slug: + resolved = branch_by_slug[mp[m.start():]] + else: + # Try matching just tNN prefix + m = re.search(r"(t\d+)", mp) + if m: + prefix = m.group(1) + "-" + for slug, br in branch_by_slug.items(): + if slug.startswith(prefix): + resolved = br + break + else: + continue + else: + continue + if resolved == branch: + continue + # Re-parent: remove from main, add under resolved parent + try: + children["main"].remove(branch) + if branch not in children[resolved]: + children[resolved].append(branch) + except ValueError: + pass + + return meta, children + + +def load_reports(shadow_dir, meta): + """Read report content and manifest metadata for each experiment.""" + for branch, info in meta.items(): + did = info["did"] + report_path = os.path.join(shadow_dir, "_dreams", did, "report.md") + info["full_report"] = "" + info["tests"] = "" + info["discoveries_count"] = 0 + + if os.path.exists(report_path): + try: + with open(report_path) as f: + content = f.read() + body = content + if content.startswith("---"): + fm_end = content.find("---", 3) + if fm_end > 0: + body = content[fm_end + 3:].strip() + info["full_report"] = body + except Exception: + pass + + manifest_path = os.path.join(shadow_dir, "_dreams", did, "manifest.json") + if os.path.exists(manifest_path): + try: + with open(manifest_path) as f: + mdata = json.load(f) + info["tests"] = str(mdata.get("tests_passed", mdata.get("test_count", ""))) + info["discoveries_count"] = len(mdata.get("discoveries", [])) + except Exception: + pass + + +# --------------------------------------------------------------------------- +# HTML generation helpers +# --------------------------------------------------------------------------- + +def md_to_html(md): + """Convert markdown to HTML, handling code blocks, lists, tables, etc.""" + # Extract fenced code blocks first (before escaping) + code_blocks = [] + def stash_code(m): + lang = m.group(1) or "" + code = htmlmod.escape(m.group(2)) + code_blocks.append(f'<pre><code>{code}</code></pre>') + return f"\x00CODE{len(code_blocks) - 1}\x00" + s = re.sub(r"```(\w*)\n(.*?)```", stash_code, md, flags=re.S) + + s = htmlmod.escape(s) + + # Headings + s = re.sub(r"^### (.+)$", r"<h4>\1</h4>", s, flags=re.M) + s = re.sub(r"^## (.+)$", r"<h3>\1</h3>", s, flags=re.M) + s = re.sub(r"^# (.+)$", r"<h2>\1</h2>", s, flags=re.M) + + # Bold (allow multiline) + s = re.sub(r"\*\*(.+?)\*\*", r"<strong>\1</strong>", s, flags=re.S) + + # Inline code + s = re.sub(r"`([^`\n]+)`", r"<code>\1</code>", s) + + # Blockquotes + s = re.sub(r"^> (.+)$", r"<blockquote>\1</blockquote>", s, flags=re.M) + + # Tables: detect header + separator + rows + def render_table(m): + lines = m.group(0).strip().split("\n") + headers = [c.strip() for c in lines[0].split("|") if c.strip()] + rows_html = "<tr>" + "".join(f"<th>{h}</th>" for h in headers) + "</tr>" + for row_line in lines[2:]: + cells = [c.strip() for c in row_line.split("|") if c.strip()] + rows_html += "<tr>" + "".join(f"<td>{c}</td>" for c in cells) + "</tr>" + return f"<table>{rows_html}</table>" + s = re.sub(r"^(\|.+\|)\n(\|[-| :]+\|)\n((?:\|.+\|\n?)+)", render_table, s, flags=re.M) + + # Numbered lists: consecutive lines starting with N. + def render_ol(m): + items = re.findall(r"^\d+\.\s+(.+)$", m.group(0), re.M) + return "<ol>" + "".join(f"<li>{it}</li>" for it in items) + "</ol>" + s = re.sub(r"(?:^\d+\.\s+.+$\n?){2,}", render_ol, s, flags=re.M) + # Single numbered item + s = re.sub(r"^(\d+)\.\s+(.+)$", r"<ol start='\1'><li>\2</li></ol>", s, flags=re.M) + + # Unordered lists (including nested via indentation). Consecutive + # `- item` lines are wrapped in a single <ul>; indented items get + # class='nested' so the panel-body CSS (li.nested margin-left: 32px) + # renders the nesting visually. Previous implementation emitted bare + # <li> tags with no <ul> wrapper, producing structurally invalid HTML. + def render_ul(m): + items_html = [] + for line in m.group(0).split('\n'): + ml = re.match(r"^( *)- (.+)$", line.rstrip()) + if not ml: + continue + cls = " class='nested'" if len(ml.group(1)) >= 2 else "" + items_html.append(f"<li{cls}>{ml.group(2)}</li>") + return "<ul>" + "".join(items_html) + "</ul>" + s = re.sub(r"(?:^ *- .+$\n?){1,}", render_ul, s, flags=re.M) + + # Paragraphs + s = re.sub(r"\n\n+", "</p><p>", s) + + # Restore code blocks + for i, block in enumerate(code_blocks): + s = s.replace(f"\x00CODE{i}\x00", block) + + return f"<p>{s}</p>" + + +def stable_id(branch): + """Generate a stable template ID from a branch name.""" + return "rpt-" + re.sub(r"[^a-zA-Z0-9]", "-", branch) + + +def flatten_chain(branch, meta, children, depth=0): + """Flatten a chain tree into an ordered list of (branch, depth) tuples.""" + result = [(branch, depth)] + for kid in children.get(branch, []): + result.extend(flatten_chain(kid, meta, children, depth + 1)) + return result + + +def node_html(branch, meta, children, with_report=True): + """Render a single node as a flat timeline row.""" + info = meta.get(branch, {}) + short = info.get("short", branch) + cat = info.get("cat", "unknown") + color = CAT_COLORS.get(cat, "#607D8B") + verdict = VERDICT_MAP.get(info.get("verdict", ""), "—") + title = htmlmod.escape(info.get("title", "")) + tests = info.get("tests", "") + disc = info.get("discoveries_count", 0) + full_report = info.get("full_report", "") + depth = info.get("_depth", 0) + + sid = stable_id(branch) + + test_badge = f'<span class="badge test">{tests} tests</span>' if tests else "" + disc_badge = f'<span class="badge disc">{disc} disc</span>' if disc else "" + + report_btn = "" + if with_report and full_report: + report_btn = ( + f' <button class="report-btn" ' + f'onclick="showPanel(\'{sid}\')">📄</button>' + ) + + return ( + f'<div class="tl-row" style="--node-color: {color}">' + f'<div class="tl-depth" style="background:{color}">{depth}</div>' + f'<div class="tl-content">' + f'<div class="node-header">' + f'<span class="name">{short}</span>' + f'<span class="verdict">{verdict}</span>' + f'{test_badge}{disc_badge}' + f'{report_btn}' + f'</div>' + f'<div class="title">{title}</div>' + f'</div>' + f'</div>' + ) + + +def compact_node(branch, meta, children, prefix="", is_last=True): + """Render a single line in the compact tree view.""" + info = meta.get(branch, {}) + short = info.get("short", branch) + cat = info.get("cat", "unknown") + color = CAT_COLORS.get(cat, "#607D8B") + verdict = VERDICT_MAP.get(info.get("verdict", ""), "—") + title = htmlmod.escape(info.get("title", "")) + tests = info.get("tests", "") + report = info.get("full_report", "") + + connector = "└── " if is_last else "├── " + test_info = f' <span class="ct-test">{tests}t</span>' if tests else "" + + sid = stable_id(branch) + report_btn = "" + if report: + report_btn = ( + f' <button class="report-btn ct-report-btn" ' + f'onclick="showPanel(\'{sid}\')">📄</button>' + ) + + line = ( + f'<div class="ct-line">' + f'<span class="ct-tree">{htmlmod.escape(prefix)}{connector}</span>' + f'<span class="ct-name" style="color:{color}">{short}</span>' + f'<span class="ct-verdict">{verdict}</span>' + f'{test_info}' + f'<span class="ct-title">{title}</span>' + f'{report_btn}' + f'</div>' + ) + + kids = children.get(branch, []) + for i, kid in enumerate(kids): + ext = " " if is_last else "│ " + line += compact_node(kid, meta, children, prefix + ext, i == len(kids) - 1) + return line + + +# --------------------------------------------------------------------------- +# Main +# --------------------------------------------------------------------------- + +def tree_depth(node, children, _seen=None): + """Depth of the subtree rooted at `node`. + + `_seen` guards against cycles in malformed indices (self-parent + or A↔B loops). On re-visit we treat the node as terminal and + return 0 — better to underestimate depth than to crash with + RecursionError on bad input. + """ + if _seen is None: + _seen = set() + if node in _seen: + return 0 + _seen = _seen | {node} + kids = children.get(node, []) + return (1 + max(tree_depth(k, children, _seen) for k in kids)) if kids else 0 + + +def generate_html(shadow_dir, output_path): + meta, children = load_index(shadow_dir) + load_reports(shadow_dir, meta) + + total = len(meta) + compound = sum(1 for p in children if p != "main" for _ in children[p]) + + # Separate chains vs fresh + chain_roots = [] + fresh_leaves = [] + for b in children.get("main", []): + d = tree_depth(b, children) + if d >= 1: + chain_roots.append({"branch": b, "depth": d}) + else: + fresh_leaves.append(b) + chain_roots.sort(key=lambda x: -x["depth"]) + + max_depth = max((r["depth"] for r in chain_roots), default=0) + 1 + + # Category counts + cat_counts = defaultdict(int) + for info in meta.values(): + cat_counts[info.get("cat", "unknown")] += 1 + + # Session count (unique date-hour groups) + sessions = set() + for info in meta.values(): + did = info.get("did", "") + m = re.match(r"(\d{8}-\d{4})", did) + if m: + sessions.add(m.group(1)) + + # --- Tab 1: Compounding Chains --- + chains_html = "" + for root_info in chain_roots: + branch = root_info["branch"] + depth = root_info["depth"] + flat = flatten_chain(branch, meta, children) + # Set depth on each node's meta for rendering + for b, d in flat: + if b in meta: + meta[b]["_depth"] = d + nodes_html = "".join(node_html(b, meta, children, with_report=True) + for b, _ in flat) + chains_html += ( + f'<div class="chain">' + f'<div class="chain-header">Depth {depth + 1} chain · {len(flat)} experiments</div>' + f'{nodes_html}' + f'</div>' + ) + + # --- Tab 2: Fresh Experiments --- + fresh_by_cat = defaultdict(list) + for b in fresh_leaves: + cat = meta.get(b, {}).get("cat", "unknown") + if b in meta: + meta[b]["_depth"] = 0 + fresh_by_cat[cat].append(b) + + fresh_html = "" + for cat in ["investigation", "bug hunting", "feature design", "refactoring", + "optimization", "security audit", "unknown"]: + branches = fresh_by_cat.get(cat, []) + if not branches: + continue + color = CAT_COLORS.get(cat, "#607D8B") + fresh_html += ( + f'<div class="fresh-group">' + f'<div class="fresh-header" style="color:{color}">{cat.title()} ({len(branches)})</div>' + f'<div class="fresh-grid">' + ) + for b in branches: + fresh_html += node_html(b, meta, children, with_report=True) + fresh_html += "</div></div>" + + # --- Tab 3: Full Tree (compact, sorted deepest-first) --- + tree_html = '<div class="compact-tree"><div class="ct-line"><span class="ct-root">🌳 main</span></div>' + main_kids = sorted( + children.get("main", []), + key=lambda b: tree_depth(b, children), + reverse=True, + ) + for i, b in enumerate(main_kids): + tree_html += compact_node(b, meta, children, "", i == len(main_kids) - 1) + tree_html += "</div>" + + # --- Category stats bar --- + cat_bar_html = "" + for cat in ["investigation", "bug hunting", "feature design", "refactoring", + "optimization", "security audit", "unknown"]: + cnt = cat_counts.get(cat, 0) + if cnt == 0: + continue + color = CAT_COLORS.get(cat, "#607D8B") + cat_bar_html += ( + f'<div class="cat-stat">' + f'<span class="legend-dot" style="background:{color}"></span>' + f'{cat.title()}: <strong>{cnt}</strong>' + f'</div>' + ) + + # Verdict legend + verdict_counts = {} + for info in meta.values(): + v = info.get("verdict", "unknown") + verdict_counts[v] = verdict_counts.get(v, 0) + 1 + verdict_bar_html = "" + useful_cnt = verdict_counts.get("useful", 0) + verdict_counts.get("confirmed", 0) + for symbol, label, cnt in [ + ("✅", "Useful", useful_cnt), + ("❌", "Dead End", verdict_counts.get("dead_end", 0)), + ("—", "Unknown", verdict_counts.get("unknown", 0) + verdict_counts.get("---", 0)), + ]: + if cnt == 0: + continue + verdict_bar_html += ( + f'<div class="cat-stat">' + f'{symbol} {label}: <strong>{cnt}</strong>' + f'</div>' + ) + + # --- Global report templates (one per experiment, shared by all tabs) --- + templates_html = "" + for branch, info in meta.items(): + report = info.get("full_report", "") + if not report: + continue + sid = stable_id(branch) + short = htmlmod.escape(info.get("short", branch)) + verdict = VERDICT_MAP.get(info.get("verdict", ""), "—") + title = htmlmod.escape(info.get("title", "")) + templates_html += ( + f'<template id="{sid}">' + f'<h2>{short}</h2>' + f'<div class="panel-meta">{verdict} {title}</div>' + f'{md_to_html(report)}' + f'</template>\n' + ) + + page = TEMPLATE.format( + total=total, compound=compound, fresh=total - compound, + sessions=len(sessions), chains=len(chain_roots), max_depth=max_depth, + cat_bar=cat_bar_html, verdict_bar=verdict_bar_html, + chains_html=chains_html, chain_count=len(chain_roots), + fresh_html=fresh_html, fresh_count=len(fresh_leaves), + tree_html=tree_html, tree_count=total, + templates=templates_html, + ) + + with open(output_path, "w") as f: + f.write(page) + print(f"Wrote {output_path} ({os.path.getsize(output_path):,} bytes)") + print(f" {total} experiments, {len(chain_roots)} chains (max depth {max_depth}), " + f"{compound} compounding, {total - compound} fresh") + + +# --------------------------------------------------------------------------- +# HTML template +# --------------------------------------------------------------------------- + +TEMPLATE = '''<!DOCTYPE html> +<html><head> +<meta charset="utf-8"> +<title>🐸 Dream Lineage + + +

        🐸 Dream Lineage

        +
        +
        {total}
        Experiments
        +
        {compound}
        Compounding
        +
        {fresh}
        Fresh
        +
        {sessions}
        Sessions
        +
        {chains}
        Chains
        +
        {max_depth}
        Max Depth
        +
        +
        {cat_bar}
        +
        {verdict_bar}
        +

        + Click 📄 on any node to view its full experiment report in the side panel. +

        + +
        +
        🌳 Chains ({chain_count})
        +
        📋 Fresh ({fresh_count})
        +
        🗂️ Full Tree ({tree_count})
        +
        + +
        +
        {chains_html}
        +
        + +
        + {fresh_html} +
        + +
        + {tree_html} +
        + + +{templates} + + +
        + +
        +
        + + +''' + + +if __name__ == "__main__": + args = parse_args() + shadow_dir = find_shadow_dir(args.shadow_dir) + generate_html(shadow_dir, args.output) diff --git a/skills/shadow-frog-viewer/shadow-viewer.py b/skills/shadow-frog-viewer/shadow-viewer.py new file mode 100755 index 0000000..7a5b3e4 --- /dev/null +++ b/skills/shadow-frog-viewer/shadow-viewer.py @@ -0,0 +1,1393 @@ +#!/usr/bin/env python3 +"""shadow-viewer: Query and browse a .shadow/ knowledge base. + +Usage: + shadow-viewer.py [options] + +Views: + --summary Overview + detailed statistics (default) + --search QUERY Universal search across files, symbols, and text + --prefs Show project-wide preferences + --recent [N] N most recent discoveries with content (default: 10) + --labels LABEL Show discoveries by label (bug, security, etc.) + --top FILE Top actionable discoveries for FILE (hook-sized) + --check-invariants Report structural violations (exit 1 if any) + +Options: + --shadow-dir DIR Path to .shadow/ directory (default: auto-detect) + +Exit codes: + 0 Success (possibly with warnings on stderr) + 1 Fatal error (shadow dir not found, no results possible) +""" + +import argparse +import json +import os +import re +import sys +import traceback +from collections import defaultdict +from datetime import datetime +from pathlib import Path + + +_DISCOVERY_META_RE = re.compile( + r"_\((\w+),\s*source:\s*(\w+)" + r"(?:,\s*labels:\s*\[([^\]]*)\])?" + r"\)_" +) + + +def warn(msg): + """Print a warning to stderr. Agents read these to adjust strategy.""" + print(f"[shadow-viewer warning] {msg}", file=sys.stderr) + + +def error(msg): + """Print an error to stderr.""" + print(f"[shadow-viewer error] {msg}", file=sys.stderr) + + +def find_shadow_dir(start="."): + """Walk up from start to find .shadow/ directory.""" + try: + p = Path(start).resolve() + while p != p.parent: + candidate = p / ".shadow" + if candidate.is_dir(): + return candidate + p = p.parent + except (OSError, PermissionError) as e: + error(f"Failed to search for .shadow/ directory from '{start}': {e}") + return None + + +def parse_discovery(line, continuation_lines=None): + """Parse a discovery bullet and its metadata line(s).""" + try: + text = line[2:].strip() if line.startswith("- ") else line.strip() + except (TypeError, AttributeError) as e: + warn(f"parse_discovery: bad line input ({type(line).__name__}): {e}") + return {"text": str(line) if line else ""} + + meta = {} + full_text = text + + if continuation_lines: + for cl in continuation_lines: + try: + stripped = cl.strip() + # _(status, source: type, labels: [l1, l2])_ or _(status, source: type)_ + m = _DISCOVERY_META_RE.match(stripped) + if m: + meta["status"] = m.group(1) + meta["source"] = m.group(2) + if m.group(3): + meta["labels"] = [ + l.strip() + for l in m.group(3).split(",") + if l.strip() + ] + else: + # _(source: type)_ (preferences format) + m2 = re.match(r"_\(source:\s*(\w+)\)_", stripped) + if m2: + meta["source"] = m2.group(1) + elif stripped.startswith("Also involves:"): + refs = re.findall(r"`([^`]+)`", stripped) + meta["also_involves"] = refs + elif stripped.startswith("Dream report:"): + m_dr = re.search(r"`([^`]+)`", stripped) + if m_dr: + meta["dream_report"] = m_dr.group(1) + else: + full_text += " " + stripped + except Exception as e: + warn(f"parse_discovery: failed parsing continuation line " + f"'{cl[:80]}': {e}") + + return {"text": full_text, **meta} + + +def parse_shadow_file(filepath): + """Parse a per-file shadow into structured data. + + Returns a result dict even on partial failure — whatever was parsed + before the error is preserved. Warnings go to stderr. + """ + result = { + "path": str(filepath), + "source_file": None, + "language": None, + "lines": None, + "last_modified": None, + "symbols": [], + "discoveries": [], + "cross_references": [], + "parse_errors": [], + } + + try: + content = filepath.read_text(encoding="utf-8") + except UnicodeDecodeError as e: + msg = (f"Cannot read {filepath}: encoding error at byte " + f"{e.start}: {e.reason}. File may not be UTF-8.") + warn(msg) + result["parse_errors"].append(msg) + return result + except OSError as e: + msg = f"Cannot read {filepath}: {e}" + warn(msg) + result["parse_errors"].append(msg) + return result + + lines = content.split("\n") + current_symbol = None + i = 0 + + while i < len(lines): + line = lines[i] + + try: + # Header: # Shadow: src/auth.py + if line.startswith("# Shadow: "): + result["source_file"] = line[len("# Shadow: "):].strip() + + # Metadata: **Language**: Python | **Lines**: 142 | ... + elif line.startswith("**Language**"): + parts = line.split("|") + for part in parts: + part = part.strip() + if part.startswith("**Language**"): + m = re.search(r"\*\*:\s*(.+)", part) + if m: + result["language"] = m.group(1).strip() + elif "Lines" in part: + m = re.search(r"(\d+)", part) + if m: + try: + result["lines"] = int(m.group(1)) + except ValueError: + pass + elif "Last modified" in part: + m = re.search(r"\*\*:\s*(.+)", part) + if m: + result["last_modified"] = m.group(1).strip() + + # Symbol heading: ## `symbol_name` or ### `Class.method` + elif re.match(r"^#{2,3}\s", line): + sym_match = re.match(r"^(#{2,3})\s+`(.+?)`", line) + if sym_match: + name = sym_match.group(2) + current_symbol = name + result["symbols"].append(name) + elif "Cross-References" in line: + current_symbol = "__cross_refs__" + elif "File-Level" in line: + current_symbol = "__file_level__" + else: + current_symbol = None + + # Discovery bullet (skip cross-reference links) + elif ( + line.strip().startswith("- ") + and current_symbol + and current_symbol != "__cross_refs__" + ): + # Collect continuation lines + continuation = [] + j = i + 1 + while j < len(lines): + next_line = lines[j] + if ( + next_line.strip() == "" + or next_line.strip().startswith("- ") + or re.match(r"^#{1,3}\s", next_line) + ): + break + continuation.append(next_line) + j += 1 + + disc = parse_discovery(line.strip(), continuation) + disc["symbol"] = ( + "file-level" if current_symbol == "__file_level__" + else current_symbol + ) + disc["file"] = result["source_file"] + result["discoveries"].append(disc) + i = j + continue + + # Cross-reference link + elif ( + current_symbol == "__cross_refs__" + and line.strip().startswith("- ") + ): + link_match = re.search(r"\[(.+?)\]", line) + if link_match: + result["cross_references"].append(link_match.group(1)) + + except Exception as e: + msg = (f"Error parsing {filepath} at line {i + 1}: " + f"{type(e).__name__}: {e}") + warn(msg) + result["parse_errors"].append(msg) + + i += 1 + + return result + + +def parse_prefs(shadow_dir): + """Parse _prefs.md into a list of preferences.""" + prefs_path = shadow_dir / "_prefs.md" + if not prefs_path.exists(): + return [] + + try: + content = prefs_path.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError) as e: + warn(f"Cannot read preferences file {prefs_path}: {e}") + return [] + + prefs = [] + lines = content.split("\n") + i = 0 + while i < len(lines): + line = lines[i] + try: + if line.strip().startswith("- ") and not line.strip().startswith( + "- [" + ): + continuation = [] + j = i + 1 + while j < len(lines): + next_line = lines[j] + if next_line.strip() == "" or next_line.strip().startswith( + "- " + ): + break + continuation.append(next_line) + j += 1 + + pref = parse_discovery(line.strip(), continuation) + pref["type"] = "preference" + prefs.append(pref) + i = j + continue + except Exception as e: + warn(f"Error parsing preference at line {i + 1} in " + f"{prefs_path}: {e}") + i += 1 + + return prefs + + +def parse_cross_cutting(shadow_dir): + """Parse all _cross/*.md files.""" + cross_dir = shadow_dir / "_cross" + if not cross_dir.exists(): + return [] + + entries = [] + try: + md_files = sorted(cross_dir.glob("*.md")) + except OSError as e: + warn(f"Cannot list cross-cutting directory {cross_dir}: {e}") + return [] + + for f in md_files: + try: + content = f.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError) as e: + warn(f"Cannot read cross-cutting file {f}: {e}") + continue + + entry = {"slug": f.stem, "file": str(f.name)} + + try: + # Title + m = re.search(r"^# (.+)", content, re.MULTILINE) + if m: + entry["title"] = m.group(1).strip() + + # Category + m = re.search(r"\*\*Category\*\*:\s*(.+)", content) + if m: + entry["category"] = m.group(1).strip() + + # Refs — only within the **Refs**: section, not backticked + # bullets elsewhere in the file (e.g., inside the Discovery body). + refs = [] + refs_block = re.search( + r"\*\*Refs\*\*:\s*\n(.*?)(?=\n[ \t]*\n|\n\*\*|\Z)", + content, + re.DOTALL, + ) + if refs_block: + refs = re.findall(r"-\s*`([^`]+)`", refs_block.group(1)) + entry["refs"] = refs + + # Discovery text + m = re.search( + r"\*\*Discovery\*\*:\s*(.+?)(?=\n\n|\n_\(|\Z)", + content, + re.DOTALL, + ) + if m: + entry["discovery"] = m.group(1).strip() + + # Status/source (with optional labels) + m = _DISCOVERY_META_RE.search(content) + if m: + entry["status"] = m.group(1) + entry["source"] = m.group(2) + if m.group(3): + entry["labels"] = [ + l.strip() + for l in m.group(3).split(",") + if l.strip() + ] + else: + # Fallback to simpler pattern + m2 = re.search( + r"_\((\w+),\s*source:\s*(\w+)\)_", content + ) + if m2: + entry["status"] = m2.group(1) + entry["source"] = m2.group(2) + except Exception as e: + warn(f"Error parsing cross-cutting file {f.name}: " + f"{type(e).__name__}: {e}") + entry.setdefault("title", f.stem) + + entries.append(entry) + + return entries + + +def load_state(shadow_dir): + """Load _meta/state.json.""" + state_path = shadow_dir / "_meta" / "state.json" + if not state_path.exists(): + return {} + try: + content = state_path.read_text(encoding="utf-8") + state = json.loads(content) + if not isinstance(state, dict): + warn(f"state.json is not a JSON object (got {type(state).__name__})") + return {} + return state + except json.JSONDecodeError as e: + warn(f"Invalid JSON in {state_path}: {e}") + return {} + except OSError as e: + warn(f"Cannot read {state_path}: {e}") + return {} + + +def get_all_shadow_files(shadow_dir): + """Get all per-file shadow .md files (excluding special files).""" + special = {"_index.md", "_prefs.md"} + special_dirs = {"_cross", "_meta", "_dreams"} + + results = [] + try: + for f in shadow_dir.rglob("*.md"): + try: + rel = f.relative_to(shadow_dir) + parts = rel.parts + if parts[0] in special_dirs: + continue + if str(rel) in special: + continue + results.append(f) + except (ValueError, IndexError) as e: + warn(f"Skipping file {f}: {e}") + except OSError as e: + warn(f"Error walking shadow directory {shadow_dir}: {e}") + + return sorted(results) + + +def collect_all_discoveries(shadow_dir): + """Parse all shadow files and collect every discovery. + + Continues past individual file failures — reports errors and moves on. + """ + all_disc = [] + failed_files = [] + for sf in get_all_shadow_files(shadow_dir): + try: + parsed = parse_shadow_file(sf) + if parsed.get("parse_errors"): + failed_files.append( + (str(sf), parsed["parse_errors"]) + ) + for d in parsed["discoveries"]: + d.setdefault("file", parsed["source_file"]) + d["shadow_path"] = str(sf.relative_to(shadow_dir)) + try: + d["shadow_mtime"] = os.path.getmtime(sf) + except OSError: + d["shadow_mtime"] = 0.0 + all_disc.append(d) + except Exception as e: + msg = f"Failed to parse {sf}: {type(e).__name__}: {e}" + warn(msg) + failed_files.append((str(sf), [msg])) + + if failed_files: + warn(f"{len(failed_files)} file(s) had parse errors " + f"(discoveries from other files still collected)") + + return all_disc + + +# --- View Functions --- + + +def view_summary(shadow_dir): + """Overview + detailed statistics. + + Each section is independently wrapped — if label stats fail, you + still get counts and the per-file table. + """ + # Load data (each can fail independently) + state = {} + shadow_files = [] + prefs = [] + cross = [] + + try: + state = load_state(shadow_dir) + except Exception as e: + warn(f"Failed to load state.json: {e}") + + try: + shadow_files = get_all_shadow_files(shadow_dir) + except Exception as e: + warn(f"Failed to list shadow files: {e}") + + try: + prefs = parse_prefs(shadow_dir) + except Exception as e: + warn(f"Failed to parse preferences: {e}") + + try: + cross = parse_cross_cutting(shadow_dir) + except Exception as e: + warn(f"Failed to parse cross-cutting discoveries: {e}") + + # Parse each file once for both stats and discoveries + file_stats = [] + all_disc = [] + total_symbols = 0 + for sf in shadow_files: + try: + parsed = parse_shadow_file(sf) + src = parsed["source_file"] or str(sf.relative_to(shadow_dir)) + n_sym = len(parsed["symbols"]) + n_disc = len(parsed["discoveries"]) + total_symbols += n_sym + file_stats.append((src, n_sym, n_disc)) + for d in parsed["discoveries"]: + d.setdefault("file", parsed["source_file"]) + d["shadow_path"] = str(sf.relative_to(shadow_dir)) + all_disc.append(d) + except Exception as e: + warn(f"Failed to process {sf}: {e}") + file_stats.sort(key=lambda x: x[2], reverse=True) + + # Header counts (always shown) + print("Shadow Knowledge Base Summary") + print("=" * 50) + print(f" Files shadowed: {len(shadow_files)}") + print(f" Symbols tracked: {total_symbols}") + print(f" Discoveries: {len(all_disc)}") + print(f" Preferences: {len(prefs)}") + print(f" Cross-cutting: {len(cross)}") + + # Source breakdown + try: + source_counts = defaultdict(int) + status_counts = defaultdict(int) + for d in all_disc: + source_counts[d.get("source", "unknown")] += 1 + status_counts[d.get("status", "unknown")] += 1 + + if source_counts: + print("\nBy source:") + for src, cnt in sorted(source_counts.items(), key=lambda x: -x[1]): + pct = cnt / len(all_disc) * 100 if all_disc else 0 + bar = "#" * int(pct / 2) + print(f" {src:15s} {cnt:4d} ({pct:5.1f}%) {bar}") + + if status_counts: + print("\nBy status:") + for st, cnt in sorted(status_counts.items(), key=lambda x: -x[1]): + pct = cnt / len(all_disc) * 100 if all_disc else 0 + bar = "#" * int(pct / 2) + print(f" {st:15s} {cnt:4d} ({pct:5.1f}%) {bar}") + except Exception as e: + warn(f"Failed to compute source/status breakdown: {e}") + + # Label breakdown + try: + label_counts = defaultdict(int) + for d in all_disc: + for lbl in d.get("labels", []): + label_counts[lbl] += 1 + if label_counts: + print("\nBy label:") + for lbl, cnt in sorted(label_counts.items(), key=lambda x: -x[1]): + print(f" {lbl:15s} {cnt:4d}") + except Exception as e: + warn(f"Failed to compute label breakdown: {e}") + + # Per-file table + try: + if file_stats: + print(f"\n{'File':<40s} {'Symbols':>8s} {'Disc.':>6s}") + print(f"{'-'*40} {'-'*8} {'-'*6}") + for src, n_sym, n_disc in file_stats[:20]: + print(f"{src:<40s} {n_sym:>8d} {n_disc:>6d}") + if len(file_stats) > 20: + print(f"... and {len(file_stats) - 20} more files") + except Exception as e: + warn(f"Failed to render per-file table: {e}") + + # Cross-cutting titles + try: + if cross: + print(f"\nCross-cutting discoveries:") + for e in cross: + title = e.get("title", e.get("slug", "?")) + cat = e.get("category", "?") + print(f" [{cat}] {title}") + except Exception as e: + warn(f"Failed to render cross-cutting list: {e}") + + # State info + try: + if state: + print(f"\nLast update: {state.get('last_update_at', '?')} " + f"({state.get('last_update_type', '?')})") + print(f"Last commit: {state.get('last_commit', '?')}") + except Exception as e: + warn(f"Failed to render state info: {e}") + + +def view_search(shadow_dir, query): + """Universal search: matches file names, symbol names, and discovery text. + + Also searches cross-cutting discoveries and preferences. + Results are grouped by match location for readability. + Each search domain (per-file, cross-cutting, prefs) is independent — + if one fails, the others still return results. + """ + query_lower = query.lower() + disc_matches = [] + cross_matches = [] + pref_matches = [] + section_errors = [] + + # Search per-file shadows + try: + for sf in get_all_shadow_files(shadow_dir): + try: + parsed = parse_shadow_file(sf) + source_file = parsed["source_file"] or str( + sf.relative_to(shadow_dir) + ) + file_name_hit = query_lower in source_file.lower() + + for d in parsed["discoveries"]: + sym = d.get("symbol", "") + text = d.get("text", "") + sym_hit = query_lower in sym.lower() + text_hit = query_lower in text.lower() + also_hit = any( + query_lower in ref.lower() + for ref in d.get("also_involves", []) + ) + + if file_name_hit or sym_hit or text_hit or also_hit: + disc_matches.append({ + "file": source_file, + "symbol": sym, + "text": text, + "status": d.get("status", "?"), + "source": d.get("source", "?"), + "also_involves": d.get("also_involves", []), + "match": ( + "file" if file_name_hit else + "symbol" if sym_hit else + "also_involves" if also_hit else "text" + ), + }) + except Exception as e: + warn(f"Search: error processing {sf}: {e}") + except Exception as e: + msg = f"Search: failed to search per-file shadows: {e}" + warn(msg) + section_errors.append(msg) + + # Search cross-cutting discoveries + try: + for e in parse_cross_cutting(shadow_dir): + title = e.get("title", "") + disc_text = e.get("discovery", "") + refs = e.get("refs", []) + if (query_lower in title.lower() + or query_lower in disc_text.lower() + or any(query_lower in r.lower() for r in refs)): + cross_matches.append(e) + except Exception as e: + msg = f"Search: failed to search cross-cutting: {e}" + warn(msg) + section_errors.append(msg) + + # Search preferences + try: + for p in parse_prefs(shadow_dir): + if query_lower in p.get("text", "").lower(): + pref_matches.append(p) + except Exception as e: + msg = f"Search: failed to search preferences: {e}" + warn(msg) + section_errors.append(msg) + + total = len(disc_matches) + len(cross_matches) + len(pref_matches) + if total == 0: + print(f"No results for '{query}'.") + if section_errors: + print(f"Note: {len(section_errors)} search section(s) had errors " + f"— results may be incomplete. Check stderr for details.") + return + + print(f"Search: '{query}' ({total} results)") + print("=" * 50) + + # Per-file discoveries, grouped by file + if disc_matches: + try: + by_file = defaultdict(list) + for d in disc_matches: + by_file[d["file"]].append(d) + + for file, discs in sorted(by_file.items()): + print(f"\n{file} ({len(discs)} matches)") + print("-" * (len(file) + 15)) + for d in discs: + sym = d["symbol"] + print(f" {file}::{sym}") + print(f" {d['text'][:120]}") + print(f" ({d['status']}, source: {d['source']})") + if d.get("also_involves"): + print( + f" Also involves: " + f"{', '.join(d['also_involves'])}" + ) + except Exception as e: + warn(f"Search: failed to render per-file results: {e}") + + # Cross-cutting + if cross_matches: + try: + print(f"\nCross-cutting ({len(cross_matches)} matches)") + print("-" * 30) + for e in cross_matches: + title = e.get("title", e.get("slug", "?")) + cat = e.get("category", "?") + status = e.get("status", "?") + source = e.get("source", "?") + print(f"\n {title}") + print(f" Category: {cat} | {status}, source: {source}") + print(f" Refs: {', '.join(e.get('refs', [])[:5])}") + if e.get("discovery"): + print(f" {e['discovery'][:120]}") + except Exception as e: + warn(f"Search: failed to render cross-cutting results: {e}") + + # Preferences + if pref_matches: + try: + print(f"\nPreferences ({len(pref_matches)} matches)") + print("-" * 30) + for p in pref_matches: + print(f" [{p.get('source', '?')}] {p['text'][:120]}") + except Exception as e: + warn(f"Search: failed to render preference results: {e}") + + +def view_prefs(shadow_dir): + """Show all preferences.""" + try: + prefs = parse_prefs(shadow_dir) + except Exception as e: + error(f"Failed to parse preferences: {e}") + return + + if not prefs: + print("No preferences recorded yet.") + return + + print(f"Project Preferences ({len(prefs)} total)") + print("=" * 40) + for p in prefs: + try: + source = p.get("source", "?") + print(f"\n [{source}] {p['text']}") + except Exception as e: + warn(f"Failed to render preference: {e}") + + +def view_labels(shadow_dir, label_filter): + """Show discoveries filtered by label(s). + + label_filter can be a single label or comma-separated list. + """ + try: + filters = [l.strip().lower() for l in label_filter.split(",")] + except Exception as e: + error(f"Invalid label filter '{label_filter}': {e}") + return + + try: + all_disc = collect_all_discoveries(shadow_dir) + except Exception as e: + error(f"Failed to collect discoveries for label filtering: {e}") + return + + # Also include cross-cutting discoveries with labels + try: + for entry in parse_cross_cutting(shadow_dir): + if entry.get("labels"): + all_disc.append({ + "file": f"_cross/{entry.get('file', '?')}", + "symbol": entry.get("title", entry.get("slug", "?")), + "text": entry.get("discovery", entry.get("title", "")), + "status": entry.get("status", "?"), + "source": entry.get("source", "?"), + "labels": entry["labels"], + }) + except Exception as e: + warn(f"Failed to include cross-cutting in label search: {e}") + + matching = [] + for d in all_disc: + try: + disc_labels = [l.lower() for l in d.get("labels", [])] + if any(f in disc_labels for f in filters): + matching.append(d) + except Exception as e: + warn(f"Failed to check labels on discovery in " + f"{d.get('file', '?')}::{d.get('symbol', '?')}: {e}") + + if not matching: + print(f"No discoveries with label(s): {', '.join(filters)}") + return + + print(f"Discoveries with label(s): {', '.join(filters)} " + f"({len(matching)} results)") + print("=" * 50) + + by_label = defaultdict(list) + for d in matching: + for lbl in d.get("labels", []): + if lbl.lower() in filters: + by_label[lbl.lower()].append(d) + + for lbl in filters: + discs = by_label.get(lbl, []) + if not discs: + continue + print(f"\n[{lbl}] ({len(discs)} discoveries)") + print("-" * 30) + for d in discs: + try: + sym = d.get("symbol", "?") + src_file = d.get("file", "?") + print(f" {src_file}::{sym}") + print(f" {d['text'][:120]}") + print(f" ({d.get('status', '?')}, " + f"source: {d.get('source', '?')})") + all_labels = d.get("labels", []) + other = [l for l in all_labels if l.lower() != lbl] + if other: + print(f" Also labeled: {', '.join(other)}") + except Exception as e: + warn(f"Failed to render labeled discovery: {e}") + + +def view_recent(shadow_dir, count=10): + """Show the N most recent discoveries (by shadow file mtime). + + Collects all discoveries across all shadow files, cross-cutting entries, + and preferences, sorts by the source file's modification time (most recent + first), and shows the actual discovery content. + Each data source is independent — if cross-cutting fails, per-file + discoveries still appear. + """ + all_items = [] + + # Per-file discoveries + try: + for sf in get_all_shadow_files(shadow_dir): + try: + mtime = os.path.getmtime(sf) + parsed = parse_shadow_file(sf) + source_file = parsed["source_file"] or str( + sf.relative_to(shadow_dir) + ) + for d in parsed["discoveries"]: + all_items.append({ + "type": "discovery", + "file": source_file, + "symbol": d.get("symbol", "?"), + "text": d.get("text", ""), + "status": d.get("status", "?"), + "source": d.get("source", "?"), + "mtime": mtime, + }) + except Exception as e: + warn(f"Recent: failed to process {sf}: {e}") + except Exception as e: + warn(f"Recent: failed to list shadow files: {e}") + + # Cross-cutting discoveries + try: + cross_entries = parse_cross_cutting(shadow_dir) + cross_by_file = defaultdict(list) + for e in cross_entries: + cross_by_file[e["file"]].append(e) + + cross_dir = shadow_dir / "_cross" + if cross_dir.exists(): + for cf in cross_dir.glob("*.md"): + try: + mtime = os.path.getmtime(cf) + for e in cross_by_file.get(cf.name, []): + all_items.append({ + "type": "cross-cutting", + "file": f"_cross/{cf.name}", + "symbol": e.get("title", e.get("slug", "?")), + "text": e.get("discovery", e.get("title", "")), + "status": e.get("status", "?"), + "source": e.get("source", "?"), + "mtime": mtime, + }) + except Exception as e: + warn(f"Recent: failed to process cross-cutting {cf}: {e}") + except Exception as e: + warn(f"Recent: failed to process cross-cutting discoveries: {e}") + + # Preferences + try: + prefs_path = shadow_dir / "_prefs.md" + if prefs_path.exists(): + mtime = os.path.getmtime(prefs_path) + for p in parse_prefs(shadow_dir): + all_items.append({ + "type": "preference", + "file": "_prefs.md", + "symbol": "-", + "text": p.get("text", ""), + "status": "-", + "source": p.get("source", "?"), + "mtime": mtime, + }) + except Exception as e: + warn(f"Recent: failed to process preferences: {e}") + + all_items.sort(key=lambda x: x.get("mtime", 0), reverse=True) + + if not all_items: + print("No discoveries found.") + return + + shown = all_items[:count] + print(f"Most Recent Discoveries (top {count})") + print("=" * 50) + for item in shown: + try: + ts = datetime.fromtimestamp( + item.get("mtime", 0) + ).strftime("%Y-%m-%d %H:%M") + kind = item.get("type", "?") + sym = item.get("symbol", "?") + + print(f"\n [{ts}] ({kind})") + if kind == "preference": + print(f" {item.get('text', '')[:120]}") + print(f" source: {item.get('source', '?')}") + else: + print(f" {item.get('file', '?')}::{sym}") + print(f" {item.get('text', '')[:120]}") + print(f" ({item.get('status', '?')}, " + f"source: {item.get('source', '?')})") + except Exception as e: + warn(f"Recent: failed to render item: {e}") + + +def view_top(shadow_dir, file_path, labels_filter, limit, max_chars): + """Show the top N actionable discoveries for a single source file. + + Designed for the preToolUse hook: concise output suitable for + inlining into additionalContext when the agent is about to mutate a + file. Pulls from both the per-file shadow and any _cross/ entries + whose refs touch this file. + + Ranking: verified > uncertain > refuted; within a tier, source + order is preserved. Output is hard-capped at max_chars (the trailing + "(...)" marker still fits). + """ + norm = file_path.strip() + if norm.startswith("./"): + norm = norm[2:] + shadow_path = shadow_dir / f"{norm}.md" + label_set = {l.strip().lower() for l in labels_filter.split(",") if l.strip()} + + candidates = [] + + if shadow_path.is_file(): + try: + parsed = parse_shadow_file(shadow_path) + for d in parsed.get("discoveries", []): + disc_labels = {l.lower() for l in d.get("labels", [])} + if label_set and not (disc_labels & label_set): + continue + candidates.append({ + "kind": "file", + "anchor": d.get("symbol") or "file-level", + "text": d.get("text", "").strip(), + "status": d.get("status", "?"), + "labels": sorted(disc_labels), + }) + except Exception as e: + warn(f"--top: failed parsing {shadow_path}: {e}") + + try: + for entry in parse_cross_cutting(shadow_dir): + refs = entry.get("refs", []) or [] + if not any(r.split("::", 1)[0].strip() == norm for r in refs): + continue + cross_labels = {l.lower() for l in entry.get("labels", [])} + if label_set and not (cross_labels & label_set): + continue + candidates.append({ + "kind": "cross", + "anchor": f"_cross/{entry.get('file', entry.get('slug', '?'))}", + "text": (entry.get("discovery") or entry.get("title") or "").strip(), + "status": entry.get("status", "?"), + "labels": sorted(cross_labels), + }) + except Exception as e: + warn(f"--top: failed scanning _cross/: {e}") + + if not candidates: + labels_disp = ",".join(sorted(label_set)) if label_set else "any" + print( + f"No actionable discoveries ({labels_disp}) for {norm}." + ) + return + + tier = {"verified": 0, "uncertain": 1, "refuted": 2} + candidates.sort(key=lambda d: tier.get(d.get("status", "?"), 3)) + + shown = candidates[:limit] + header = ( + f"Top {len(shown)} of {len(candidates)} actionable discoveries " + f"for {norm}:" + ) + lines = [header] + for d in shown: + labels = ",".join(d["labels"]) if d["labels"] else "—" + anchor = d["anchor"] + text = d["text"].replace("\n", " ").strip() + lines.append( + f"- [{labels}] `{anchor}` ({d['status']}): {text}" + ) + + out = "\n".join(lines) + if max_chars and len(out) > max_chars: + truncated = out[: max_chars - 6].rstrip() + out = truncated + "\n(...)" + print(out) + + +def view_check_invariants(shadow_dir): + """Walk the shadow knowledge base and report invariant violations. + + Statically-checkable invariants from shadow-frog/SKILL.md: + #3 (partial) Per-file 'Also involves:' uses file::symbol notation + #4 Cross-ref back-pointers match: _cross/.md refs <-> + per-file ## Cross-References + #5 Every ## Cross-References entry has a matching _cross/*.md + + Plus syntactic guards that catch the most common drift: + - Symbol headings use the required backtick form + - Discovery metadata uses valid status enum + - Discovery metadata uses valid source enum + - Discovery labels are from the allowed set + - _cross/ Category field uses a known value + + Invariants #1, #2, #7 are NOT checked (would require source parsing + and semantic match); #6 is filesystem-enforced. + + Exit 0 = clean, 1 = at least one violation. Violations print one per + line in `path:line: kind: message` form so grep/editors can navigate. + """ + VALID_STATUS = {"verified", "uncertain", "refuted"} + VALID_SOURCE = {"exploration", "user", "interaction"} + VALID_LABELS = {"bug", "performance", "security", + "feature-gap", "tech-debt"} + VALID_CATEGORIES = { + "pattern", "behavior", "edge-case", "contract", + "performance", "intent", "warning", "history", "convention", + } + + violations = [] + def v(path, line, kind, msg): + violations.append(f"{path}:{line}: {kind}: {msg}") + + # Pass 1: walk per-file shadows -> collect cross-reference entries + # they declare and validate their internal format. + per_file_xref_targets = {} # rel_shadow_path -> set(slug declared) + cross_dir = shadow_dir / "_cross" + cross_slugs_on_disk = set() + if cross_dir.is_dir(): + try: + cross_slugs_on_disk = {f.stem for f in cross_dir.glob("*.md")} + except OSError as e: + warn(f"Cannot list {cross_dir}: {e}") + + md_heading_re = re.compile(r"^(#{2,3})\s+(.*)$") + backtick_heading_re = re.compile(r"^(#{2,3})\s+`[^`]+`\s*$") + also_involves_re = re.compile(r"^\s*Also involves:\s*(.+)$", re.I) + file_sym_re = re.compile(r"`([^`]+::[^`]+)`") + + for shadow_path in get_all_shadow_files(shadow_dir): + try: + rel = shadow_path.relative_to(shadow_dir) + except ValueError: + continue + try: + text = shadow_path.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError) as e: + v(rel, 0, "unreadable", str(e)) + continue + + in_cross_refs = False + declared = set() + for ln, raw in enumerate(text.split("\n"), 1): + line = raw.rstrip() + + heading = md_heading_re.match(line) + if heading: + title = heading.group(2).strip() + if title.lower().startswith("cross-references"): + in_cross_refs = True + continue + in_cross_refs = False + # Skip special headings ("File-Level Notes", "Notes", etc.) + if ( + title.lower().startswith("file-level") + or title.lower() in {"notes", "metadata"} + ): + continue + # Symbol heading must use backtick form + if not backtick_heading_re.match(line): + v(rel, ln, "heading", + f"symbol heading must be `## `name`` or " + f"`### `Class.name``; got: {line[:80]}") + continue + + if in_cross_refs and line.strip().startswith("- "): + # Format: - [slug](.shadow/_cross/slug.md) — title + slug_match = re.search( + r"_cross/([^)\s]+?)\.md", line + ) + if slug_match: + declared.add(slug_match.group(1)) + else: + # Looser fallback: bare slug in brackets + alt = re.search(r"\[([^\]]+)\]", line) + if alt: + declared.add(alt.group(1).strip()) + + # Discovery metadata line + md = _DISCOVERY_META_RE.search(line) + if md: + status, source = md.group(1), md.group(2) + labels_raw = md.group(3) or "" + if status not in VALID_STATUS: + v(rel, ln, "enum", + f"status '{status}' not in {sorted(VALID_STATUS)}") + if source not in VALID_SOURCE: + v(rel, ln, "enum", + f"source '{source}' not in {sorted(VALID_SOURCE)}") + for lbl in (l.strip() for l in labels_raw.split(",") if l.strip()): + if lbl not in VALID_LABELS: + v(rel, ln, "enum", + f"label '{lbl}' not in {sorted(VALID_LABELS)}") + + # `Also involves:` must list file::symbol anchors in backticks + ai = also_involves_re.match(line) + if ai: + rest = ai.group(1) + anchors = file_sym_re.findall(rest) + if not anchors: + v(rel, ln, "anchor", + "Also involves: needs `file::symbol` " + "backtick anchors") + # Light sanity: every anchor has both file and symbol + for a in anchors: + if "::" not in a or not a.split("::", 1)[1].strip(): + v(rel, ln, "anchor", + f"anchor '{a}' missing symbol after ::") + + per_file_xref_targets[str(rel)] = declared + + # Invariant #5: every declared cross slug must exist on disk + for slug in declared: + if slug not in cross_slugs_on_disk: + v(rel, 0, "cross-ref", + f"references _cross/{slug}.md but file does not exist") + + # Pass 2: walk _cross/*.md -> validate refs format + back-pointer. + # Build the reverse map: cross_slug -> set(file::symbol it points at). + cross_back = {} # slug -> set(file paths it should be linked from) + if cross_dir.is_dir(): + for cf in sorted(cross_dir.glob("*.md")): + slug = cf.stem + try: + text = cf.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError) as e: + v(cf.relative_to(shadow_dir), 0, "unreadable", str(e)) + continue + + rel_cf = cf.relative_to(shadow_dir) + + # Category enum check + cat_m = re.search(r"\*\*Category\*\*:\s*(.+)", text) + if cat_m: + cat = cat_m.group(1).strip().lower() + if cat not in VALID_CATEGORIES: + v(rel_cf, 0, "enum", + f"Category '{cat}' not in {sorted(VALID_CATEGORIES)}") + else: + v(rel_cf, 0, "schema", + "missing **Category**: field") + + # Discovery metadata + md = _DISCOVERY_META_RE.search(text) + if md: + status, source = md.group(1), md.group(2) + if status not in VALID_STATUS: + v(rel_cf, 0, "enum", + f"status '{status}' not in {sorted(VALID_STATUS)}") + if source not in VALID_SOURCE: + v(rel_cf, 0, "enum", + f"source '{source}' not in {sorted(VALID_SOURCE)}") + else: + v(rel_cf, 0, "schema", + "missing trailing _(status, source: ...)_ metadata") + + # Refs must be `file::symbol` anchors + refs_block = re.search( + r"\*\*Refs\*\*:\s*\n((?:\s*-\s+`[^`]+`\s*\n?)+)", + text, + ) + if not refs_block: + v(rel_cf, 0, "schema", + "missing **Refs**: block (one per line, " + "`- `file::symbol``)") + else: + anchors = file_sym_re.findall(refs_block.group(1)) + if not anchors: + v(rel_cf, 0, "anchor", + "Refs block has no `file::symbol` entries") + for a in anchors: + if "::" not in a or not a.split("::", 1)[1].strip(): + v(rel_cf, 0, "anchor", + f"ref '{a}' missing symbol after ::") + else: + # Convert file part to shadow path: + # src/foo.py -> src/foo.py.md (relative to shadow_dir) + file_part = a.split("::", 1)[0].strip() + shadow_rel = f"{file_part}.md" + cross_back.setdefault(slug, set()).add(shadow_rel) + + # Invariant #4 back-pointer: every file referenced by a cross slug + # must declare that slug in its ## Cross-References. + for slug, expected_files in cross_back.items(): + for shadow_rel in expected_files: + declared = per_file_xref_targets.get(shadow_rel) + if declared is None: + v(f"_cross/{slug}.md", 0, "cross-ref", + f"refs {shadow_rel} but no such shadow file exists") + elif slug not in declared: + v(f"_cross/{slug}.md", 0, "cross-ref", + f"refs {shadow_rel} but that shadow's ## " + f"Cross-References does not link back to " + f"_cross/{slug}.md") + + # Output + if not violations: + print(f"✓ Invariants OK ({len(per_file_xref_targets)} per-file " + f"shadows, {len(cross_slugs_on_disk)} cross-cutting " + f"discoveries)") + return 0 + + for line in violations: + print(line) + print(f"\n{len(violations)} invariant violation(s) found.", + file=sys.stderr) + return 1 + + +def main(): + try: + parser = argparse.ArgumentParser( + description="Query and browse a .shadow/ knowledge base.", + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + + # Views (mutually exclusive) + views = parser.add_mutually_exclusive_group() + views.add_argument( + "--summary", action="store_true", + help="Overview + detailed statistics (default)", + ) + views.add_argument( + "--search", metavar="QUERY", + help="Universal search: files, symbols, and discovery text", + ) + views.add_argument( + "--prefs", action="store_true", + help="Show project-wide preferences", + ) + views.add_argument( + "--recent", nargs="?", const=10, type=int, metavar="N", + help="N most recent discoveries with content (default: 10)", + ) + views.add_argument( + "--labels", metavar="LABEL", + help=( + "Show discoveries by label " + "(e.g., bug, security, bug,performance)" + ), + ) + views.add_argument( + "--top", metavar="FILE", + help=( + "Top actionable discoveries for FILE (a source path " + "like src/auth.py). Concise output for the preToolUse " + "hook: filters to actionable labels (default: " + "bug,security), includes both per-file and _cross/ " + "entries that reference FILE, ranks verified first." + ), + ) + views.add_argument( + "--check-invariants", action="store_true", + help=( + "Walk the shadow and report structural violations: " + "missing back-pointers, dangling _cross/ refs, invalid " + "enums, bad heading format. Exit 1 if any are found." + ), + ) + + # Options + parser.add_argument( + "--shadow-dir", default=None, + help="Path to .shadow/ directory (default: auto-detect)", + ) + parser.add_argument( + "--top-labels", default="bug,security", metavar="LABELS", + help=( + "Comma-separated labels to include in --top " + "(default: bug,security). Pass empty string to include " + "all labeled discoveries." + ), + ) + parser.add_argument( + "--top-limit", type=int, default=3, metavar="N", + help="Max discoveries to show in --top (default: 3)", + ) + parser.add_argument( + "--top-max-chars", type=int, default=600, metavar="N", + help=( + "Hard cap on --top total output length " + "(default: 600). Use 0 for no cap." + ), + ) + + args = parser.parse_args() + + # Find shadow dir + if args.shadow_dir: + shadow_dir = Path(args.shadow_dir) + else: + shadow_dir = find_shadow_dir() + + if not shadow_dir or not shadow_dir.is_dir(): + cwd = os.getcwd() + error( + f"No .shadow/ directory found. " + f"Searched from: {cwd}\n" + f"[shadow-viewer error] " + f"Run /shadow-frog-init first to create the shadow, " + f"or pass --shadow-dir /path/to/.shadow/ explicitly." + ) + if args.shadow_dir: + error( + f"Provided --shadow-dir '{args.shadow_dir}' does not " + f"exist or is not a directory." + ) + sys.exit(1) + + # Dispatch + if args.check_invariants: + sys.exit(view_check_invariants(shadow_dir)) + if args.top: + view_top( + shadow_dir, + args.top, + args.top_labels, + args.top_limit, + args.top_max_chars, + ) + elif args.search: + view_search(shadow_dir, args.search) + elif args.prefs: + view_prefs(shadow_dir) + elif args.labels: + view_labels(shadow_dir, args.labels) + elif args.recent is not None: + view_recent(shadow_dir, args.recent) + else: + view_summary(shadow_dir) + + except SystemExit: + raise + except KeyboardInterrupt: + error("Interrupted by user.") + sys.exit(130) + except Exception as e: + error( + f"Unexpected error: {type(e).__name__}: {e}\n" + f"[shadow-viewer error] Full traceback:\n" + f"{traceback.format_exc()}" + f"This is likely a bug in shadow-viewer.py. " + f"The shadow data may be in an unexpected format. " + f"Try running with --shadow-dir to confirm the path, " + f"or inspect the .shadow/ files manually." + ) + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/skills/shadow-frog/SKILL.md b/skills/shadow-frog/SKILL.md new file mode 100644 index 0000000..30ffeab --- /dev/null +++ b/skills/shadow-frog/SKILL.md @@ -0,0 +1,373 @@ +--- +name: shadow-frog +description: >- + Use a shadow knowledge base to understand any codebase. The .shadow/ + directory mirrors the source tree with markdown files containing + behavioral insights — known bugs, edge cases, implicit contracts, + and user preferences. Always check the shadow before editing, + debugging, or investigating code. When the user shares important + context, write it to the shadow immediately. Invoke shadow-frog-init + to create it, shadow-frog-update to refresh it, shadow-frog-dream + for autonomous exploration, shadow-frog-meditate for shadow hygiene, + or shadow-frog-viewer to browse it. +--- + +# ShadowFrog + +`.shadow/` mirrors the source tree. Each source file has a `.md` shadow organized +by symbol. Each symbol section contains discoveries — behavioral insights anchored +to that code location. + +## Required Actions + +**Every time you work on code in a repo with `.shadow/`:** + +1. **Read `_prefs.md` first** — it contains project-wide conventions, + user preferences, and things the user explicitly wants to avoid. + Violating a preference wastes the user's time. +2. **Read `_cross/` discoveries** — these are the highest-value findings, + spanning multiple files. List `_cross/` and read any files whose titles + relate to the area you're working in. Cross-cutting discoveries reveal + hidden contracts, interaction bugs, and design patterns that per-file + shadows alone cannot capture. +3. **Check `_dreams/` for experiment results** — `_dreams/_index.md` lists + autonomous exploration experiments. Read reports relevant to your task — + they contain verified bug analyses, attempted fixes, and architectural + insights. Dreams may contain knowledge not yet distilled into per-file + shadows, so always check when investigating a bug or unfamiliar area. +4. **Before editing any file**: read its shadow (`.shadow/.md`), + check `_cross/` for cross-cutting discoveries about it, and apply + what you learn. The shadow contains known bugs, edge cases, and + implicit contracts discovered by previous sessions. + **Note**: `_index.md` discovery counts may be stale — always check + per-file shadows and `_cross/` directly rather than relying solely on + the index summary. +5. **When the user explains something about code** (gotcha, design intent, + warning, history): write a `source: user` discovery to the shadow + immediately. Do not ask where to put it — resolve the `file::symbol` + anchor yourself by searching `_index.md`, shadow files, and session + context (current file, recent edits). +6. **When the user states a preference or convention** (not tied to any + specific file): write it to `_prefs.md` immediately. +7. **After code changes**: run `/shadow-frog-update` + +## Directory Layout + +``` +.shadow/ + .shadowignore Gitignore-syntax file for excluding paths from the shadow + _index.md File list with symbol counts and discovery counts + _prefs.md Project-wide user preferences (not tied to any file/symbol) + _cross/ Cross-cutting discoveries (span multiple files) + .md One file per cross-cutting discovery (descriptive kebab-case name) + _meta/ + state.json Last commit, timestamps, counts + _dreams/ Dream experiment archive (detailed reports + diffs) + _index.md Table of all experiments with verdicts + / One folder per experiment + report.md Structured report with YAML frontmatter + patch.diff Full implementation diff against base_commit + / Per-file shadows + file.py.md Organized by symbol +``` + +## Reference Notation + +Canonical format: `file_path::symbol_name` + +Examples: `src/auth.py::authenticate_user`, `src/auth.py::UserAuth.validate`, +`src/auth.py` (file-level, no symbol) + +The symbol name is the stable anchor. + +## Per-File Shadow Format + +```markdown +# Shadow: src/auth.py + +**Language**: Python | **Lines**: 142 | **Last modified**: 2025-01-15 + +## File-Level + +- This module has no __all__ — all top-level names are public. + _(verified, source: exploration)_ + +## `class UserAuth` + +### `UserAuth.validate` + +- Catches ALL exceptions and returns False — swallows + connection errors, making network failures look like invalid tokens. + _(verified, source: exploration)_ + +## `authenticate_user` + +- Silently returns None on expired tokens. Callers must check. + _(verified, source: exploration, labels: [bug])_ + Also involves: `src/middleware.py::require_auth` + +## Cross-References + +- [db-connection-lifecycle](../_cross/db-connection-lifecycle.md) + (involves `src/db/connection.py::ConnectionPool`, `src/api/routes.py::get_user`) +``` + +Heading format (**hard rule — parsers depend on this**): +- Top-level symbols (classes, functions, constants): `##` heading with symbol in backticks +- Nested symbols (methods): `###` heading with symbol in backticks +- `## Cross-References` — always last section + +The viewer parser only matches the backtick form. A heading written as +`### UserAuth.validate` (no backticks) will have its discoveries +silently dropped from search/top output. Always wrap the symbol in +backticks, including for nested symbols. + +Examples: +``` +## `authenticate_user` +## `class UserAuth` +### `UserAuth.validate` +``` + +## Cross-Cutting File Format (`_cross/.md`) + +```markdown +# Database connection lifecycle + +**Category**: pattern +**Refs**: +- `src/db/connection.py::ConnectionPool.get` +- `src/auth.py::authenticate_user` +- `src/api/routes.py::get_user` + +**Discovery**: All database access goes through a connection pool that +silently reconnects on failure. First request after DB restart is slow (~2s). + +_(verified, source: exploration)_ +``` + +## Preferences File (`_prefs.md`) + +Project-wide user preferences and conventions that are not tied to any +specific file or symbol. These guide all agent work across the codebase. + +```markdown +# Preferences + +- No backward compatibility — only keep the latest code, no shims or aliases. + _(source: user)_ + +- Use snake_case for all Python function and variable names. + _(source: user)_ + +- Prefer small, focused PRs over large sweeping changes. + _(source: interaction)_ +``` + +Format: +``` +- + _(source: )_ +``` + +Preferences are always trusted (same rank as `source: user`). They don't +need `verified/uncertain/refuted` — if the user said it, it's a directive. + +When to write to `_prefs.md` vs per-file shadow vs `_cross/`: +- Applies to the whole repo, no specific file → `_prefs.md` +- Applies to a specific file or symbol → per-file shadow +- Applies to 3+ specific files → `_cross/.md` + +## Discovery Format + +Per-file discoveries (no IDs — anchored by their `file::symbol` heading): +``` +- + _(, source: )_ + Also involves: `file::symbol`, `file::symbol` +``` + +With labels (optional — only when the discovery is actionable): +``` +- + _(, source: , labels: [bug, security])_ + Also involves: `file::symbol` +``` + +With dream report link (optional — only for experiment-derived discoveries): +``` +- + _(, source: )_ + Dream report: `_dreams//` +``` + +Cross-cutting discoveries (one per `_cross/.md` file): +``` +# + +**Category**: <category> +**Refs**: +- `file::symbol` + +**Discovery**: <behavioral statement> + +_(<verified|uncertain|refuted>, source: <exploration|user|interaction>)_ +``` + +Slug naming: use descriptive kebab-case derived from the title. +Example: title "Database connection lifecycle" → filename `db-connection-lifecycle.md` + +### Labels + +Labels mark actionable discoveries so agents can quickly scan for specific +types. Most discoveries are just knowledge — labels are only for findings +that call for action. + +| Label | Use when | +|-------|----------| +| `bug` | A defect that should be fixed | +| `performance` | A bottleneck or inefficiency | +| `security` | A vulnerability or unsafe pattern | +| `feature-gap` | Missing functionality or improvement opportunity | +| `tech-debt` | Code smell, duplication, refactoring opportunity | + +A discovery can have multiple labels: `labels: [bug, security]`. +Omit labels entirely for pure observational knowledge. + +Labels go in the metadata line: +``` +_(verified, source: exploration, labels: [bug])_ +``` + +Cross-cutting discoveries can also have labels — add them to the metadata line. + +### Fields + +- `verified|uncertain|refuted` — verification status +- `source: exploration` — agent discovered via code analysis +- `source: user` — human stated it in conversation +- `source: interaction` — emerged from collaborative work (debugging, refactoring) +- `labels: [...]` — optional, actionable labels (see table above) +- `Also involves:` — `file::symbol` refs to other code locations (required if discovery touches other files) +- `Dream report:` — optional, `_dreams/<dream-id>/` link for experiment-derived discoveries +- `Category` (cross-cutting only): pattern, behavior, edge-case, contract, performance, intent, warning, history, convention + +## Trust Order + +1. `source: user` — highest trust, always `verified` +2. `source: interaction` — always `verified` +3. `verified` from exploration +4. `uncertain` — not yet confirmed +5. `refuted` — skip + +## Five Reference Links (all must be maintained) + +1. **File mapping**: `src/auth.py` ↔ `.shadow/src/auth.py.md` +2. **Symbol anchoring**: every source symbol has a `##`/`###` heading in its shadow +3. **Also involves**: per-file discoveries list other `file::symbol` locations +4. **Cross-ref back-pointers**: per-file `## Cross-References` links to `_cross/<slug>.md` entries +5. **Cross-cutting refs**: `_cross/<slug>.md` `**Refs**:` lists all involved `file::symbol` locations + +Links 4 and 5 are bidirectional: if `_cross/db-connection-lifecycle.md` references +`src/auth.py::fn`, then `src/auth.py.md` must list it in `## Cross-References`, and vice versa. + +## Seven Invariants + +1. Every included source file has exactly one shadow at `.shadow/<path>.md` +2. Every symbol in source has a `##`/`###` heading in its shadow +3. Per-file discoveries touching other files have `Also involves:` with `file::symbol` +4. Cross-ref back-pointers match: `_cross/<slug>.md` refs ↔ per-file `## Cross-References` +5. Every entry in `## Cross-References` has a corresponding `_cross/<slug>.md` file +6. Cross-cutting filenames are unique (enforced by filesystem) +7. No duplicate discoveries (same behavioral claim at same symbol) + +To audit a shadow for structural drift (invariant 3 format, invariants 4–5, +plus enum and heading-format guards), locate the viewer script and run it: + +```bash +VIEWER="" +for DIR in .github/skills/shadow-frog-viewer .claude/skills/shadow-frog-viewer; do + [ -f "$DIR/shadow-viewer.py" ] && VIEWER="$DIR/shadow-viewer.py" && break +done +python3 "$VIEWER" --check-invariants +``` + +Exits 0 if clean, 1 with one violation per line otherwise. Invariant 3 is +checked for anchor *format* only (not existence of the referenced file or +symbol); invariants 1, 2, and 7 require source parsing / semantic match and +are not statically checked; invariant 6 is filesystem-enforced. + +## Lookup Commands + +```bash +# File's shadow +cat .shadow/src/auth.py.md + +# Specific symbol's knowledge +grep -A 20 "## \`authenticate_user\`" .shadow/src/auth.py.md + +# Cross-cutting discoveries for a file +grep -rl "src/auth.py::" .shadow/_cross/ + +# Search by topic +grep -rl "error.handling\|exception" .shadow/ --include="*.md" + +# All user-shared knowledge +grep -r "source: user" .shadow/ --include="*.md" + +# Project-wide preferences +cat .shadow/_prefs.md + +# List all cross-cutting discovery files +ls .shadow/_cross/ +``` + +## Verification + +Two methods, use whichever fits the claim: + +**Observe-based** (for simpler claims — code reading suffices): +1. Read the source code at the relevant `file::symbol` +2. Trace the logic: does the behavioral claim hold? +3. If confirmed → `verified`. If contradicted → `refuted`. If unclear → `uncertain`. + +**Do-based** (for harder claims — requires execution): +1. Write a short verification script (test, assertion, or probe) that would + confirm or refute the claim +2. Run it +3. Based on the result → `verified` or `refuted` + +Prefer do-based for claims about runtime behavior, performance, error handling +paths, or race conditions. Prefer observe-based for claims about code structure, +types, or static properties. + +## Dedup and Writing Rules + +Before writing any discovery, follow this procedure: + +1. **Read before write**: Read all existing discoveries under the target + `file::symbol`. If an existing discovery makes the same behavioral + claim (even if worded differently) → update the existing one. If the + new one extends an existing one → merge into a single richer entry. + If they conflict → investigate the code, keep the correct one, mark + the other `refuted`. +2. **Append, don't replace**: If the symbol already has discoveries, + append your new one after them. If there is a `_No discoveries yet._` + placeholder, remove it and write your discovery. Never use the + placeholder as an edit anchor if it's already gone — read the file first. +3. Write in canonical format (see Discovery Format above) +4. **Fix bad format**: if you see any existing content that doesn't + follow the canonical format, fix it in place + +For cross-cutting, search `_cross/` for overlapping `**Refs**:` sets +before creating a new entry. Only create `_cross/<slug>.md` for +discoveries spanning **3+ files** — otherwise use per-file entries with +`Also involves:` references. + +## Related Skills + +- `/shadow-frog-init` — create `.shadow/` for a new repo +- `/shadow-frog-update` — refresh shadows after changes or from conversation +- `/shadow-frog-dream` — autonomous exploration and experimentation while user is AFK +- `/shadow-frog-meditate` — deduplicate, merge, and resolve conflicting discoveries +- `/shadow-frog-viewer` — browse and query the shadow (overview, search, preferences, recent) diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..aabcbe6 --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,131 @@ +"""Repo-wide pytest fixtures for ShadowFrog tests. + +Most scripts in this repo are hyphenated CLI files (e.g. `shadow-init.py`) +that aren't importable as regular Python modules. The `_load_script` +helper provides an importlib-based loader so tests can reach the internal +functions without subprocess overhead — but every public CLI path +(argparse dispatcher) should still also be exercised via subprocess in +at least one integration test, since the importable view of a script +bypasses the argument parser. +""" +import importlib.util +import shutil +import subprocess +import sys +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent + + +def _load_script(path): + """Load a hyphenated Python script as a callable module. + + Subsequent loads of the same path return a fresh module — tests that + need isolated module-level state (e.g. tweaking module-level + constants) should NOT share the fixture across tests. + """ + path = Path(path).resolve() + spec = importlib.util.spec_from_file_location(f"_loaded_{path.stem}", path) + mod = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = mod + spec.loader.exec_module(mod) + return mod + + +# --- Script-loader fixtures (session-scoped — module state is read-only) --- + +@pytest.fixture(scope="session") +def repo_root(): + """Absolute path to the ShadowFrog repo root.""" + return REPO_ROOT + + +@pytest.fixture(scope="session") +def shadow_init(repo_root): + return _load_script(repo_root / "skills/shadow-frog-init/shadow-init.py") + + +@pytest.fixture(scope="session") +def shadow_viewer(repo_root): + return _load_script(repo_root / "skills/shadow-frog-viewer/shadow-viewer.py") + + +@pytest.fixture(scope="session") +def dream_reconcile(repo_root): + return _load_script(repo_root / "skills/shadow-frog-dream/dream-reconcile.py") + + +@pytest.fixture(scope="session") +def dream_validate(repo_root): + return _load_script(repo_root / "skills/shadow-frog-dream/dream-validate.py") + + +@pytest.fixture(scope="session") +def dream_coverage(repo_root): + return _load_script(repo_root / "skills/shadow-frog-dream/dream-coverage.py") + + +@pytest.fixture(scope="session") +def dream_lineage(repo_root): + return _load_script(repo_root / "skills/shadow-frog-viewer/dream-lineage.py") + + +@pytest.fixture(scope="session") +def meditate_repair(repo_root): + return _load_script(repo_root / "skills/shadow-frog-meditate/meditate-repair.py") + + +# --- Filesystem fixtures --- + +@pytest.fixture(scope="session") +def coupon_demo_src(repo_root): + """Read-only path to the canonical coupon-demo. Tests MUST NOT mutate this.""" + return repo_root / "examples/coupon-demo" + + +@pytest.fixture +def coupon_demo(tmp_path, coupon_demo_src): + """A mutable, fresh-per-test copy of coupon-demo, initialized as a git repo. + + Use this for any test that needs to read/write `.shadow/` or any + source file. Each test gets its own copy. + """ + dst = tmp_path / "coupon-demo" + shutil.copytree( + coupon_demo_src, dst, + ignore=shutil.ignore_patterns("__pycache__", "*.pyc"), + ) + _git_init(dst, commit_all=True) + return dst + + +@pytest.fixture +def tmp_git_repo(tmp_path): + """An empty initialized git repo (no files committed). For init tests.""" + repo = tmp_path / "repo" + repo.mkdir() + _git_init(repo, commit_all=False) + return repo + + +def _git_init(path, commit_all): + """Init a git repo at `path` with deterministic identity. Optional initial commit.""" + env = { + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "HOME": str(path), + "PATH": "/usr/bin:/bin:/usr/local/bin", + } + subprocess.run(["git", "init", "-q", "-b", "main"], cwd=path, check=True, env=env) + subprocess.run(["git", "config", "user.email", "test@shadowfrog.invalid"], + cwd=path, check=True, env=env) + subprocess.run(["git", "config", "user.name", "ShadowFrog Test"], + cwd=path, check=True, env=env) + subprocess.run(["git", "config", "commit.gpgsign", "false"], + cwd=path, check=True, env=env) + if commit_all: + subprocess.run(["git", "add", "-A"], cwd=path, check=True, env=env) + subprocess.run(["git", "commit", "-q", "--allow-empty", "-m", "test initial"], + cwd=path, check=True, env=env) diff --git a/tests/hooks/__init__.py b/tests/hooks/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/hooks/test_check_hook_failopen.py b/tests/hooks/test_check_hook_failopen.py new file mode 100644 index 0000000..178260e --- /dev/null +++ b/tests/hooks/test_check_hook_failopen.py @@ -0,0 +1,277 @@ +"""Tests for hook-templates/check-hook-failopen.py — the static guard that +enforces the fail-open contract on advisory hook scripts. + +Every known regression vector must be detected by this checker. These tests +are the unit-level mirror of the adversarial probe used to develop it.""" +import subprocess +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent +CHECKER = REPO_ROOT / "hook-templates" / "check-hook-failopen.py" + + +def _run(tmp_path: Path, body: str) -> tuple[int, str]: + """Write `body` to a hook file under tmp_path and run the checker on it.""" + p = tmp_path / "hook.sh" + p.write_text(body) + r = subprocess.run( + ["python3", str(CHECKER), str(p)], + capture_output=True, text=True, timeout=10, + ) + return r.returncode, r.stdout + r.stderr + + +# Good hook patterns that must PASS the checker +GOOD_CASES = [ + pytest.param("""#!/usr/bin/env bash +trap 'exit 0' EXIT +trap 'exit 0' TERM HUP INT +INPUT=$(cat) +echo "ok" +""", id="canonical-pattern"), + pytest.param("""#!/usr/bin/env bash +trap 'exit 0' EXIT TERM HUP INT +echo "ok" +""", id="combined-trap-form"), + pytest.param("""#!/usr/bin/env bash +trap 'exit 0' EXIT +trap 'exit 0' TERM HUP INT +X=$(python3 - <<'PYEOF' +import subprocess +subprocess.run(["git", "diff"], timeout=1.0) +PYEOF +) +echo "ok" +""", id="git-inside-python-heredoc-allowed"), +] + + +@pytest.mark.parametrize("body", GOOD_CASES) +def test_checker_passes_good_hooks(tmp_path, body): + rc, output = _run(tmp_path, body) + assert rc == 0, f"checker incorrectly flagged good hook:\n{output}" + + +# Bad hook patterns that must FAIL the checker — one for every known +# regression vector. +BAD_CASES = [ + pytest.param("set -e", "set -e", + id="short-form-set-e"), + pytest.param("set -euo pipefail (original v1.0.57 bug)", + "set -euo pipefail", id="set-euo-pipefail-the-original-bug"), + pytest.param("set -o errexit (long-form bypass)", + "set -o errexit", id="long-form-errexit"), + pytest.param("set -o nounset (long-form bypass)", + "set -o nounset", id="long-form-nounset"), + pytest.param("set -o pipefail (long-form bypass)", + "set -o pipefail", id="long-form-pipefail"), + pytest.param("set -o errtrace (long-form bypass)", + "set -o errtrace", id="long-form-errtrace"), + pytest.param("source ./helpers.sh", "source ./helpers.sh", + id="source-external-file"), + pytest.param(". ./helpers.sh", ". ./helpers.sh", + id="dot-source-external-file"), + pytest.param("unbounded git rev-parse (line-94 bug class)", + "REPO_ROOT=$(git rev-parse --show-toplevel)", + id="unbounded-git-rev-parse"), + pytest.param("unbounded git diff", + "git diff HEAD~1 HEAD", + id="unbounded-git-diff"), +] + + +@pytest.mark.parametrize("description,injection", BAD_CASES) +def test_checker_flags_bad_hooks(tmp_path, description, injection): + body = f"""#!/usr/bin/env bash +trap 'exit 0' EXIT +trap 'exit 0' TERM HUP INT +{injection} +echo "fail" +""" + rc, output = _run(tmp_path, body) + assert rc != 0, ( + f"checker FAILED to flag known regression '{description}'.\n" + f"This injection should have been detected:\n {injection}\n" + f"Checker output:\n{output}" + ) + + +def test_checker_rejects_missing_exit_trap(tmp_path): + body = """#!/usr/bin/env bash +trap 'exit 0' TERM HUP INT +echo "missing EXIT" +""" + rc, output = _run(tmp_path, body) + assert rc != 0 + assert "EXIT" in output + + +def test_checker_rejects_missing_term_trap(tmp_path): + body = """#!/usr/bin/env bash +trap 'exit 0' EXIT +echo "missing TERM - SIGTERM returns 143" +""" + rc, output = _run(tmp_path, body) + assert rc != 0 + assert "TERM" in output + + +def test_checker_rejects_comment_masquerading_as_trap(tmp_path): + """A comment containing 'trap exit 0 EXIT' must not satisfy the requirement.""" + body = """#!/usr/bin/env bash +# This script does have: trap 'exit 0' EXIT TERM HUP INT +echo "comment masquerade" +""" + rc, output = _run(tmp_path, body) + assert rc != 0 + + +def test_checker_runs_against_actual_hook_scripts(): + """Sanity: the current production hook scripts must satisfy the contract.""" + hooks = sorted((REPO_ROOT / "hook-templates" / "scripts").glob("*.sh")) + assert hooks, "no hook scripts found" + r = subprocess.run( + ["python3", str(CHECKER), *map(str, hooks)], + capture_output=True, text=True, timeout=10, + ) + assert r.returncode == 0, ( + f"Production hook scripts FAIL the fail-open checker — this PR broke " + f"the contract.\nCheckpoint:\n{r.stdout}\n{r.stderr}" + ) + + +# --------------------------------------------------------------------------- +# CI guard evasion + raw-python3 detection +# Each known bypass pattern gets a dedicated regression test. If a future +# regex tweak breaks one of these, the failure message says exactly which +# bypass slipped through. +# --------------------------------------------------------------------------- + +# Evasion patterns that previously bypassed `FORBIDDEN_SET_RE`. Each must +# FAIL the checker (strict mode flag detected anywhere in the body). +SET_E_EVASION_CASES = [ + pytest.param("[[ True ]] && set -e", + id="evasion-combinator-and"), + pytest.param("true || set -e", + id="evasion-combinator-or"), + pytest.param("true; set -e", + id="evasion-semicolon"), + pytest.param("set \\\n -e", + id="evasion-line-continuation"), + pytest.param("eval 'set -e'", + id="evasion-eval-literal"), + pytest.param("eval \"set -o pipefail\"", + id="evasion-eval-long-form"), + pytest.param("if true; then set -euo pipefail; fi", + id="evasion-conditional-block"), + pytest.param("{ set -e; true; }", + id="evasion-group-command"), +] + + +@pytest.mark.parametrize("injection", SET_E_EVASION_CASES) +def test_checker_flags_set_e_evasions(tmp_path, injection): + """These bypass the original `^\\s*set\\s+` anchor. Substring scanning + + line-continuation joining must catch each one.""" + body = f"""#!/usr/bin/env bash +trap 'exit 0' EXIT +trap 'exit 0' TERM HUP INT +{injection} +echo "should be flagged" +""" + rc, output = _run(tmp_path, body) + assert rc != 0, ( + f"checker MISSED set -e evasion:\n{injection!r}\nOutput:\n{output}" + ) + + +# Signal-alias cases — these must PASS the checker because `SIGTERM` and +# numeric `15` are bash-equivalent to `TERM`. +SIGNAL_ALIAS_GOOD_CASES = [ + pytest.param("""#!/usr/bin/env bash +trap 'exit 0' EXIT +trap 'exit 0' SIGTERM SIGHUP SIGINT +echo "SIGTERM alias for TERM" +""", id="sigterm-spelled-out"), + pytest.param("""#!/usr/bin/env bash +trap 'exit 0' EXIT +trap 'exit 0' 15 1 2 +echo "numeric signal codes (15=TERM, 1=HUP, 2=INT)" +""", id="numeric-signal-codes"), + pytest.param("""#!/usr/bin/env bash +trap 'exit 0' EXIT SIGTERM +echo "combined EXIT and SIGTERM in one trap" +""", id="combined-exit-sigterm"), +] + + +@pytest.mark.parametrize("body", SIGNAL_ALIAS_GOOD_CASES) +def test_checker_accepts_signal_aliases(tmp_path, body): + """`SIGTERM` and numeric `15` must be recognized as TERM-equivalent.""" + rc, output = _run(tmp_path, body) + assert rc == 0, ( + f"checker false-rejected valid signal alias:\n{body}\nOutput:\n{output}" + ) + + +# Raw python3 invocation cases — these must FAIL the checker. +RAW_PYTHON_BAD_CASES = [ + pytest.param("python3 my_script.py", + id="raw-python3-script-file"), + pytest.param("python3 -m mymodule", + id="raw-python3-module"), + pytest.param("python3 -u long_running.py", + id="raw-python3-unbuffered-script"), + pytest.param("OUTPUT=$(python3 my_script.py)", + id="raw-python3-script-captured"), +] + + +@pytest.mark.parametrize("injection", RAW_PYTHON_BAD_CASES) +def test_checker_flags_raw_python3_scripts(tmp_path, injection): + """Bash-level `python3 path.py` is unbounded — same bug class as + `git rev-parse` hang. Must be detected.""" + body = f"""#!/usr/bin/env bash +trap 'exit 0' EXIT +trap 'exit 0' TERM HUP INT +{injection} +echo "should be flagged" +""" + rc, output = _run(tmp_path, body) + assert rc != 0, ( + f"checker MISSED raw python3 script invocation:\n{injection!r}\n" + f"Output:\n{output}" + ) + + +# Raw python3 SAFE forms — these must PASS (no false-positives). +RAW_PYTHON_OK_CASES = [ + pytest.param("""#!/usr/bin/env bash +trap 'exit 0' EXIT +trap 'exit 0' TERM HUP INT +TOOL=$(python3 -c "import json,sys; print('ok')" 2>/dev/null || echo "") +echo "short -c literal is allowed" +""", id="python3-dash-c-short-literal"), + pytest.param("""#!/usr/bin/env bash +trap 'exit 0' EXIT +trap 'exit 0' TERM HUP INT +OUT=$(python3 - <<'PYEOF' 2>/dev/null || echo "" +import subprocess +subprocess.run(["echo","hello"], timeout=1.0) +PYEOF +) +echo "heredoc with bounded internal logic is allowed" +""", id="python3-stdin-heredoc"), +] + + +@pytest.mark.parametrize("body", RAW_PYTHON_OK_CASES) +def test_checker_accepts_safe_python3_forms(tmp_path, body): + """`python3 -c` short literals and `python3 - <<HEREDOC` patterns must + NOT be flagged.""" + rc, output = _run(tmp_path, body) + assert rc == 0, ( + f"checker false-positive on safe python3 form:\n{body}\nOutput:\n{output}" + ) diff --git a/tests/hooks/test_check_init_sh.py b/tests/hooks/test_check_init_sh.py new file mode 100644 index 0000000..34a52c9 --- /dev/null +++ b/tests/hooks/test_check_init_sh.py @@ -0,0 +1,267 @@ +"""Tests for hook-templates/scripts/shadow-frog-check-init.sh — SessionStart hook. + +Exercises: no-shadow guidance, fresh shadow status, B2 sentinel handling, +staleness detection. +""" +import json +import os +import subprocess +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent +HOOK_SCRIPT = REPO_ROOT / "hook-templates" / "scripts" / "shadow-frog-check-init.sh" + + +def _base_env(cwd: Path, extras: dict | None = None) -> dict: + env = { + "PATH": os.environ.get("PATH", "/usr/bin:/bin:/usr/local/bin"), + "HOME": str(cwd), + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "LANG": "en_US.UTF-8", + } + if extras: + env.update(extras) + return env + + +def run_hook(cwd: Path, env_extra: dict | None = None) -> subprocess.CompletedProcess: + """Run the check-init hook (stdin is ignored but must exist).""" + env = _base_env(cwd, env_extra) + return subprocess.run( + ["bash", str(HOOK_SCRIPT)], + input="{}", + capture_output=True, + text=True, + cwd=cwd, + env=env, + ) + + +@pytest.mark.slow +@pytest.mark.integration +class TestCheckInitDualOutputFormat: + """Output carries both Copilot (top-level) and Claude Code (nested) shapes.""" + + def test_no_shadow_emits_both_shapes(self, tmp_git_repo): + result = run_hook(cwd=tmp_git_repo) + data = json.loads(result.stdout) + assert "additionalContext" in data + hso = data.get("hookSpecificOutput") + assert hso is not None + assert hso["hookEventName"] == "SessionStart" + assert hso["additionalContext"] == data["additionalContext"] + + def test_fresh_shadow_emits_both_shapes(self, coupon_demo): + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, + ).stdout.strip() + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = head + state_file.write_text(json.dumps(state)) + + result = run_hook(cwd=coupon_demo) + assert result.returncode == 0 + data = json.loads(result.stdout) + hso = data.get("hookSpecificOutput") + assert hso is not None + assert hso["hookEventName"] == "SessionStart" + assert hso["additionalContext"] == data["additionalContext"] + + +@pytest.mark.slow +@pytest.mark.integration +class TestCheckInitNoShadow: + """No .shadow/ → emits init guidance.""" + + def test_no_shadow_emits_init_guidance(self, tmp_git_repo): + result = run_hook(cwd=tmp_git_repo) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "additionalContext" in data + assert "shadow-frog-init" in data["additionalContext"] + + +@pytest.mark.slow +@pytest.mark.integration +class TestCheckInitFreshShadow: + """Fresh .shadow/ with current commit → no staleness warning.""" + + def test_fresh_shadow_no_staleness(self, coupon_demo): + # Update state.json to point at current HEAD + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, + ).stdout.strip() + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = head + state_file.write_text(json.dumps(state)) + + result = run_hook(cwd=coupon_demo) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "additionalContext" in data + assert "Shadow loaded" in data["additionalContext"] + assert "WARNING" not in data["additionalContext"] + + +@pytest.mark.slow +@pytest.mark.integration +class TestCheckInitSentinels: + """B2: Sentinel values for last_commit don't crash the hook.""" + + def test_last_commit_none_sentinel(self, coupon_demo): + """state.json with last_commit='none' — no crash, no false staleness.""" + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = "none" + state_file.write_text(json.dumps(state)) + + result = run_hook(cwd=coupon_demo) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "additionalContext" in data + # "none" is not a valid git ref, so git rev-parse --verify fails, + # which means the staleness check is skipped entirely. + assert "WARNING" not in data["additionalContext"] + + def test_missing_last_commit_field(self, coupon_demo): + """state.json without last_commit key — no crash.""" + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state.pop("last_commit", None) + state_file.write_text(json.dumps(state)) + + result = run_hook(cwd=coupon_demo) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "additionalContext" in data + # Default is 'none' which won't verify, so no false warning + assert "WARNING" not in data["additionalContext"] + + +@pytest.mark.slow +@pytest.mark.integration +class TestCheckInitStaleness: + """Stale shadow → emits staleness warning.""" + + def test_stale_shadow_emits_warning(self, coupon_demo): + # Get the initial commit + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, + ).stdout.strip() + + # Set state to current HEAD + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = head + state_file.write_text(json.dumps(state)) + + # Make new commits with file changes to create staleness + env = _base_env(coupon_demo) + for i in range(3): + (coupon_demo / f"file{i}.py").write_text(f"# file {i}\n") + subprocess.run(["git", "add", "-A"], cwd=coupon_demo, check=True, env=env) + subprocess.run( + ["git", "commit", "-q", "-m", "add files"], + cwd=coupon_demo, check=True, env=env, + ) + + result = run_hook(cwd=coupon_demo) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "WARNING" in data["additionalContext"] + assert "shadow-frog-update" in data["additionalContext"] + + def test_shadow_only_commit_no_staleness(self, coupon_demo): + """Committing only .shadow/ changes must NOT trigger a staleness + warning — otherwise git-tracked shadows warn forever after every + update commit.""" + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, + ).stdout.strip() + + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = head + state_file.write_text(json.dumps(state)) + + # Advance HEAD with a commit that touches ONLY .shadow/ + env = _base_env(coupon_demo) + (coupon_demo / ".shadow" / "newshadow.py.md").write_text("# x\n") + subprocess.run(["git", "add", "-A"], cwd=coupon_demo, check=True, env=env) + subprocess.run( + ["git", "commit", "-q", "-m", "shadow only"], + cwd=coupon_demo, check=True, env=env, + ) + + result = run_hook(cwd=coupon_demo) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "WARNING" not in data["additionalContext"] + + +def _make_failing_git_stub(stub_dir: Path, fail_subcommand: str = "diff") -> Path: + """Create a `git` shim that fails for one subcommand, passing the rest to + the real git. Used to prove the hook stays robust when git misbehaves.""" + import shutil + stub_dir.mkdir(parents=True, exist_ok=True) + real_git = shutil.which("git") + shim = stub_dir / "git" + shim.write_text( + "#!/bin/bash\n" + f'if [ "$1" = "{fail_subcommand}" ]; then\n' + ' echo "warning: simulated git failure" >&2\n' + " exit 1\n" + "fi\n" + f'exec "{real_git}" "$@"\n' + ) + shim.chmod(0o755) + return shim + + +@pytest.mark.slow +@pytest.mark.integration +class TestCheckInitRobustness: + """sessionStart is fail-open by contract, but must also stay quiet and + clean (exit 0, no stderr leak) when state.json is missing or git hiccups.""" + + def test_missing_state_json_exits_zero_no_stderr(self, coupon_demo): + (coupon_demo / ".shadow" / "_meta" / "state.json").unlink() + result = run_hook(cwd=coupon_demo) + assert result.returncode == 0 + assert result.stderr.strip() == "", f"unexpected stderr: {result.stderr!r}" + data = json.loads(result.stdout) + assert "additionalContext" in data + + def test_git_diff_failure_exits_zero(self, coupon_demo, tmp_path): + # Make the shadow stale so the staleness/git-diff path runs. + env = _base_env(coupon_demo) + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, env=env, + ).stdout.strip() + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = head + state_file.write_text(json.dumps(state)) + (coupon_demo / "newcode.py").write_text("x = 1\n") + subprocess.run(["git", "add", "-A"], cwd=coupon_demo, check=True, env=env) + subprocess.run(["git", "commit", "-q", "-m", "advance"], + cwd=coupon_demo, check=True, env=env) + + stub = tmp_path / "stubbin" + _make_failing_git_stub(stub, "diff") + result = run_hook( + cwd=coupon_demo, + env_extra={"PATH": f"{stub}:{os.environ.get('PATH', '')}"}, + ) + assert result.returncode == 0, f"stderr={result.stderr}" + json.loads(result.stdout) diff --git a/tests/hooks/test_hook_fault_injection.py b/tests/hooks/test_hook_fault_injection.py new file mode 100644 index 0000000..52b1398 --- /dev/null +++ b/tests/hooks/test_hook_fault_injection.py @@ -0,0 +1,585 @@ +"""Fault-injection matrix for shadow-frog-pre-tool.sh and +shadow-frog-check-init.sh. + +Both hooks MUST be fail-open (exit 0, no stderr, valid JSON or empty stdout) +under every plausible perturbation, because Copilot CLI >= 1.0.57 denies the +tool call if a preToolUse command hook exits non-zero. + +This file systematically perturbs four axes and asserts the contract holds: + + 1. Stubbed binary — git (fail/hang/various subcommands), python3, ps, + mkdir (read-only), tr + 2. Input payload — malformed JSON, missing fields, huge, non-ASCII paths, + shell-metacharacter paths, deeply nested + 3. CWD state — non-git, no commits yet, missing _meta, corrupt + state.json, .shadow read-only + 4. Env state — C locale, empty PPID, missing SHADOWFROG_TMP_DIR + +Each parametrized case asserts: + result.returncode == 0 + result.stderr.strip() == '' (no shell or tool noise leaked) + valid JSON OR empty stdout (Copilot treats empty as default-allow) + elapsed wall-clock < HOOK_BUDGET_SEC (must beat the runner's timeoutSec + or Copilot SIGTERM kills us) + +When this file fails, the failure name pinpoints which axis × value broke the +contract, making future regressions easy to triage. + +Why the matrix tests create `.shadow/<target>.md` (per Opus-4.7/4.8 review): +the pre-tool viewer-discovery branch only executes when a shadow exists for +the file being edited. Without seeding that shadow file, the entire branch +(including the git/viewer subprocesses that were the original bug class) +goes unexercised and bugs hide in plain sight. +""" +import json +import os +import shutil +import stat +import subprocess +import time +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent +PRE_TOOL_HOOK = REPO_ROOT / "hook-templates" / "scripts" / "shadow-frog-pre-tool.sh" +CHECK_INIT_HOOK = REPO_ROOT / "hook-templates" / "scripts" / "shadow-frog-check-init.sh" + +# Configured hook timeout in hook-templates/shadow-frog-hooks.json. Tests must complete +# below this or production would have been killed by the runner. +HOOK_BUDGET_SEC = 5.0 +# Soft wall-clock cap. A real unbounded-call regression hits the 10s subprocess +# timeout in _run (raising TimeoutExpired), so this assertion exists to catch +# *creeping* slowdowns — e.g., a new ~2s subprocess added without a timeout +# bound. The cap is intentionally generous (well above the runner's 5s +# timeoutSec) because shared CI runners under parallel pytest load routinely +# show 3-5x Python cold-start latency vs a quiet developer machine; locally +# the hook completes in ~1s, on busy CI it has been observed at ~6.3s while +# doing identical bounded work. We accept that latency variance and rely on +# (a) the 10s hard subprocess timeout for true hangs, and (b) the CI guard +# (hook-templates/check-hook-failopen.py) for static unbounded-call +# detection. The bounded-work budget itself is ~3.5s; doubling that for CI +# headroom yields the 7.0s soft cap below. +WALL_CLOCK_LIMIT_SEC = 7.0 + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +def _base_env(cwd: Path, extras: dict | None = None) -> dict: + env = { + "PATH": os.environ.get("PATH", "/usr/bin:/bin:/usr/local/bin"), + "HOME": str(cwd), + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "LANG": "en_US.UTF-8", + } + if extras: + env.update(extras) + return env + + +def _run(hook: Path, cwd: Path, stdin: str, env_extra: dict | None = None, + timeout: float = 10.0) -> tuple[subprocess.CompletedProcess, float]: + """Run hook and return (completed_process, elapsed_seconds). + + Per-test dedup isolation: without this, every parametrized matrix cell + shares the same pytest PPID AND parent start-time, which collide on + `${TMP_ROOT}/shadowfrog-hook-<PPID>-<start>/`. The first cell to touch + `a.py` creates `a.py.injected`; every subsequent cell finds the dedup + file present and SKIPS the entire viewer subprocess (pre-tool.sh + `[ ! -f "$DEDUP_FILE" ]` guard). Result: the viewer-branch git rev-parse + + viewer process are exercised exactly ONCE per pytest run, masking + unbounded-call regressions (reproduced: replacing the bounded + `_git(['rev-parse',...], 0.5)` with an unbounded call passed 70 cells + even though production would hang for 31s on a slow `git rev-parse`). + + The fix: give each test a fresh SHADOWFROG_TMP_DIR rooted under its own + `cwd` so dedup state cannot bleed across cells. Tests that explicitly + set their own SHADOWFROG_TMP_DIR (e.g., the `*-tmpdir` env cells) take + precedence. + """ + extras = dict(env_extra or {}) + if "SHADOWFROG_TMP_DIR" not in extras: + # Use a stable subdir of cwd so multiple invocations within ONE + # test (e.g., dedup tests) share state, but cross-test invocations + # do not (each test gets a unique cwd from pytest's tmp_path). + extras["SHADOWFROG_TMP_DIR"] = str(cwd / "_sf_dedup") + t0 = time.perf_counter() + cp = subprocess.run( + ["bash", str(hook)], + input=stdin, capture_output=True, text=True, + cwd=cwd, env=_base_env(cwd, extras), timeout=timeout, + ) + return cp, time.perf_counter() - t0 + + +def _assert_fail_open(result: subprocess.CompletedProcess, case: str, + elapsed: float | None = None) -> None: + """The contract every cell must satisfy. Empty stdout is acceptable + (Copilot treats it as default-allow); non-empty must be valid JSON.""" + assert result.returncode == 0, ( + f"[{case}] HOOK DENIED TOOL (exit={result.returncode})\n" + f"stderr={result.stderr!r}\nstdout={result.stdout!r}" + ) + assert result.stderr.strip() == "", ( + f"[{case}] stderr leak (would be visible to user):\n{result.stderr}" + ) + if elapsed is not None: + assert elapsed < WALL_CLOCK_LIMIT_SEC, ( + f"[{case}] WALL CLOCK BUDGET EXCEEDED: {elapsed:.2f}s " + f">= {WALL_CLOCK_LIMIT_SEC}s. Runner's timeoutSec={HOOK_BUDGET_SEC}s " + f"would have SIGTERM-killed us → tool DENY in production." + ) + out = result.stdout.strip() + if out: + try: + data = json.loads(out) + except json.JSONDecodeError as e: + pytest.fail(f"[{case}] invalid JSON output: {e}\nstdout={out!r}") + # If JSON is produced, it must carry advisory context (the hook's job) + assert "additionalContext" in data, ( + f"[{case}] JSON missing additionalContext: {data!r}" + ) + + +def _make_binary_stub(stub_dir: Path, name: str, body: str) -> Path: + """Create an executable shim at stub_dir/<name>. Prepend stub_dir to PATH + to activate it.""" + stub_dir.mkdir(parents=True, exist_ok=True) + shim = stub_dir / name + shim.write_text("#!/bin/bash\n" + body) + shim.chmod(0o755) + return shim + + +def _git_passthrough_with_failure(stub_dir: Path, fail_pattern: str, + exit_code: int = 1, + stderr_msg: str = "warning: simulated") -> None: + """Stub git to fail when the first arg matches `fail_pattern` + (shell-glob), passthrough otherwise. Use 'ALL' to fail every call.""" + real = shutil.which("git") + if fail_pattern == "ALL": + body = f'echo "{stderr_msg}" >&2\nexit {exit_code}\n' + else: + body = ( + f'case "$1" in\n' + f' {fail_pattern}) echo "{stderr_msg}" >&2; exit {exit_code} ;;\n' + f'esac\n' + f'exec "{real}" "$@"\n' + ) + _make_binary_stub(stub_dir, "git", body) + + +def _git_hanging_stub(stub_dir: Path, hang_pattern: str, sleep_secs: int = 30) -> None: + """Stub git to hang for `hang_pattern` calls (simulates a locked repo / + network filesystem). Other calls passthrough. This proves our subprocess + timeouts catch the hang well below the hook's 5s budget.""" + real = shutil.which("git") + body = ( + f'case "$1" in\n' + f' {hang_pattern}) sleep {sleep_secs}; exit 0 ;;\n' + f'esac\n' + f'exec "{real}" "$@"\n' + ) + _make_binary_stub(stub_dir, "git", body) + + +def _init_minimal_shadow(cwd: Path, last_commit: str = "deadbeef0000", + state_extra: dict | None = None, + seed_target: str | None = "a.py") -> None: + """Create a minimal .shadow/ layout in cwd. + + When `seed_target` is non-empty, also creates .shadow/<seed_target>.md + with a bug-labeled discovery so the pre-tool viewer-discovery branch + (which only runs when the per-file shadow exists) is reachable in + binary-fault and cwd-state tests. The original matrix omitted this + seeding, leaving the viewer branch — including the very git rev-parse + call that was a latent unbounded-hang vector — completely uncovered. + """ + meta = cwd / ".shadow" / "_meta" + meta.mkdir(parents=True, exist_ok=True) + state = {"version": 1, "last_commit": last_commit, + "total_files": 0, "total_discoveries": 0} + if state_extra: + state.update(state_extra) + (meta / "state.json").write_text(json.dumps(state)) + if seed_target: + shadow_md = cwd / ".shadow" / f"{seed_target}.md" + shadow_md.parent.mkdir(parents=True, exist_ok=True) + # Use a `bug`-labeled discovery so the viewer's `--top-labels bug,security` + # filter actually returns content (forcing the viewer subprocess to run + # to completion under fault injection, not exit early). + shadow_md.write_text( + f"# {seed_target}\n\n" + f"## `dummy_function`\n\n" + f"- Returns None on empty input instead of raising — historical bug.\n" + f" _(verified, source: exploration, labels: [bug])_\n" + ) + + +def _make_git_repo(cwd: Path, with_commit: bool = True, + advance_head: bool = False) -> str: + """Initialize a real git repo at cwd; return HEAD SHA (or '' if no commit).""" + env = _base_env(cwd) + subprocess.run(["git", "init", "-q", "-b", "main"], cwd=cwd, check=True, env=env) + subprocess.run(["git", "config", "user.email", "t@t.co"], cwd=cwd, check=True, env=env) + subprocess.run(["git", "config", "user.name", "t"], cwd=cwd, check=True, env=env) + if not with_commit: + return "" + (cwd / "a.py").write_text("x = 1\n") + subprocess.run(["git", "add", "-A"], cwd=cwd, check=True, env=env) + subprocess.run(["git", "commit", "-qm", "init"], cwd=cwd, check=True, env=env) + head = subprocess.run(["git", "rev-parse", "HEAD"], cwd=cwd, check=True, + capture_output=True, text=True, env=env).stdout.strip() + if advance_head: + (cwd / "b.py").write_text("y = 2\n") + subprocess.run(["git", "add", "-A"], cwd=cwd, check=True, env=env) + subprocess.run(["git", "commit", "-qm", "advance"], cwd=cwd, check=True, env=env) + return head + + +# --------------------------------------------------------------------------- +# Axis 1: Stubbed binary failures (preToolUse + sessionStart) +# --------------------------------------------------------------------------- + +# Each tuple is (binary_name, body, scenario_id). Each test runs the hook in a +# real git repo with a stale shadow so the staleness/git path executes. +BINARY_FAULTS = [ + # git failures by subcommand + ("git-fail-all", lambda d: _git_passthrough_with_failure(d, "ALL")), + ("git-fail-diff", lambda d: _git_passthrough_with_failure(d, "diff")), + ("git-fail-rev-parse", lambda d: _git_passthrough_with_failure(d, "rev-parse")), + ("git-fail-show-toplevel", lambda d: _git_passthrough_with_failure(d, "rev-parse", stderr_msg="not a repo")), + ("git-noisy-warnings", lambda d: _git_passthrough_with_failure(d, "diff", exit_code=128, stderr_msg="fatal: bad ref")), + # git hangs (proves bounded timeouts work). The "rev-parse" hang exercises + # both the staleness-check rev-parse AND — critically — the viewer-branch + # rev-parse --show-toplevel at pre-tool.sh that was previously unbounded + # (caught by Opus-4.7/4.8 review, reproduced as a 31s hang in production). + # The seed_target=.shadow/a.py.md added by _init_minimal_shadow now ensures + # the viewer branch actually executes under this fault. + ("git-hang-diff", lambda d: _git_hanging_stub(d, "diff", sleep_secs=30)), + ("git-hang-rev-parse", lambda d: _git_hanging_stub(d, "rev-parse", sleep_secs=30)), + # missing python3 (rare but possible in stripped containers) + ("python3-missing", lambda d: _make_binary_stub(d, "python3", "exit 127\n")), + # missing ps (some sandboxed containers) + ("ps-missing", lambda d: _make_binary_stub(d, "ps", "exit 127\n")), + # tr complaining on locale (macOS quirk) + ("tr-fail", lambda d: _make_binary_stub(d, "tr", "exit 1\n")), +] + + +@pytest.mark.slow +@pytest.mark.integration +@pytest.mark.parametrize("hook", [PRE_TOOL_HOOK, CHECK_INIT_HOOK], + ids=["pre-tool", "check-init"]) +@pytest.mark.parametrize("scenario_id,stub_factory", BINARY_FAULTS, + ids=[s[0] for s in BINARY_FAULTS]) +def test_binary_fault_injection(tmp_path, hook, scenario_id, stub_factory): + """For every (hook × stubbed-binary-failure) combination, the hook must + still exit 0 with no stderr and either empty stdout or valid JSON.""" + repo = tmp_path / "repo" + repo.mkdir() + head = _make_git_repo(repo, with_commit=True, advance_head=True) + # Set state.json one commit behind so the staleness path triggers + behind = subprocess.run( + ["git", "rev-parse", "HEAD~1"], cwd=repo, + capture_output=True, text=True, check=True, env=_base_env(repo), + ).stdout.strip() + _init_minimal_shadow(repo, last_commit=behind) + # Regression guard: the viewer-discovery branch only fires when the + # per-file shadow exists. If a future change ever defaults seed_target + # to None, the entire viewer branch (git rev-parse, viewer subprocess) + # silently goes uncovered — exactly the blind spot that hid the + # original unbounded-git regression. Asserting + # here makes that drift impossible to merge. + assert (repo / ".shadow" / "a.py.md").exists(), ( + "test setup regression: _init_minimal_shadow did not seed the shadow " + "file. The viewer-discovery branch will not execute under fault " + "injection. Check seed_target default in _init_minimal_shadow." + ) + + stub_dir = tmp_path / "stubbin" + stub_factory(stub_dir) + env = {"PATH": f"{stub_dir}:{os.environ.get('PATH', '')}"} + + payload = json.dumps({"toolName": "edit", + "toolInput": {"file_path": "a.py"}}) + # Bound test execution at 10s — proves the hook itself stays within budget + # even when stubs hang for 30s. _assert_fail_open additionally asserts + # the wall clock is under WALL_CLOCK_LIMIT_SEC, which catches creeping + # slowdowns while tolerating CI cold-start latency. + result, elapsed = _run(hook, repo, stdin=payload, env_extra=env, timeout=10.0) + _assert_fail_open(result, f"{hook.name} | {scenario_id}", elapsed=elapsed) + + +# --------------------------------------------------------------------------- +# Axis 2: Malicious / malformed input payloads (preToolUse only — check-init +# ignores stdin) +# --------------------------------------------------------------------------- + +PAYLOAD_FAULTS = [ + ("empty", ""), + ("not-json", "this is not json at all"), + ("empty-object", "{}"), + ("null", "null"), + ("array-not-object", '[1, 2, 3]'), + ("missing-toolName", '{"toolInput": {"file_path": "x.py"}}'), + ("missing-toolInput", '{"toolName": "edit"}'), + ("wrong-type-toolInput", '{"toolName": "edit", "toolInput": "not-an-object"}'), + ("wrong-type-toolName", '{"toolName": 123, "toolInput": {}}'), + ("huge-payload", '{"toolName": "edit", "toolInput": {"file_path": "' + "A" * 100_000 + '.py"}}'), + ("non-ascii-path", '{"toolName": "edit", "toolInput": {"file_path": "café/日本語/файл.py"}}'), + ("shell-meta-path", '{"toolName": "edit", "toolInput": {"file_path": "x.py; rm -rf /; #"}}'), + ("backtick-path", '{"toolName": "edit", "toolInput": {"file_path": "`whoami`.py"}}'), + ("quote-injection", '{"toolName": "edit", "toolInput": {"file_path": "a\\"); import os; os.system(\\"touch /tmp/PWNED_FUZZ\\"); #"}}'), + ("newline-in-path", '{"toolName": "edit", "toolInput": {"file_path": "x\\ny.py"}}'), + ("null-byte-payload", '{"toolName": "edit", "toolInput": {"file_path": "x\\u0000y.py"}}'), + ("deeply-nested", json.dumps({"toolName": "edit", "toolInput": {"path": "x.py", "deep": {"a": {"b": {"c": {"d": {"e": {"f": "g"}}}}}}}})), + ("unknown-tool", '{"toolName": "frobnicate", "toolInput": {"file_path": "x.py"}}'), + ("snake-case-fields", '{"tool_name": "edit", "tool_input": {"file_path": "x.py"}}'), + ("toolArgs-fallback", '{"toolName": "edit", "toolArgs": {"path": "x.py"}}'), +] + + +@pytest.mark.slow +@pytest.mark.integration +@pytest.mark.parametrize("scenario_id,payload", PAYLOAD_FAULTS, + ids=[s[0] for s in PAYLOAD_FAULTS]) +def test_pretool_payload_fault_injection(coupon_demo, scenario_id, payload, + tmp_path): + """preToolUse hook must survive any plausible / malicious JSON payload + without denying the tool or executing the payload.""" + # Align state.json with HEAD so the git/staleness path runs cleanly + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, env=_base_env(coupon_demo), + ).stdout.strip() + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = head + state_file.write_text(json.dumps(state)) + + # Inject a test-scoped sentinel path. Replace the hardcoded /tmp/PWNED_FUZZ + # marker (which made tests flaky across runs and could false-positive when + # an unrelated leftover existed) with a tmp_path-scoped file. + pwn_marker = tmp_path / "PWNED_FUZZ" + scoped_payload = payload.replace("/tmp/PWNED_FUZZ", str(pwn_marker)) + result, elapsed = _run(PRE_TOOL_HOOK, coupon_demo, stdin=scoped_payload, + timeout=10.0) + _assert_fail_open(result, f"pre-tool | payload={scenario_id}", + elapsed=elapsed) + # Confirm no injection payload was executed + assert not pwn_marker.exists(), \ + f"[{scenario_id}] payload was EXECUTED — RCE!" + + +# --------------------------------------------------------------------------- +# Axis 3: Filesystem / CWD state perturbations +# --------------------------------------------------------------------------- + +CWD_FAULTS = [ + "non-git-dir", + "git-no-commits", + "missing-meta-dir", + "missing-state-json", + "state-json-empty", + "state-json-corrupt", + "state-json-wrong-type", + "state-json-no-last-commit", + "state-json-last-commit-not-a-sha", + "shadow-read-only", + "detached-head", + # Pathological-but-recoverable state.json variants that the hook MUST + # silently absorb without exiting non-zero. + "state-json-symlink-to-devnull", + "state-json-is-a-directory", + "state-json-huge-file", + "state-json-future-version", + "state-json-null-bytes", + "state-json-deeply-nested", +] + + +def _setup_cwd_state(tmp_path: Path, scenario: str) -> Path: + """Build a cwd that exhibits `scenario`. Returns the cwd path.""" + cwd = tmp_path / "cwd" + cwd.mkdir() + + if scenario == "non-git-dir": + _init_minimal_shadow(cwd) + return cwd + + if scenario == "git-no-commits": + _make_git_repo(cwd, with_commit=False) + _init_minimal_shadow(cwd) + return cwd + + _make_git_repo(cwd, with_commit=True) + + if scenario == "missing-meta-dir": + (cwd / ".shadow").mkdir() # exists but no _meta/ + return cwd + if scenario == "missing-state-json": + (cwd / ".shadow" / "_meta").mkdir(parents=True) + return cwd + if scenario == "state-json-empty": + meta = cwd / ".shadow" / "_meta"; meta.mkdir(parents=True) + (meta / "state.json").write_text("") + return cwd + if scenario == "state-json-corrupt": + meta = cwd / ".shadow" / "_meta"; meta.mkdir(parents=True) + (meta / "state.json").write_text("{this is not valid json") + return cwd + if scenario == "state-json-wrong-type": + meta = cwd / ".shadow" / "_meta"; meta.mkdir(parents=True) + (meta / "state.json").write_text('"a string, not an object"') + return cwd + if scenario == "state-json-no-last-commit": + meta = cwd / ".shadow" / "_meta"; meta.mkdir(parents=True) + (meta / "state.json").write_text('{"version": 1, "total_files": 0}') + return cwd + if scenario == "state-json-last-commit-not-a-sha": + _init_minimal_shadow(cwd, last_commit="totally not a sha 🐸") + return cwd + if scenario == "shadow-read-only": + # The hook only writes to SHADOWFROG_TMP_DIR (dedup markers); .shadow/ + # is read-only by design. Making state.json read-only doesn't actually + # exercise any failure path. Instead, make the dedup tmpdir target + # read-only — that's where a real write failure could occur (and + # where the hook must fail-open silently). + _init_minimal_shadow(cwd) + readonly_tmp = cwd / "readonly_tmp" + readonly_tmp.mkdir() + readonly_tmp.chmod(stat.S_IRUSR | stat.S_IXUSR) # r-x, no write + # Also keep state.json read-only as a defensive check that read-only + # state still works (it shouldn't — the hook only reads it). + (cwd / ".shadow" / "_meta" / "state.json").chmod(stat.S_IRUSR | stat.S_IRGRP | stat.S_IROTH) + return cwd + if scenario == "detached-head": + head = subprocess.run(["git", "rev-parse", "HEAD"], cwd=cwd, check=True, + capture_output=True, text=True, + env=_base_env(cwd)).stdout.strip() + subprocess.run(["git", "checkout", "-q", head], cwd=cwd, check=True, + env=_base_env(cwd)) + _init_minimal_shadow(cwd, last_commit=head) + return cwd + # ----- pathological state.json variants ----- + if scenario == "state-json-symlink-to-devnull": + # Adversarial: state.json is a symlink to /dev/null. Reading it + # returns empty → `json.load` raises → except absorbs → hook OK. + meta = cwd / ".shadow" / "_meta"; meta.mkdir(parents=True) + sj = meta / "state.json" + try: + sj.symlink_to("/dev/null") + except (OSError, NotImplementedError): + # If the platform can't make symlinks, fall back to empty file + # (same effective failure mode). + sj.write_text("") + return cwd + if scenario == "state-json-is-a-directory": + # Pathological: state.json is a DIRECTORY, not a file. `open()` + # raises IsADirectoryError → except absorbs. + meta = cwd / ".shadow" / "_meta"; meta.mkdir(parents=True) + (meta / "state.json").mkdir() + return cwd + if scenario == "state-json-huge-file": + # Defense-in-depth: an oversized state.json (~10 MB of valid JSON + # padding). Reading it is bounded by the python subprocess timeout + # in the staleness check. + meta = cwd / ".shadow" / "_meta"; meta.mkdir(parents=True) + big = {"version": 1, "last_commit": "deadbeef0000", + "padding": "x" * 10_000_000} + (meta / "state.json").write_text(json.dumps(big)) + return cwd + if scenario == "state-json-future-version": + # Forward-compat: version field with an unexpected schema marker. + # Hook should ignore unknown fields and read what it can. + _init_minimal_shadow(cwd, state_extra={"version": 9999, + "future_field": {"x": 1}}) + return cwd + if scenario == "state-json-null-bytes": + # Corruption: file contains embedded NUL bytes. `json.load` raises; + # except absorbs. + meta = cwd / ".shadow" / "_meta"; meta.mkdir(parents=True) + (meta / "state.json").write_bytes(b'{"last_commit":"abc\x00\x00def"}') + return cwd + if scenario == "state-json-deeply-nested": + # Pathological nesting that some JSON decoders might choke on. + # Python's json module accepts ~1000 levels by default, so this + # parses fine — but we want to verify the hook doesn't blow up + # building any data structures from it. + deep = {"version": 1, "last_commit": "abc"} + cur = deep + for _ in range(200): + cur["x"] = {} + cur = cur["x"] + meta = cwd / ".shadow" / "_meta"; meta.mkdir(parents=True) + (meta / "state.json").write_text(json.dumps(deep)) + return cwd + + raise ValueError(f"unknown scenario {scenario}") + + +@pytest.mark.slow +@pytest.mark.integration +@pytest.mark.parametrize("hook", [PRE_TOOL_HOOK, CHECK_INIT_HOOK], + ids=["pre-tool", "check-init"]) +@pytest.mark.parametrize("scenario", CWD_FAULTS) +def test_cwd_state_fault_injection(tmp_path, hook, scenario): + """Both hooks must survive any plausible filesystem/git state.""" + cwd = _setup_cwd_state(tmp_path, scenario) + payload = json.dumps({"toolName": "edit", + "toolInput": {"file_path": "a.py"}}) + result, elapsed = _run(hook, cwd, stdin=payload, timeout=10.0) + _assert_fail_open(result, f"{hook.name} | cwd={scenario}", elapsed=elapsed) + + +# --------------------------------------------------------------------------- +# Axis 4: Environment perturbations +# --------------------------------------------------------------------------- + +ENV_FAULTS = [ + ("C-locale", {"LANG": "C", "LC_ALL": "C"}), + ("posix-locale", {"LANG": "POSIX", "LC_ALL": "POSIX"}), + ("empty-tmpdir-override", {"SHADOWFROG_TMP_DIR": ""}), + ("nonexistent-tmpdir", {"SHADOWFROG_TMP_DIR": "/nonexistent/path/xyz"}), + # Cross-platform readonly-tmpdir cell: /proc is Linux-only and was + # silently skipped on macOS. Sentinel triggers a chmod 0500 tmp dir + # wired up by the test body so both platforms exercise the case. + ("readonly-tmpdir", {"__SF_TEST_USE_READONLY_TMPDIR__": "1"}), +] + + +@pytest.mark.slow +@pytest.mark.integration +@pytest.mark.parametrize("hook", [PRE_TOOL_HOOK, CHECK_INIT_HOOK], + ids=["pre-tool", "check-init"]) +@pytest.mark.parametrize("scenario,env_extra", ENV_FAULTS, + ids=[s[0] for s in ENV_FAULTS]) +def test_env_fault_injection(coupon_demo, hook, scenario, env_extra, tmp_path): + """Both hooks must survive constrained environments (C locale, hostile + tmpdir overrides).""" + # For the readonly-tmpdir cell, dynamically wire a read-only tmpdir. + # chmod 0500 (r-x, no write) on a tmp_path subdir works on both macOS + # and Linux — the previous /proc approach was Linux-only. + if "__SF_TEST_USE_READONLY_TMPDIR__" in env_extra: + ro = tmp_path / "readonly_tmp" + ro.mkdir() + ro.chmod(stat.S_IRUSR | stat.S_IXUSR) # r-x, no write + env_extra = {"SHADOWFROG_TMP_DIR": str(ro)} + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, env=_base_env(coupon_demo), + ).stdout.strip() + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = head + state_file.write_text(json.dumps(state)) + + payload = json.dumps({"toolName": "edit", + "toolInput": {"file_path": "cart.py"}}) + result, elapsed = _run(hook, coupon_demo, stdin=payload, env_extra=env_extra, + timeout=10.0) + _assert_fail_open(result, f"{hook.name} | env={scenario}", elapsed=elapsed) diff --git a/tests/hooks/test_pre_tool_sh.py b/tests/hooks/test_pre_tool_sh.py new file mode 100644 index 0000000..b0d2625 --- /dev/null +++ b/tests/hooks/test_pre_tool_sh.py @@ -0,0 +1,664 @@ +"""Tests for hook-templates/scripts/shadow-frog-pre-tool.sh — PreToolUse hook. + +Exercises: dedup isolation, injection resistance, field-name fallback, +tool filtering, SHADOWFROG_TMP_DIR override, happy-path discovery output, +PascalCase tool normalization, SIGTERM behavioral trap verification, the +strict production wall-clock budget, and file_path / path precedence. +""" +import json +import os +import shutil +import signal +import subprocess +import time +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent +HOOK_SCRIPT = REPO_ROOT / "hook-templates" / "scripts" / "shadow-frog-pre-tool.sh" + + +def _make_failing_git_stub(stub_dir: Path, fail_subcommand: str = "diff") -> Path: + """Create a `git` shim that fails for one subcommand and passes the rest + through to the real git. Prepend stub_dir to PATH to activate it. + + Simulates real-world transient `git diff` failures (warnings tripping + pipefail, exclude-pathspec quirks, version-specific non-zero exits) that + must never deny a tool call under Copilot CLI >= 1.0.57. + """ + stub_dir.mkdir(parents=True, exist_ok=True) + real_git = shutil.which("git") + shim = stub_dir / "git" + shim.write_text( + "#!/bin/bash\n" + f'if [ "$1" = "{fail_subcommand}" ]; then\n' + ' echo "warning: simulated git failure" >&2\n' + " exit 1\n" + "fi\n" + f'exec "{real_git}" "$@"\n' + ) + shim.chmod(0o755) + return shim + + +def _base_env(cwd: Path, extras: dict | None = None) -> dict: + """Minimal env isolating from user environment but preserving PATH for python3/git.""" + env = { + "PATH": os.environ.get("PATH", "/usr/bin:/bin:/usr/local/bin"), + "HOME": str(cwd), + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "LANG": "en_US.UTF-8", + } + if extras: + env.update(extras) + return env + + +def run_hook(json_input: dict, cwd: Path, env_extra: dict | None = None) -> subprocess.CompletedProcess: + """Run the pre-tool hook with given JSON on stdin.""" + env = _base_env(cwd, env_extra) + return subprocess.run( + ["bash", str(HOOK_SCRIPT)], + input=json.dumps(json_input), + capture_output=True, + text=True, + cwd=cwd, + env=env, + ) + + +def _fix_state_json_for_test(coupon_demo: Path) -> None: + """Update state.json so last_commit points to the test repo's HEAD. + + The coupon_demo fixture creates a fresh git repo, so the original + last_commit (from the source example) doesn't exist in history. + This causes 'git diff' to fail with exit 128 under pipefail. + """ + env = _base_env(coupon_demo) + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, env=env, + ).stdout.strip() + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = head + state_file.write_text(json.dumps(state)) + + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolNoShadow: + """When .shadow/ doesn't exist, hook exits cleanly.""" + + def test_no_shadow_dir_exits_zero(self, tmp_path): + result = run_hook( + {"tool_name": "Read", "tool_input": {"file_path": "foo.py"}}, + cwd=tmp_path, + ) + assert result.returncode == 0 + # No output or empty output (script exits before printing) + assert result.stdout.strip() == "" + + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolHappyPath: + """With coupon_demo .shadow/, hook returns discoveries.""" + + def test_mutation_tool_returns_json_with_context(self, coupon_demo): + _fix_state_json_for_test(coupon_demo) + result = run_hook( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, + ) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "additionalContext" in data + assert "ShadowFrog" in data["additionalContext"] + + def test_read_tool_returns_base_reminder(self, coupon_demo): + """Read/Bash are NOT mutation tools — should get base reminder only.""" + _fix_state_json_for_test(coupon_demo) + result = run_hook( + {"tool_name": "Read", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, + ) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "additionalContext" in data + # Base reminder is emitted (not file-specific actionable) + assert "ShadowFrog" in data["additionalContext"] + + def test_emits_both_output_shapes(self, coupon_demo): + """Output carries top-level (Copilot) and nested (Claude Code) context.""" + _fix_state_json_for_test(coupon_demo) + result = run_hook( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, + ) + data = json.loads(result.stdout) + hso = data.get("hookSpecificOutput") + assert hso is not None + assert hso["hookEventName"] == "PreToolUse" + assert hso["additionalContext"] == data["additionalContext"] + + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolDedup: + """PPID+lstart dedup prevents repeated injection for same file.""" + + def test_same_session_same_file_deduplicates(self, coupon_demo, tmp_path): + """Same SHADOWFROG_TMP_DIR → second call uses cached dedup marker.""" + _fix_state_json_for_test(coupon_demo) + dedup_dir = tmp_path / "dedup" + dedup_dir.mkdir() + extras = {"SHADOWFROG_TMP_DIR": str(dedup_dir)} + + # First call + r1 = run_hook( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, env_extra=extras, + ) + # Second call — same session (same PPID inherently), same file + r2 = run_hook( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, env_extra=extras, + ) + assert r1.returncode == 0 + assert r2.returncode == 0 + d1 = json.loads(r1.stdout) + d2 = json.loads(r2.stdout) + # Both return valid JSON with additionalContext + assert "additionalContext" in d1 + assert "additionalContext" in d2 + # If first had actionable content, second should NOT re-run --top. + # The dedup file should exist after first call. + dedup_files = list(dedup_dir.rglob("*.injected")) + if "Actionable" in d1["additionalContext"]: + assert len(dedup_files) >= 1 + # Second call should have the pointer message (not full --top output) + # because dedup file already exists + assert "Actionable" not in d2["additionalContext"] or \ + d2["additionalContext"] == d1["additionalContext"] + + def test_different_sessions_no_shared_dedup(self, coupon_demo, tmp_path): + """Different SHADOWFROG_TMP_DIR paths → no shared dedup state.""" + _fix_state_json_for_test(coupon_demo) + dedup1 = tmp_path / "dedup1" + dedup1.mkdir() + dedup2 = tmp_path / "dedup2" + dedup2.mkdir() + + r1 = run_hook( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, + env_extra={"SHADOWFROG_TMP_DIR": str(dedup1)}, + ) + r2 = run_hook( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, + env_extra={"SHADOWFROG_TMP_DIR": str(dedup2)}, + ) + assert r1.returncode == 0 + assert r2.returncode == 0 + d1 = json.loads(r1.stdout) + d2 = json.loads(r2.stdout) + # Both should have context (no cross-session dedup) + assert "additionalContext" in d1 + assert "additionalContext" in d2 + # Both should have the same level of detail (both first-time) + if "Actionable" in d1["additionalContext"]: + assert "Actionable" in d2["additionalContext"] + + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolInjection: + """Injection resistance — malicious file_path must not execute.""" + + def test_injection_in_file_path(self, coupon_demo, tmp_path): + _fix_state_json_for_test(coupon_demo) + # Use tmp_path-scoped marker instead of /tmp/PWNED so the test + # cannot false-positive on a stale file from a previous failure + # and cannot collide with parallel test runs (per opus-4.7xh review). + pwn_marker = tmp_path / "PWNED" + malicious_path = f'cart.py"); import os; os.system("touch {pwn_marker}"); #' + result = run_hook( + {"tool_name": "edit", "tool_input": {"file_path": malicious_path}}, + cwd=coupon_demo, + env_extra={"SHADOWFROG_TMP_DIR": str(tmp_path / "dedup")}, + ) + # Hook should not crash hard (exit 0 still outputs JSON, or exits gracefully) + assert result.returncode == 0 + # The exploit file must NOT exist + assert not pwn_marker.exists() + + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolFieldFallback: + """I6: Hook handles both tool_input.file_path and tool_input.path.""" + + def test_file_path_field(self, coupon_demo): + _fix_state_json_for_test(coupon_demo) + result = run_hook( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, + ) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "cart.py" in data["additionalContext"] + + def test_path_field_fallback(self, coupon_demo): + _fix_state_json_for_test(coupon_demo) + result = run_hook( + {"tool_name": "edit", "tool_input": {"path": "cart.py"}}, + cwd=coupon_demo, + ) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "cart.py" in data["additionalContext"] + + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolNonMutationTool: + """Non-mutation tools (Bash, grep, etc.) get base reminder, no file lookup.""" + + def test_bash_tool_no_file_lookup(self, coupon_demo): + _fix_state_json_for_test(coupon_demo) + result = run_hook( + {"tool_name": "Bash", "tool_input": {"command": "ls"}}, + cwd=coupon_demo, + ) + assert result.returncode == 0 + data = json.loads(result.stdout) + assert "additionalContext" in data + # Should be the generic base reminder + assert ".shadow/ mirrors" in data["additionalContext"] + + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolTmpDirOverride: + """SHADOWFROG_TMP_DIR env override redirects dedup state.""" + + def test_dedup_state_uses_override_dir(self, coupon_demo, tmp_path): + _fix_state_json_for_test(coupon_demo) + custom_tmp = tmp_path / "custom_tmp" + custom_tmp.mkdir() + run_hook( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, + env_extra={"SHADOWFROG_TMP_DIR": str(custom_tmp)}, + ) + # Dedup state should be in custom_tmp, not /tmp + shadowfrog_dirs = list(custom_tmp.glob("shadowfrog-hook-*")) + assert len(shadowfrog_dirs) >= 1 + + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolFailOpen: + """Regression: preToolUse is ADVISORY and must ALWAYS exit 0. + + Copilot CLI >= 1.0.57 denies the tool call when a preToolUse command hook + exits non-zero ("Denied by preToolUse hook (hook errored)"). The hook must + therefore run fail-open: any internal failure (git hiccup, missing state, + non-git dir, locale issue) still exits 0 so the user's tool is never blocked. + """ + + def _set_stale_state(self, coupon_demo: Path) -> None: + """Point last_commit at HEAD, then advance HEAD so the shadow is stale + and the staleness/git-diff path runs.""" + env = _base_env(coupon_demo) + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, env=env, + ).stdout.strip() + state_file = coupon_demo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_file.read_text()) + state["last_commit"] = head + state_file.write_text(json.dumps(state)) + (coupon_demo / "newcode.py").write_text("x = 1\n") + subprocess.run(["git", "add", "-A"], cwd=coupon_demo, check=True, env=env) + subprocess.run( + ["git", "commit", "-q", "-m", "advance"], + cwd=coupon_demo, check=True, env=env, + ) + + def test_git_diff_failure_does_not_deny(self, coupon_demo, tmp_path): + """The original bug: stale shadow + a failing `git diff` made the hook + exit non-zero under `set -euo pipefail`, denying the tool.""" + self._set_stale_state(coupon_demo) + stub = tmp_path / "stubbin" + _make_failing_git_stub(stub, "diff") + env = {"PATH": f"{stub}:{os.environ.get('PATH', '')}"} + result = run_hook( + {"tool_name": "Bash", "tool_input": {"command": "ls"}}, + cwd=coupon_demo, env_extra=env, + ) + assert result.returncode == 0, f"hook denied tool! stderr={result.stderr}" + # Still emits valid JSON (default-allow with advisory context). + json.loads(result.stdout) + + def test_missing_state_json_exits_zero_no_stderr(self, coupon_demo): + """Missing state.json must exit 0 AND not leak a shell redirection + error to stderr (state.json is opened inside Python now).""" + (coupon_demo / ".shadow" / "_meta" / "state.json").unlink() + result = run_hook( + {"tool_name": "Bash", "tool_input": {"command": "ls"}}, + cwd=coupon_demo, + ) + assert result.returncode == 0 + assert result.stderr.strip() == "", f"unexpected stderr: {result.stderr!r}" + json.loads(result.stdout) + + def test_non_git_dir_exits_zero(self, tmp_path): + """A .shadow/ in a directory that is NOT a git repo must not deny.""" + shadow_meta = tmp_path / ".shadow" / "_meta" + shadow_meta.mkdir(parents=True) + (shadow_meta / "state.json").write_text( + json.dumps({"last_commit": "abc1234", "total_files": 0, + "total_discoveries": 0}) + ) + result = run_hook( + {"tool_name": "edit", "tool_input": {"path": str(tmp_path / "x.py")}}, + cwd=tmp_path, + ) + assert result.returncode == 0 + json.loads(result.stdout) + + def test_git_rev_parse_failure_does_not_deny(self, coupon_demo, tmp_path): + """Even if `git rev-parse` itself fails, the hook stays fail-open.""" + self._set_stale_state(coupon_demo) + stub = tmp_path / "stubbin" + _make_failing_git_stub(stub, "rev-parse") + env = {"PATH": f"{stub}:{os.environ.get('PATH', '')}"} + result = run_hook( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, env_extra=env, + ) + assert result.returncode == 0, f"hook denied tool! stderr={result.stderr}" + + +# --------------------------------------------------------------------------- +# PascalCase tool names must trigger the file-specific actionable branch, not +# just lowercase ones. Claude Code emits `Edit`/`Write`/`MultiEdit`/ +# `NotebookEdit`; Copilot emits `edit`/`write`/`str_replace`. Removing +# `.lower()` from TOOL_NAME normalization silently downgrades all Claude +# Code mutations to the base reminder. +# --------------------------------------------------------------------------- + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolPascalCaseToolNames: + """File-specific actionable branch must fire for every spelling of every + mutation tool, across both Copilot CLI and Claude Code casing conventions.""" + + @pytest.mark.parametrize("tool_name", [ + # Copilot CLI lowercase forms + "edit", "create", "str_replace", "write", + # Claude Code PascalCase forms + "Edit", "Write", "MultiEdit", "NotebookEdit", + # Pathological edge cases — uppercased / mixed + "EDIT", "WRITE", "Create", + ]) + def test_mutation_tool_triggers_file_specific_branch( + self, coupon_demo, tmp_path, tool_name + ): + """The actionable message MUST reference the target file. Falling + back to the base reminder (which omits the file path) is a silent + regression of `.lower()` normalization.""" + _fix_state_json_for_test(coupon_demo) + # Per-test isolated dedup tmpdir so each parametrized cell exercises + # the viewer-discovery branch fresh. + result = run_hook( + {"tool_name": tool_name, "tool_input": {"file_path": "cart.py"}}, + cwd=coupon_demo, + env_extra={"SHADOWFROG_TMP_DIR": str(tmp_path / "dedup")}, + ) + assert result.returncode == 0, ( + f"{tool_name}: hook denied (exit={result.returncode})\n" + f"stderr={result.stderr!r}" + ) + data = json.loads(result.stdout) + ctx = data["additionalContext"] + # The file-specific branches (actionable OR pointer) both include + # the target file's name. The base reminder does not — so this + # assertion proves the IS_MUTATION branch fired correctly. + assert "cart.py" in ctx, ( + f"{tool_name}: TOOL NORMALIZATION REGRESSION. Mutation tool got\n" + f"the base reminder instead of file-specific output. Either\n" + f"`.lower()` was removed from TOOL_NAME, or the case statement\n" + f"lost the matching alias.\nContext: {ctx!r}" + ) + + +# --------------------------------------------------------------------------- +# subprocess.run(timeout=) sends SIGKILL, not SIGTERM, so the matrix has zero +# behavioral coverage for the TERM trap. Removing `trap 'exit 0' TERM` is +# invisible to pytest because every test completes well before the 10s +# subprocess timeout. This test exercises the REAL trap by sending SIGTERM +# to a running hook process. +# --------------------------------------------------------------------------- + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolSigterm: + """Behavioral verification of `trap 'exit 0' TERM` — distinct from + static CI checker coverage.""" + + def _spawn_and_signal(self, coupon_demo, tmp_path, env_extra=None, + signal_delay=0.05): + """Spawn the hook, send SIGTERM after `signal_delay` seconds, + return (returncode, stdout, stderr, wall_clock).""" + import threading + env = _base_env(coupon_demo, env_extra) + payload = json.dumps( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}} + ) + t0 = time.perf_counter() + proc = subprocess.Popen( + ["bash", str(HOOK_SCRIPT)], + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + cwd=coupon_demo, + env=env, + text=True, + ) + # Schedule SIGTERM in a background thread so communicate() can + # handle the stdin write + reads atomically. This avoids the + # "I/O operation on closed file" race when we close stdin + # manually and then call communicate(). + def _kill_after(): + time.sleep(signal_delay) + try: + proc.send_signal(signal.SIGTERM) + except ProcessLookupError: + pass # already exited + t = threading.Thread(target=_kill_after, daemon=True) + t.start() + try: + stdout, stderr = proc.communicate(input=payload, timeout=8.0) + except subprocess.TimeoutExpired: + proc.kill() + stdout, stderr = proc.communicate() + t.join(timeout=1.0) + pytest.fail( + f"hook did not exit within 8s of SIGTERM (signal_delay={signal_delay}).\n" + f"Likely cause: an unbounded foreground subprocess queued the signal.\n" + f"stdout={stdout!r}\nstderr={stderr!r}" + ) + t.join(timeout=1.0) + return proc.returncode, stdout, stderr, time.perf_counter() - t0 + + def test_sigterm_returns_zero_exit(self, coupon_demo, tmp_path): + """The trap pyramid must convert SIGTERM to exit 0. Without the + TERM trap, bash returns 143/-15 → Copilot CLI denies the tool. + + We run several attempts at varied delays to maximize the chance + of catching bash in the gap between bounded subprocesses (where + the trap MUST fire). The bounded subprocesses themselves ensure + the queued signal is eventually delivered.""" + _fix_state_json_for_test(coupon_demo) + failures = [] + for attempt, delay in enumerate([0.01, 0.05, 0.1, 0.2, 0.5]): + rc, stdout, stderr, elapsed = self._spawn_and_signal( + coupon_demo, tmp_path, + env_extra={"SHADOWFROG_TMP_DIR": str(tmp_path / f"dedup_{attempt}")}, + signal_delay=delay, + ) + if rc != 0 or stderr.strip(): + failures.append( + f" attempt {attempt} (delay={delay}s): rc={rc} " + f"elapsed={elapsed:.2f}s stderr={stderr!r}" + ) + assert not failures, ( + "SIGTERM did not produce exit 0 with empty stderr (TERM trap broken):\n" + + "\n".join(failures) + ) + + def test_sigterm_during_hung_subprocess_still_eventually_exits( + self, coupon_demo, tmp_path + ): + """Even when bash is waiting on a foreground subprocess (signals + queue), the subprocess's bounded timeout ensures the signal is + eventually delivered and the trap fires. The trap pyramid is a + safety net for fast failures; bounded subprocess.run(timeout=) is + the actual hang defense.""" + _fix_state_json_for_test(coupon_demo) + # Stub git so the viewer-branch rev-parse takes 1.5s (within the + # 0.5s python-level timeout, so subprocess.run will TIMEOUT and + # raise → bash continues → SIGTERM trap fires). + stub = tmp_path / "stubbin" + stub.mkdir() + real_git = shutil.which("git") + (stub / "git").write_text( + "#!/bin/bash\n" + 'if [ "$1" = "rev-parse" ] && [ "$2" = "--show-toplevel" ]; then\n' + " sleep 1.5\n" + " exit 0\n" + "fi\n" + f'exec "{real_git}" "$@"\n' + ) + (stub / "git").chmod(0o755) + rc, stdout, stderr, elapsed = self._spawn_and_signal( + coupon_demo, tmp_path, + env_extra={ + "PATH": f"{stub}:{os.environ.get('PATH','')}", + "SHADOWFROG_TMP_DIR": str(tmp_path / "dedup"), + }, + signal_delay=0.05, + ) + assert rc == 0, ( + f"SIGTERM mid-subprocess: exit={rc} (Copilot would DENY).\n" + f"elapsed={elapsed:.2f}s\nstderr={stderr!r}" + ) + # Total wall-clock must still be under the runner's 5s budget, + # proving bounded subprocesses cap the signal-queue latency. + assert elapsed < 5.0, ( + f"SIGTERM was queued behind subprocess for {elapsed:.2f}s " + f"(>= 5.0s Copilot deny threshold). The viewer-branch " + f"rev-parse subprocess timeout is no longer bounded." + ) + + +# --------------------------------------------------------------------------- +# The matrix's 7s wall-clock cap is generous to tolerate CI cold-start +# variance, but the HAPPY PATH (no faults, real environment) must complete +# within Copilot's strict 5s deny threshold. A 6.3s hook passes the matrix +# but is denied in production. +# --------------------------------------------------------------------------- + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolStrictBudget: + """Production-realistic happy-path budget — distinct from fault-cell budget.""" + + HAPPY_PATH_BUDGET_SEC = 5.0 # Copilot CLI's preToolUse deny threshold + + def test_happy_path_meets_strict_production_budget( + self, coupon_demo, tmp_path + ): + """With no faults, the hook MUST complete under 5.0s — the actual + production deny threshold. Run 3 times and assert the MIN elapsed + (best of 3) clears the bar, so transient CI spikes don't cause + false failures while still catching genuine creeping slowdown.""" + _fix_state_json_for_test(coupon_demo) + payload = json.dumps( + {"tool_name": "edit", "tool_input": {"file_path": "cart.py"}} + ) + elapseds = [] + for attempt in range(3): + t0 = time.perf_counter() + result = subprocess.run( + ["bash", str(HOOK_SCRIPT)], + input=payload, + capture_output=True, + text=True, + cwd=coupon_demo, + env=_base_env(coupon_demo, { + "SHADOWFROG_TMP_DIR": str(tmp_path / f"dedup_{attempt}"), + }), + timeout=10.0, + ) + elapsed = time.perf_counter() - t0 + elapseds.append(elapsed) + assert result.returncode == 0, ( + f"attempt {attempt}: hook returned {result.returncode}, " + f"stderr={result.stderr!r}" + ) + best = min(elapseds) + assert best < self.HAPPY_PATH_BUDGET_SEC, ( + f"Happy-path hook is too slow: best of 3 = {best:.2f}s, " + f"all elapseds = {[f'{e:.2f}' for e in elapseds]}. " + f"Copilot CLI denies tool calls when hook exceeds " + f"{self.HAPPY_PATH_BUDGET_SEC}s. This is the production " + f"reality, not the matrix's lenient {7.0}s cap." + ) + + +# --------------------------------------------------------------------------- +# When both `file_path` and `path` are present, precedence must prefer +# `file_path` (the canonical key for both Copilot CLI and Claude Code). +# Flipping precedence would silently route to the wrong file's shadow. +# --------------------------------------------------------------------------- + +@pytest.mark.slow +@pytest.mark.integration +class TestPreToolPathPrecedence: + """When both file_path and path are present, file_path wins.""" + + def test_file_path_takes_precedence_over_path(self, coupon_demo, tmp_path): + """A payload carrying BOTH keys: the hook should reference the + `file_path` value's shadow, not the `path` value's.""" + _fix_state_json_for_test(coupon_demo) + # cart.py has a shadow; inventory.py also has one. The hook should + # pick cart.py (from file_path), not inventory.py (from path). + result = run_hook( + {"tool_name": "edit", "tool_input": { + "file_path": "cart.py", + "path": "inventory.py", + }}, + cwd=coupon_demo, + env_extra={"SHADOWFROG_TMP_DIR": str(tmp_path / "dedup")}, + ) + assert result.returncode == 0 + data = json.loads(result.stdout) + ctx = data["additionalContext"] + # Must reference cart.py (the file_path) + assert "cart.py" in ctx, ( + f"file_path was ignored; precedence is wrong. ctx={ctx!r}" + ) + # Must NOT reference inventory.py (the path) — that would mean + # `path` took precedence, which is the bug. + assert "inventory.py" not in ctx, ( + f"PATH PRECEDENCE BUG: `path` value won over `file_path`. " + f"This routes Copilot/Claude Code edits to the wrong shadow. " + f"ctx={ctx!r}" + ) diff --git a/tests/skills/__init__.py b/tests/skills/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/skills/shadow_frog_dream/__init__.py b/tests/skills/shadow_frog_dream/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/skills/shadow_frog_dream/test_dream_cleanup_sh.py b/tests/skills/shadow_frog_dream/test_dream_cleanup_sh.py new file mode 100644 index 0000000..258999a --- /dev/null +++ b/tests/skills/shadow_frog_dream/test_dream_cleanup_sh.py @@ -0,0 +1,360 @@ +"""Tests for skills/shadow-frog-dream/dream-cleanup.sh. + +Two layers to exercise: +1. Happy paths — proper worktrees get cleaned (git path succeeds). +2. Fallback paths — when `git worktree remove` fails (dead .git pointer, + no repo-root available, …) rm -rf takes over AND the safety gate + refuses anything outside `$DREAM_WORKTREE_BASE/<ns>/dream-<slug>`. + +The bug this fixes — `git worktree remove ... 2>/dev/null` swallowing +the error AND leaving the directory on disk — is the test below +`test_fallback_rm_when_gitdir_broken`. +""" +import os +import subprocess +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent +CLEANUP_SH = REPO_ROOT / "skills" / "shadow-frog-dream" / "dream-cleanup.sh" + + +def _base_env(extras: dict | None = None) -> dict: + env = { + "PATH": os.environ.get("PATH", "/usr/bin:/bin:/usr/local/bin"), + "HOME": os.environ.get("HOME", "/tmp"), + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "LANG": "en_US.UTF-8", + } + if extras: + env.update(extras) + return env + + +def _make_repo(path: Path) -> Path: + """Create a tiny git repo to host worktrees.""" + env = _base_env() + subprocess.run(["git", "init", "-q", "-b", "main", str(path)], check=True, env=env) + subprocess.run(["git", "-C", str(path), "config", "user.email", "t@t.invalid"], check=True, env=env) + subprocess.run(["git", "-C", str(path), "config", "user.name", "T"], check=True, env=env) + subprocess.run(["git", "-C", str(path), "config", "commit.gpgsign", "false"], check=True, env=env) + (path / "r.md").write_text("hi\n") + subprocess.run(["git", "-C", str(path), "add", "-A"], check=True, env=env) + subprocess.run(["git", "-C", str(path), "commit", "-q", "-m", "init"], check=True, env=env) + return path + + +def _run(args: list[str], env_extra: dict | None = None) -> subprocess.CompletedProcess: + return subprocess.run( + ["bash", str(CLEANUP_SH), *args], + capture_output=True, text=True, env=_base_env(env_extra), + ) + + +# =========================================================================== +# Usage / arg parsing +# =========================================================================== + +class TestUsage: + def test_help_exits_zero(self): + r = _run(["--help"]) + assert r.returncode == 0 + assert "Usage" in r.stdout or "dream-cleanup.sh" in r.stdout + + def test_no_args_errors(self): + r = _run([]) + assert r.returncode == 2 + assert "required" in r.stderr.lower() + + def test_unknown_flag_errors(self): + r = _run(["/tmp/x", "--bogus"]) + assert r.returncode == 2 + + def test_extra_positional_errors(self): + r = _run(["/tmp/a", "/tmp/b"]) + assert r.returncode == 2 + + +# =========================================================================== +# Happy path: git worktree remove succeeds +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +class TestHappyPath: + def test_removes_real_worktree(self, tmp_path): + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + target = base / "proj" / "dream-foo" + target.parent.mkdir(parents=True) + subprocess.run( + ["git", "-C", str(repo), "worktree", "add", "-q", + str(target), "-b", "dream/proj/20260101-000000Z-foo"], + check=True, env=_base_env(), + ) + assert target.is_dir() + + r = _run( + [str(target), "--repo-root", str(repo)], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, f"stderr: {r.stderr}\nstdout: {r.stdout}" + assert not target.exists() + # git worktree bookkeeping is clean. + list_r = subprocess.run( + ["git", "-C", str(repo), "worktree", "list"], + capture_output=True, text=True, env=_base_env(), + ) + assert str(target) not in list_r.stdout + + def test_idempotent_when_path_missing(self, tmp_path): + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + base.mkdir() + # Path never existed. + r = _run( + [str(base / "proj" / "dream-foo"), "--repo-root", str(repo)], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert "Nothing to clean" in r.stdout + + +# =========================================================================== +# Fallback path: rm -rf fires when git can't help +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +class TestFallback: + def test_fallback_rm_when_gitdir_broken(self, tmp_path): + """Repro for bug-worktree-leak.md: a dream worktree whose .git + pointer is dangling (git's perspective: gone; disk's perspective: + still there). The pre-fix snippet silently leaked this — this test + FAILS on the old `git worktree remove ... 2>/dev/null` snippet.""" + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + target = base / "proj" / "dream-leaked" + target.mkdir(parents=True) + # Dangling gitdir pointer. + (target / ".git").write_text("gitdir: /nonexistent/path\n") + (target / "leaked.pyc").write_text("# nobody owns me\n") + + r = _run( + [str(target), "--repo-root", str(repo)], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, f"stderr: {r.stderr}\nstdout: {r.stdout}" + assert not target.exists(), "fallback rm -rf should have cleared it" + + def test_fallback_rm_without_repo_root(self, tmp_path): + """When the caller doesn't pass --repo-root and the dream worktree + is dead, the script should still remove the directory.""" + base = tmp_path / "wt-base" + target = base / "proj" / "dream-orphan" + target.mkdir(parents=True) + (target / "leaked.txt").write_text("orphan\n") + # No .git file at all → looks like never-was-a-worktree. + + r = _run( + [str(target)], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, f"stderr: {r.stderr}\nstdout: {r.stdout}" + assert not target.exists() + + +# =========================================================================== +# Safety gate refuses +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +class TestSafetyGate: + def test_refuses_path_outside_base(self, tmp_path): + """If WORKTREE_DIR is outside $DREAM_WORKTREE_BASE the script + refuses AND leaves the directory untouched.""" + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + base.mkdir() + # Decoy outside the base. + outside = tmp_path / "decoy" / "proj" / "dream-foo" + outside.mkdir(parents=True) + (outside / "important.txt").write_text("DO NOT DELETE\n") + + r = _run( + [str(outside), "--repo-root", str(repo)], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 1, f"expected refuse: {r.stdout}\n{r.stderr}" + assert outside.is_dir(), "directory must survive a refused cleanup" + assert (outside / "important.txt").read_text() == "DO NOT DELETE\n" + + def test_refuses_base_itself(self, tmp_path): + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + base.mkdir() + r = _run( + [str(base), "--repo-root", str(repo)], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 1 + assert base.is_dir() + + def test_refuses_sensitive_base_via_env(self, tmp_path): + """DREAM_WORKTREE_BASE=/tmp must be refused even though /tmp/x/ + looks shape-correct. Verifies the macOS /private/tmp bypass is + closed end-to-end through the bash wrapper.""" + repo = _make_repo(tmp_path / "repo") + # The path lives under /tmp — DON'T create it; the safety gate + # should refuse based on base alone. + r = _run( + ["/tmp/should-never-rm/proj/dream-foo", + "--repo-root", str(repo)], + env_extra={"DREAM_WORKTREE_BASE": "/tmp"}, + ) + # Note: refused-because-base-is-/tmp surfaces as rc=1 from the + # safety gate. rc=0 with "Nothing to clean" would mean the gate + # accepted /tmp as a valid base — regression alarm. + assert r.returncode == 1, ( + f"sensitive base /tmp was NOT refused — gate regression!\n" + f"stdout: {r.stdout}\nstderr: {r.stderr}" + ) + + def test_refuses_path_with_traversal(self, tmp_path): + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + base.mkdir() + bad = f"{base}/proj/../escape/dream-foo" + r = _run( + [bad, "--repo-root", str(repo)], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 1 + + def test_refuses_path_with_wrong_shape(self, tmp_path): + """`<base>/proj/notdream-foo` shouldn't be deletable — the leaf + must start with `dream-`.""" + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + target = base / "proj" / "notdream-foo" + target.mkdir(parents=True) + r = _run( + [str(target), "--repo-root", str(repo)], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 1 + assert target.is_dir() + + +# =========================================================================== +# Regressions for the 5-model review panel findings (S1-S3, C1-C2) +# =========================================================================== + +class TestSafetyModuleMissing: + """C1: if _worktree_safety.py is missing, python3 itself exits 2 — + the same code as the gate's "safe but path missing" sentinel. The + script MUST distinguish these and refuse loudly when the gate is + unloadable, not silently report success. + """ + + def _copy_script_without_safety(self, dst: Path) -> Path: + """Copy dream-cleanup.sh + _worktree_safety.py to dst/, then + DELETE the safety module to simulate a broken install.""" + import shutil + src = REPO_ROOT / "skills" / "shadow-frog-dream" + dst.mkdir() + shutil.copy(src / "dream-cleanup.sh", dst / "dream-cleanup.sh") + # intentionally do NOT copy _worktree_safety.py + os.chmod(dst / "dream-cleanup.sh", 0o755) + return dst / "dream-cleanup.sh" + + def test_missing_safety_module_exits_4(self, tmp_path): + cleanup = self._copy_script_without_safety(tmp_path / "broken") + base = tmp_path / "wt-base" + wt = base / "proj" / "dream-foo" + wt.mkdir(parents=True) + r = subprocess.run( + ["bash", str(cleanup), str(wt)], + capture_output=True, text=True, + env=_base_env({"DREAM_WORKTREE_BASE": str(base)}), + ) + assert r.returncode == 4, ( + f"missing _worktree_safety.py should exit 4, got {r.returncode}\n" + f"stderr: {r.stderr}" + ) + assert "safety module" in r.stderr.lower() + # CRITICAL: the worktree must NOT be deleted when the gate is unloadable. + assert wt.is_dir(), "worktree was deleted without a safety gate!" + + +class TestRepoRootEnvInherited: + """C2: $REPO_ROOT env var must be respected when --repo-root flag is + absent. The previous `REPO_ROOT="$REPO_ROOT_OVERRIDE"` clobbered any + inherited value.""" + + def test_repo_root_from_env_is_used(self, tmp_path): + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + base.mkdir() + wt = base / "proj" / "dream-foo" + # Real worktree registered to repo (so git can remove it). + subprocess.run( + ["git", "-C", str(repo), "worktree", "add", "-b", "tmp-test", str(wt)], + check=True, capture_output=True, env=_base_env(), + ) + r = _run( + [str(wt)], # NO --repo-root flag + env_extra={ + "DREAM_WORKTREE_BASE": str(base), + "REPO_ROOT": str(repo), + }, + ) + assert r.returncode == 0, f"stdout: {r.stdout}\nstderr: {r.stderr}" + assert not wt.exists(), "worktree should have been removed via env REPO_ROOT" + # The polite git path should have run (not just the fallback rm). + assert "Removed via git worktree" in r.stdout + + +class TestRepoRootAsWorktree: + """S3: when REPO_ROOT is itself a worktree, `.git` is a file (not a + directory). The previous `[[ -d "$REPO_ROOT/.git" ]]` check silently + skipped the polite git-worktree path.""" + + def test_repo_root_as_worktree(self, tmp_path): + main_repo = _make_repo(tmp_path / "main") + # Add a worktree that we'll USE AS THE REPO_ROOT for the cleanup call. + worktree_repo = tmp_path / "wt-as-root" + subprocess.run( + ["git", "-C", str(main_repo), "worktree", "add", + "-b", "tmp-root", str(worktree_repo)], + check=True, capture_output=True, env=_base_env(), + ) + # Sanity: this is a worktree (`.git` is a file, not a directory). + assert (worktree_repo / ".git").is_file() + + # Now register a dream-shaped worktree against the SAME main repo. + base = tmp_path / "wt-base" + base.mkdir() + dream_wt = base / "proj" / "dream-foo" + subprocess.run( + ["git", "-C", str(main_repo), "worktree", "add", + "-b", "tmp-dream", str(dream_wt)], + check=True, capture_output=True, env=_base_env(), + ) + + # Cleanup, passing the WORKTREE (not the main repo) as --repo-root. + # Pre-fix this would skip the polite git path silently. + r = _run( + [str(dream_wt), "--repo-root", str(worktree_repo)], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, f"stdout: {r.stdout}\nstderr: {r.stderr}" + assert not dream_wt.exists() + # The polite git path SHOULD have fired — that's the regression. + assert "Removed via git worktree" in r.stdout, ( + f"polite git path was skipped (regression of -d .git check):\n" + f"stdout: {r.stdout}" + ) diff --git a/tests/skills/shadow_frog_dream/test_dream_coverage.py b/tests/skills/shadow_frog_dream/test_dream_coverage.py new file mode 100644 index 0000000..062e6f1 --- /dev/null +++ b/tests/skills/shadow_frog_dream/test_dream_coverage.py @@ -0,0 +1,332 @@ +"""Tests for skills/shadow-frog-dream/dream-coverage.py. + +The coverage script computes "is this source file shadowed?" stats for +a repo's tracked files, plus fan-in heuristics for prioritization. +Tests exercise the four pure-ish helpers plus a CLI smoke run against +the coupon-demo (which has a stable, committed `.shadow/` tree). +""" +import os +import subprocess +import sys +from pathlib import Path + +import pytest + + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent +SCRIPT = REPO_ROOT / "skills" / "shadow-frog-dream" / "dream-coverage.py" + + +def _git_env(home: Path) -> dict: + return { + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "HOME": str(home), + "PATH": "/usr/bin:/bin:/usr/local/bin:/opt/homebrew/bin", + } + + +def _seed_tree(repo: Path, files: dict[str, str]): + """Write `files` (rel-path → content) into `repo`, commit, return SHA.""" + env = _git_env(repo) + for rel, content in files.items(): + path = repo / rel + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(content) + subprocess.run(["git", "add", "-A"], cwd=repo, env=env, check=True) + subprocess.run(["git", "commit", "-q", "-m", "seed"], cwd=repo, env=env, check=True) + + +# =========================================================================== +# get_source_files — exclusion + scoping +# =========================================================================== + +def test_get_source_files_matches_init_selection(dream_coverage, tmp_git_repo): + """Coverage's source set mirrors shadow-init.py exactly: tests are + INCLUDED (init shadows them); only EXCLUDE_DIRS components and minified/ + map/lock suffixes are dropped.""" + _seed_tree(tmp_git_repo, { + "src/main.py": "print()\n", + "src/util.py": "x\n", + "tests/test_main.py": "x\n", # tests INCLUDED (init shadows them) + "test_util.py": "x\n", # test_ prefix INCLUDED + "util_test.py": "x\n", # _test suffix INCLUDED + "node_modules/x.js": "x\n", # node_modules/ excluded + "vendor/lib.py": "x\n", # vendor/ excluded + "dist/bundle.js": "x\n", # dist/ excluded + "src/app.min.js": "x\n", # .min.js excluded + "yarn.lock": "x\n", # .lock excluded (but not a source ext anyway) + }) + files = dream_coverage.get_source_files(str(tmp_git_repo)) + # Included (init shadows all of these): + for good in ["src/main.py", "src/util.py", "tests/test_main.py", + "test_util.py", "util_test.py"]: + assert good in files, f"{good!r} should be included (matches init)" + # Excluded (EXCLUDE_DIRS / minified): + for bad in ["node_modules/x.js", "vendor/lib.py", "dist/bundle.js", + "src/app.min.js"]: + assert bad not in files, f"{bad!r} should be excluded" + + +def test_get_source_files_excludes_non_source_files(dream_coverage, tmp_git_repo): + """B17: docs/images/data files are tracked but NOT shadowed by init, so + they must not inflate coverage's denominator.""" + _seed_tree(tmp_git_repo, { + "src/main.py": "print()\n", + "README.md": "# docs\n", # markdown — not source + "docs/guide.rst": "guide\n", # rst — not source + "assets/logo.png": "binary\n", # image — not source + "LICENSE": "MIT\n", # no extension, not a basename + "data.csv": "a,b\n", # data — not source + }) + files = dream_coverage.get_source_files(str(tmp_git_repo)) + assert "src/main.py" in files + for bad in ["README.md", "docs/guide.rst", "assets/logo.png", + "LICENSE", "data.csv"]: + assert bad not in files, f"{bad!r} should not be counted as source" + + +@pytest.mark.parametrize("rel,is_source", [ + ("src/a.py", True), + ("a.ts", True), + ("Makefile", True), + ("svc/Dockerfile", True), + ("README.md", False), + ("notes.txt", False), + ("logo.png", False), + ("LICENSE", False), +]) +def test_is_source_file(dream_coverage, rel, is_source): + assert dream_coverage._is_source_file(rel) is is_source + + +def test_get_source_files_filters_by_scope(dream_coverage, tmp_git_repo): + _seed_tree(tmp_git_repo, { + "src/auth/login.py": "x\n", + "src/auth/token.py": "x\n", + "src/db/conn.py": "x\n", + "src/util.py": "x\n", + }) + scoped = dream_coverage.get_source_files( + str(tmp_git_repo), scopes=["src/auth/"] + ) + assert "src/auth/login.py" in scoped + assert "src/auth/token.py" in scoped + assert "src/db/conn.py" not in scoped + assert "src/util.py" not in scoped + + +def test_get_source_files_multiple_scopes(dream_coverage, tmp_git_repo): + _seed_tree(tmp_git_repo, { + "src/auth/login.py": "x\n", + "src/db/conn.py": "x\n", + "src/util.py": "x\n", + }) + scoped = dream_coverage.get_source_files( + str(tmp_git_repo), scopes=["src/auth/", "src/db/"] + ) + assert set(scoped) == {"src/auth/login.py", "src/db/conn.py"} + + +def test_get_source_files_empty_scopes_returns_all(dream_coverage, tmp_git_repo): + _seed_tree(tmp_git_repo, {"a.py": "x\n", "b.py": "x\n"}) + files = dream_coverage.get_source_files(str(tmp_git_repo), scopes=[]) + assert set(files) == {"a.py", "b.py"} + + +# =========================================================================== +# _count_discoveries — identical shape to dream-reconcile's version +# =========================================================================== + +def test_count_discoveries_excludes_cross_references(dream_coverage, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## File-Level\n\n" + "- fl one\n" + "- fl two\n" + "\n" + "## `foo`\n\n" + "- sym one\n" + "- sym two\n" + "\n" + "## Cross-References\n\n" + "- [bp one](_cross/a.md)\n" + "- [bp two](_cross/b.md)\n" + ) + assert dream_coverage._count_discoveries(str(shadow)) == 4 + + +def test_count_discoveries_handles_lowercase_xref(dream_coverage, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n- real\n\n## cross-references\n\n- [bp](_cross/x.md)\n" + ) + assert dream_coverage._count_discoveries(str(shadow)) == 1 + + +def test_count_discoveries_matches_coupon_demo(dream_coverage, coupon_demo): + """Same totals as committed _index.md: cart=14, inventory=10, test_cart=9.""" + assert dream_coverage._count_discoveries( + str(coupon_demo / ".shadow" / "cart.py.md")) == 14 + assert dream_coverage._count_discoveries( + str(coupon_demo / ".shadow" / "inventory.py.md")) == 10 + assert dream_coverage._count_discoveries( + str(coupon_demo / ".shadow" / "test_cart.py.md")) == 9 + + +# =========================================================================== +# check_coverage +# =========================================================================== + +def test_check_coverage_classifies_files(dream_coverage, tmp_git_repo): + # Three source files; one fully covered, one placeholder-only, one no shadow. + _seed_tree(tmp_git_repo, { + "a.py": "x\n", + "b.py": "x\n", + "c.py": "x\n", + }) + (tmp_git_repo / ".shadow").mkdir() + (tmp_git_repo / ".shadow" / "a.py.md").write_text( + "## `a`\n\n- one\n- two\n" + ) + (tmp_git_repo / ".shadow" / "b.py.md").write_text( + "## `b`\n\n_No discoveries yet._\n" + ) + # c.py: no shadow at all. + + files = dream_coverage.get_source_files(str(tmp_git_repo)) + covered, uncovered, saturated = dream_coverage.check_coverage( + str(tmp_git_repo), files + ) + covered_paths = {f for f, _ in covered} + assert "a.py" in covered_paths + assert "b.py" in uncovered + assert "c.py" in uncovered + assert saturated == [] # 2 discoveries < 8 + + +def test_check_coverage_saturated_threshold(dream_coverage, tmp_git_repo): + _seed_tree(tmp_git_repo, {"big.py": "x\n"}) + (tmp_git_repo / ".shadow").mkdir() + (tmp_git_repo / ".shadow" / "big.py.md").write_text( + "## `big`\n\n" + "\n".join(f"- discovery {i}" for i in range(8)) + "\n" + ) + files = dream_coverage.get_source_files(str(tmp_git_repo)) + covered, uncovered, saturated = dream_coverage.check_coverage( + str(tmp_git_repo), files + ) + assert ("big.py", 8) in saturated + + +def test_check_coverage_against_coupon_demo(dream_coverage, coupon_demo): + """All 3 known source files are covered (cart, inventory, test_cart).""" + files = dream_coverage.get_source_files(str(coupon_demo)) + covered, uncovered, saturated = dream_coverage.check_coverage( + str(coupon_demo), files + ) + covered_paths = {f for f, _ in covered} + # All 3 are shadowed by init, so all 3 must be in the coverage denominator. + assert "cart.py" in covered_paths + assert "inventory.py" in covered_paths + assert "test_cart.py" in covered_paths + + +# =========================================================================== +# compute_fan_in +# =========================================================================== + +def test_compute_fan_in_counts_cross_file_references(dream_coverage, tmp_git_repo): + # `git grep -F -l <basename>` counts files whose CONTENT mentions the + # basename — so put the literal string into multiple files. + _seed_tree(tmp_git_repo, { + "src/auth.py": "# auth module\ndef login(): pass\n", # mentions auth + "src/api.py": "from src.auth import login\n", # mentions auth + "src/admin.py": "import src.auth as a\n", # mentions auth + "src/unrelated.py": "x = 1\n", + }) + uncovered = ["src/auth.py"] + fan = dream_coverage.compute_fan_in(str(tmp_git_repo), uncovered) + # `git grep -F -l auth` should match auth.py + api.py + admin.py = 3. + assert fan["src/auth.py"] == 3 + + +def test_compute_fan_in_returns_zero_for_unreferenced(dream_coverage, tmp_git_repo): + _seed_tree(tmp_git_repo, { + "src/lonely.py": "x = 1\n", + "src/other.py": "y = 2\n", + }) + fan = dream_coverage.compute_fan_in(str(tmp_git_repo), ["src/lonely.py"]) + assert fan["src/lonely.py"] == 0 + + +def test_compute_fan_in_empty_input(dream_coverage, tmp_git_repo): + assert dream_coverage.compute_fan_in(str(tmp_git_repo), []) == {} + + +def test_compute_fan_in_respects_max_files_cap(dream_coverage, tmp_git_repo): + """Files beyond the cap are not measured.""" + _seed_tree(tmp_git_repo, { + "src/a.py": "alpha\n", + "src/b.py": "beta\n", + "src/c.py": "gamma\n", + }) + uncovered = ["src/a.py", "src/b.py", "src/c.py"] + fan = dream_coverage.compute_fan_in(str(tmp_git_repo), uncovered, max_files=1) + # Only a.py was measured. + assert "src/a.py" in fan + assert "src/b.py" not in fan + assert "src/c.py" not in fan + + +# =========================================================================== +# CLI smoke +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_help_exits_zero(): + result = subprocess.run( + [sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True, + ) + assert result.returncode == 0 + assert "Dream coverage" in result.stdout or "coverage" in result.stdout.lower() + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_runs_against_coupon_demo(coupon_demo): + result = subprocess.run( + [sys.executable, str(SCRIPT), str(coupon_demo)], + capture_output=True, text=True, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "EXPLORATION COVERAGE MAP" in result.stdout + assert "Total source files:" in result.stdout + assert "Covered:" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_scope_with_no_matches_exits_nonzero(coupon_demo): + """Empty scope match is a hard exit-code-2 so typos in --scope are visible.""" + result = subprocess.run( + [sys.executable, str(SCRIPT), str(coupon_demo), + "--scope", "no-such-prefix/"], + capture_output=True, text=True, + ) + assert result.returncode == 2 + assert "No source files matched" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_scope_filters_to_subset(coupon_demo): + """A scope that matches at least one file → exits 0 with scoped totals.""" + result = subprocess.run( + [sys.executable, str(SCRIPT), str(coupon_demo), "--scope", "cart.py"], + capture_output=True, text=True, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "Total source files: 1" in result.stdout + assert "cart.py" in result.stdout diff --git a/tests/skills/shadow_frog_dream/test_dream_gc_sh.py b/tests/skills/shadow_frog_dream/test_dream_gc_sh.py new file mode 100644 index 0000000..3cba88c --- /dev/null +++ b/tests/skills/shadow_frog_dream/test_dream_gc_sh.py @@ -0,0 +1,741 @@ +"""Tests for skills/shadow-frog-dream/dream-gc.sh. + +Sweeper for orphan dream worktrees — defense in depth for the cases +`dream-cleanup.sh` missed (machine crash, OOM-killed agent, …). Every +candidate is re-validated through the safety gate before removal, so +even a misconfigured `$DREAM_WORKTREE_BASE` cannot cause data loss. +""" +import os +import subprocess +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent +GC_SH = REPO_ROOT / "skills" / "shadow-frog-dream" / "dream-gc.sh" + + +def _base_env(extras: dict | None = None) -> dict: + env = { + "PATH": os.environ.get("PATH", "/usr/bin:/bin:/usr/local/bin"), + "HOME": os.environ.get("HOME", "/tmp"), + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "LANG": "en_US.UTF-8", + } + if extras: + env.update(extras) + return env + + +def _make_repo(path: Path) -> Path: + env = _base_env() + subprocess.run(["git", "init", "-q", "-b", "main", str(path)], check=True, env=env) + subprocess.run(["git", "-C", str(path), "config", "user.email", "t@t.invalid"], check=True, env=env) + subprocess.run(["git", "-C", str(path), "config", "user.name", "T"], check=True, env=env) + subprocess.run(["git", "-C", str(path), "config", "commit.gpgsign", "false"], check=True, env=env) + (path / "r.md").write_text("hi\n") + subprocess.run(["git", "-C", str(path), "add", "-A"], check=True, env=env) + subprocess.run(["git", "-C", str(path), "commit", "-q", "-m", "init"], check=True, env=env) + return path + + +def _orphan_worktree(parent: Path, name: str = "dream-orphan", old: bool = True) -> Path: + """Build a dir under `parent` that looks like an abandoned worktree.""" + d = parent / name + d.mkdir(parents=True) + (d / ".git").write_text("gitdir: /nonexistent/path/worktrees/missing\n") + (d / "leftover.txt").write_text("orphaned\n") + if old: + # Force ancient mtime so --min-age-min finds it. Use os.utime + # because `touch -t` syntax differs across platforms. + ancient = 946684800 # 2000-01-01 00:00 UTC + os.utime(d, (ancient, ancient)) + return d + + +def _run(args: list[str], env_extra: dict | None = None) -> subprocess.CompletedProcess: + return subprocess.run( + ["bash", str(GC_SH), *args], + capture_output=True, text=True, env=_base_env(env_extra), + ) + + +# =========================================================================== +# Usage +# =========================================================================== + +class TestUsage: + def test_help_exits_zero(self): + r = _run(["--help"]) + assert r.returncode == 0 + assert "Usage" in r.stdout or "dream-gc.sh" in r.stdout + + def test_unknown_flag_errors(self): + r = _run(["--bogus"]) + assert r.returncode == 2 + + @pytest.mark.parametrize("bad_age", ["foo", "-5", "5min", "1.5", ""]) + def test_invalid_min_age_errors(self, bad_age): + r = _run(["--min-age-min", bad_age]) + assert r.returncode == 2, f"--min-age-min {bad_age!r} should reject" + + +# =========================================================================== +# Safety: base validation +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +class TestBaseSafety: + @pytest.mark.parametrize("base", [ + "/", "/tmp", "/etc", "/var", "/home", "/Users", + "/private/tmp", "/private/etc", + ]) + def test_refuses_sensitive_base(self, base): + r = _run(["--dry-run"], env_extra={"DREAM_WORKTREE_BASE": base}) + assert r.returncode == 1, ( + f"sensitive base {base!r} should be refused. stdout: {r.stdout}\n" + f"stderr: {r.stderr}" + ) + + def test_noop_when_base_missing(self, tmp_path): + base = tmp_path / "never-existed" + # No mkdir — base must not exist + r = _run([], env_extra={"DREAM_WORKTREE_BASE": str(base)}) + assert r.returncode == 0 + assert "does not exist" in r.stdout + + +# =========================================================================== +# Sweep behavior +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +class TestSweep: + def test_removes_orphan(self, tmp_path): + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + ns_dir = base / "proj" + orphan = _orphan_worktree(ns_dir, "dream-foo") + + r = _run( + ["--repo-root", str(repo), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert not orphan.exists(), "orphan should be swept" + + def test_keeps_live_worktree(self, tmp_path): + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + live = base / "proj" / "dream-live" + live.parent.mkdir(parents=True) + subprocess.run( + ["git", "-C", str(repo), "worktree", "add", "-q", + str(live), "-b", "dream/proj/20260101-000000Z-live"], + check=True, env=_base_env(), + ) + # Force ancient mtime so --min-age-min doesn't save it. Only the + # orphan-check should save it. + ancient = 946684800 + os.utime(live, (ancient, ancient)) + + r = _run( + ["--repo-root", str(repo), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert live.is_dir(), "live worktree must NOT be swept" + + def test_min_age_skips_fresh_orphan(self, tmp_path): + """A fresh orphan (mtime = now) gets skipped — protects races + with `dream-setup.sh` that just created the dir but hasn't + finished registering the worktree yet.""" + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + fresh = _orphan_worktree(base / "proj", "dream-fresh", old=False) + # leave mtime at "now" + + r = _run( + ["--repo-root", str(repo), "--min-age-min", "5"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert fresh.is_dir(), "fresh dir must NOT be swept" + + def test_dry_run_removes_nothing(self, tmp_path): + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + orphan = _orphan_worktree(base / "proj", "dream-foo") + + r = _run( + ["--repo-root", str(repo), "--min-age-min", "0", "--dry-run"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert orphan.is_dir(), "dry-run must NOT remove" + assert "WOULD REMOVE" in r.stdout + + def test_skips_non_dream_dir(self, tmp_path): + """`<base>/proj/notdream` (no dream- prefix) must be ignored + outright — the shape check filters it before the gate.""" + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + non_dream = base / "proj" / "notdream-foo" + non_dream.mkdir(parents=True) + (non_dream / "data.txt").write_text("important\n") + ancient = 946684800 + os.utime(non_dream, (ancient, ancient)) + + r = _run( + ["--repo-root", str(repo), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert non_dream.is_dir(), "non-dream-prefix dir must be ignored" + + def test_orphan_with_missing_git_file(self, tmp_path): + """A dir with no .git at all is also considered orphan.""" + repo = _make_repo(tmp_path / "repo") + base = tmp_path / "wt-base" + d = base / "proj" / "dream-bare" + d.mkdir(parents=True) + (d / "stuff.txt").write_text("x\n") + ancient = 946684800 + os.utime(d, (ancient, ancient)) + + r = _run( + ["--repo-root", str(repo), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert not d.exists() + + +# =========================================================================== +# Regressions for the 5-model review panel findings (S1, C1) +# =========================================================================== + +class TestSafetyModuleMissingGc: + """C1: a missing _worktree_safety.py file makes python3 exit 2 — + the same code as the gate's success-but-missing-path. dream-gc.sh + must refuse to sweep when its gate is unloadable, not silently + sweep with no gate.""" + + def test_missing_safety_module_exits_4(self, tmp_path): + import shutil + broken = tmp_path / "broken" + broken.mkdir() + shutil.copy(GC_SH, broken / "dream-gc.sh") + # No _worktree_safety.py copied — gate is unloadable. + os.chmod(broken / "dream-gc.sh", 0o755) + + base = tmp_path / "wt-base" + wt = _orphan_worktree(base / "proj", "dream-foo", old=True) + assert wt.exists() + + r = subprocess.run( + ["bash", str(broken / "dream-gc.sh")], + capture_output=True, text=True, + env=_base_env({"DREAM_WORKTREE_BASE": str(base)}), + ) + assert r.returncode == 4, ( + f"missing safety must exit 4, got {r.returncode}\n" + f"stderr: {r.stderr}" + ) + assert "safety module" in r.stderr.lower() + # CRITICAL: nothing was removed. + assert wt.exists(), "orphan was swept without a safety gate!" + + +class TestGitdirParserRobust: + """S1: the old `awk -F': *'` parser split on every `:`, so a valid + gitdir line like `gitdir: /repo:with-colon/.git/worktrees/foo` got + truncated and the live worktree was misclassified as orphan and + DELETED. New parser must preserve `:` chars after the `gitdir: ` prefix. + """ + + def test_gitdir_path_with_colon_is_not_orphan(self, tmp_path): + # Build a fake target the parser will think exists. + gitdir_real = tmp_path / "container:with:colons" / "worktrees" / "foo" + gitdir_real.mkdir(parents=True) + # Build the candidate with a `.git` file pointing at the colon-laden path. + base = tmp_path / "wt-base" + candidate = base / "proj" / "dream-foo" + candidate.mkdir(parents=True) + (candidate / ".git").write_text(f"gitdir: {gitdir_real}\n") + # Age it so --min-age-min finds it. + ancient = 946684800 + os.utime(candidate, (ancient, ancient)) + + r = _run( + ["--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, f"stdout: {r.stdout}\nstderr: {r.stderr}" + # The candidate must NOT have been removed — its gitdir target exists. + assert candidate.exists(), ( + f"live worktree with colon-in-gitdir was DELETED — awk parser regression\n" + f"stdout: {r.stdout}" + ) + assert "removed=0" in r.stdout + assert "kept=1" in r.stdout + + def test_gitdir_with_crlf_endings_is_not_orphan(self, tmp_path): + """Opus 4.7-xhigh nit: CRLF endings would leave a trailing \\r + in the parsed gitdir, making `-e` falsely return false.""" + gitdir_real = tmp_path / "worktrees" / "foo" + gitdir_real.mkdir(parents=True) + base = tmp_path / "wt-base" + candidate = base / "proj" / "dream-foo" + candidate.mkdir(parents=True) + # Write the .git file with CRLF line endings. + (candidate / ".git").write_bytes(f"gitdir: {gitdir_real}\r\n".encode()) + ancient = 946684800 + os.utime(candidate, (ancient, ancient)) + + r = _run( + ["--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert candidate.exists(), ( + f"CRLF-terminated gitdir was misparsed → live worktree DELETED\n" + f"stdout: {r.stdout}" + ) + assert "kept=1" in r.stdout + + def test_gitdir_with_relative_path(self, tmp_path): + """Rare-but-legal: relative `gitdir:` paths must resolve relative + to the .git file's directory, not the cwd.""" + base = tmp_path / "wt-base" + candidate = base / "proj" / "dream-foo" + candidate.mkdir(parents=True) + # Build a real target adjacent to the candidate. + target = candidate / ".relative-gitdir" + target.mkdir() + (candidate / ".git").write_text("gitdir: .relative-gitdir\n") + ancient = 946684800 + os.utime(candidate, (ancient, ancient)) + + r = _run( + ["--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert candidate.exists(), ( + f"relative gitdir was treated as orphan → live worktree DELETED\n" + f"stdout: {r.stdout}" + ) + assert "kept=1" in r.stdout + + +# =========================================================================== +# --task-complete mode (Bug B fix from bug-cleanup-gaps.md) +# +# Round-2 review (5-model panel on commit 5ab361d) flagged that the first +# implementation of these tests planted FAKE `.git` files instead of using +# real `git worktree add`, so every assertion was satisfied by the dangerous +# `rm -rf` fallback rather than the polite `git worktree remove` path. The +# helper below uses a real `git worktree add` so: +# 1. `git worktree remove --force` actually succeeds in the happy path. +# 2. The cross-namespace and locked-worktree regression tests can plant +# worktrees that git ACTUALLY recognizes (and refuses to touch from +# the wrong repo, or when locked). +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +class TestTaskComplete: + """--task-complete mode sweeps registered worktrees in a single namespace. + + Bug B from bug-cleanup-gaps.md: worktrees that finished pushing but + never had `dream-cleanup.sh` called on them have VALID `.git` pointers + and so survive the standard orphan check forever. --task-complete + broadens the sweep to catch them. The round-2 review surfaced that + the original implementation walked the WHOLE base, which would have + swept other repos' live worktrees — so the flag is now strictly + namespace-scoped and refuses without `--namespace`. + """ + + def _real_registered_worktree( + self, + base: Path, + ns: str, + name: str, + repo: Path | None = None, + old: bool = True, + repos_root: Path | None = None, + ) -> tuple[Path, Path]: + """Plant a REAL `git worktree add`'d dream-* dir under <base>/<ns>. + + Returns (candidate_path, repo_path). When repo is None, creates a + fresh repo under `repos_root` (or `base.parent` if not given). + Uses a unique branch name so multiple worktrees per repo don't + collide. + """ + if repo is None: + parent = repos_root if repos_root is not None else base.parent + parent.mkdir(parents=True, exist_ok=True) + repo = parent / f"repo-{ns}" + if not (repo / ".git").exists(): + _make_repo(repo) + candidate = base / ns / name + candidate.parent.mkdir(parents=True, exist_ok=True) + branch = f"dream/{ns}/{name}" + env = _base_env() + subprocess.run( + ["git", "-C", str(repo), "worktree", "add", "-q", str(candidate), "-b", branch], + check=True, env=env, + ) + (candidate / "leftover.txt").write_text("stale\n") + if old: + ancient = 946684800 # 2000-01-01 + os.utime(candidate, (ancient, ancient)) + return candidate, repo + + # --- The happy path (real git worktree, polite remove succeeds) ---- + + def test_default_mode_keeps_registered_worktree(self, tmp_path): + """Sanity: default mode (no --task-complete) leaves registered dirs alone.""" + base = tmp_path / "wt-base" + candidate, _ = self._real_registered_worktree(base, ns="proj", name="dream-stale") + r = _run( + ["--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert candidate.exists(), ( + f"Default mode must NOT touch registered worktrees\nstdout: {r.stdout}" + ) + assert "kept=1" in r.stdout + + def test_task_complete_sweeps_registered_worktree_via_polite_path(self, tmp_path): + """--task-complete uses `git worktree remove --force` for registered dirs. + + Verifies BOTH: (1) the dir is gone, (2) git's bookkeeping was + updated (`git worktree list` no longer shows it). If the script + ever silently regressed to the unsafe `rm -rf` fallback, the + bookkeeping assertion would still pass file-existence but git + would still list it as "prunable" — so we assert it's gone from + the list entirely. + """ + base = tmp_path / "wt-base" + candidate, repo = self._real_registered_worktree(base, ns="proj", name="dream-stale") + r = _run( + ["--task-complete", "--namespace", "proj", "--repo-root", str(repo), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, ( + f"--task-complete should succeed\nstdout: {r.stdout}\nstderr: {r.stderr}" + ) + assert not candidate.exists(), ( + f"Registered worktree must be removed\nstdout: {r.stdout}\nstderr: {r.stderr}" + ) + assert "removed=1" in r.stdout + assert "stale-registered" in r.stdout + # Git's worktree list must no longer show it (polite path succeeded). + wt_list = subprocess.run( + ["git", "-C", str(repo), "worktree", "list", "--porcelain"], + capture_output=True, text=True, env=_base_env(), + ) + assert str(candidate) not in wt_list.stdout, ( + f"`git worktree list` still shows the path — polite remove must have failed\n" + f"list: {wt_list.stdout}\n" + f"gc-stdout: {r.stdout}\n" + f"gc-stderr: {r.stderr}" + ) + + def test_task_complete_sweeps_orphans_AND_registered(self, tmp_path): + """--task-complete is a superset: catches both orphans and registered.""" + base = tmp_path / "wt-base" + # Both must live under the same namespace now that scoping is enforced. + orphan = _orphan_worktree(base / "proj", name="dream-orphan") + registered, repo = self._real_registered_worktree(base, ns="proj", name="dream-reg") + r = _run( + ["--task-complete", "--namespace", "proj", "--repo-root", str(repo), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert not orphan.exists(), "orphan should be swept" + assert not registered.exists(), "registered should be swept" + assert "removed=2" in r.stdout + + def test_task_complete_respects_min_age(self, tmp_path): + """Fresh dirs (< --min-age-min) survive even in task-complete mode. + + NOTE: this is an mtime gate, NOT a liveness check. A real long-running + dream that simply hasn't written to disk in N minutes is still + eligible — the agent is responsible for asserting end-of-session. + """ + base = tmp_path / "wt-base" + candidate, repo = self._real_registered_worktree(base, ns="proj", name="dream-fresh", old=False) + r = _run( + ["--task-complete", "--namespace", "proj", "--repo-root", str(repo), "--min-age-min", "1"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert candidate.exists(), ( + f"Fresh dir must survive --min-age-min\nstdout: {r.stdout}" + ) + assert "removed=0" in r.stdout + + def test_task_complete_dry_run_only_logs(self, tmp_path): + """--task-complete + --dry-run logs but removes nothing.""" + base = tmp_path / "wt-base" + candidate, repo = self._real_registered_worktree(base, ns="proj", name="dream-stale") + r = _run( + ["--task-complete", "--namespace", "proj", "--repo-root", str(repo), "--dry-run", "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0 + assert candidate.exists(), "Dry-run must NOT remove" + assert "WOULD REMOVE" in r.stdout + assert "stale-registered" in r.stdout + + def test_task_complete_safety_gate_still_holds(self, tmp_path): + """Safety gate cannot be bypassed by --task-complete. + + Even an asserted task_complete sweep must refuse a sensitive base. + """ + r = _run( + ["--task-complete", "--namespace", "proj", "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": "/tmp"}, + ) + assert r.returncode == 1, "Sensitive base refused regardless of mode" + + # --- Round-2 regressions (the panel's blocker + criticals) ----------- + + def test_task_complete_requires_namespace(self, tmp_path): + """`--task-complete` without a namespace must exit 2 (not silently sweep all). + + BLOCKER from the 5-model panel: the prior implementation walked the + WHOLE `$DREAM_WORKTREE_BASE`, so an end-of-session sweep in repo A + would delete repo B's live worktrees in a shared-base fleet. The + script now refuses unless a namespace can be resolved. + """ + base = tmp_path / "wt-base" + base.mkdir() + r = _run( + ["--task-complete", "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + # No namespace, no DREAM_NAMESPACE env, no --repo-root → refuse. + assert r.returncode == 2, ( + f"Expected refusal exit 2\nstdout: {r.stdout}\nstderr: {r.stderr}" + ) + assert "requires --namespace" in r.stderr or "Refusing" in r.stderr + + def test_task_complete_refuses_unsafe_namespace_chars(self, tmp_path): + """Namespace input is validated before reaching `find`.""" + base = tmp_path / "wt-base" + base.mkdir() + r = _run( + ["--task-complete", "--namespace", "../escape", "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 2 + assert "namespace must match" in r.stderr.lower() or "--namespace" in r.stderr + + def test_task_complete_does_NOT_cross_namespaces(self, tmp_path): + """Cross-namespace data-loss prevention: nsA's task-complete must NEVER touch nsB. + + This is the round-2 BLOCKER from all 5 reviewers — reproduced + empirically against the prior implementation. Two real + git-worktree-add'd dreams in two namespaces, A's sweep must leave + B intact. + """ + base = tmp_path / "wt-base" + cand_a, repo_a = self._real_registered_worktree(base, ns="nsA", name="dream-a") + cand_b, repo_b = self._real_registered_worktree(base, ns="nsB", name="dream-b") + r = _run( + ["--task-complete", "--namespace", "nsA", "--repo-root", str(repo_a), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, f"stderr: {r.stderr}" + assert not cand_a.exists(), "nsA dream should be removed" + assert cand_b.exists(), ( + f"nsB dream MUST survive nsA's sweep — cross-namespace data loss!\n" + f"stdout: {r.stdout}\nstderr: {r.stderr}" + ) + # And git's bookkeeping for repoB is intact. + list_b = subprocess.run( + ["git", "-C", str(repo_b), "worktree", "list", "--porcelain"], + capture_output=True, text=True, env=_base_env(), + ) + assert str(cand_b) in list_b.stdout + + def test_task_complete_refuses_locked_worktree_no_rm_fallback(self, tmp_path): + """If `git worktree remove --force` refuses (locked), we WARN and skip. + + CRITICAL from the round-2 review: the prior code fell through to + `rm -rf` whenever git failed, silently destroying deliberately + locked worktrees. New behavior: WARN, refuse, leave the dir. + """ + base = tmp_path / "wt-base" + candidate, repo = self._real_registered_worktree(base, ns="proj", name="dream-locked") + # Lock it via git — `--force` cannot override a lock. + subprocess.run( + ["git", "-C", str(repo), "worktree", "lock", str(candidate), "--reason", "test"], + check=True, env=_base_env(), + ) + r = _run( + ["--task-complete", "--namespace", "proj", "--repo-root", str(repo), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, f"stderr: {r.stderr}" + # Locked worktree MUST survive — no rm -rf fallback. + assert candidate.exists(), ( + f"Locked worktree must not be force-removed via rm fallback\n" + f"stdout: {r.stdout}\nstderr: {r.stderr}" + ) + # And the script logged a WARN for diagnosability. + assert "WARN" in r.stderr + assert "refused=1" in r.stdout + + def test_task_complete_skips_other_repos_worktree_no_rm_fallback(self, tmp_path): + """Defense-in-depth: even if scoping were bypassed, the rm fallback + no longer destroys worktrees registered with a DIFFERENT repo. + + Plants nsA's worktree but invokes --task-complete with a UNRELATED + repo's --repo-root. `git worktree remove` errors out ("not a working + tree"); the script must NOT fall through to rm -rf. + """ + base = tmp_path / "wt-base" + candidate, repo_owner = self._real_registered_worktree(base, ns="nsA", name="dream-owned") + # An unrelated repo — must be a valid git repo so `_is_git_root` passes + # and we actually exercise the `git worktree remove` failure path. + other_repo = tmp_path / "other-repo" + _make_repo(other_repo) + r = _run( + ["--task-complete", "--namespace", "nsA", "--repo-root", str(other_repo), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, f"stderr: {r.stderr}" + assert candidate.exists(), ( + f"Worktree owned by another repo MUST survive\n" + f"stdout: {r.stdout}\nstderr: {r.stderr}" + ) + assert "WARN" in r.stderr + # And the rightful owner's bookkeeping is still intact. + list_owner = subprocess.run( + ["git", "-C", str(repo_owner), "worktree", "list", "--porcelain"], + capture_output=True, text=True, env=_base_env(), + ) + assert str(candidate) in list_owner.stdout + + def test_task_complete_dream_namespace_env_works(self, tmp_path): + """DREAM_NAMESPACE env satisfies the --namespace requirement.""" + base = tmp_path / "wt-base" + candidate, repo = self._real_registered_worktree(base, ns="env-ns", name="dream-x") + r = _run( + ["--task-complete", "--repo-root", str(repo), "--min-age-min", "0"], + env_extra={ + "DREAM_WORKTREE_BASE": str(base), + "DREAM_NAMESPACE": "env-ns", + }, + ) + assert r.returncode == 0, f"stderr: {r.stderr}" + assert not candidate.exists() + + def test_task_complete_does_not_auto_derive_from_repo_root_basename(self, tmp_path): + """`--task-complete --repo-root <repo>` alone must NOT derive ns from basename. + + REPO_ROOT can itself be inferred from cwd, so deriving namespace + from `basename "$REPO_ROOT"` could silently sweep the wrong + namespace. The script must require an explicit --namespace or + DREAM_NAMESPACE env. + """ + base = tmp_path / "wt-base" + repo = tmp_path / "some-repo" + _make_repo(repo) + r = _run( + ["--task-complete", "--repo-root", str(repo), "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 2, ( + f"Expected refusal (no --namespace, no DREAM_NAMESPACE)\n" + f"stdout: {r.stdout}\nstderr: {r.stderr}" + ) + assert "requires --namespace" in r.stderr + + def test_task_complete_min_age_min_0_catches_fresh_worktree(self, tmp_path): + """Regression: `--min-age-min 0` MUST sweep a worktree just created. + + Previously the find walk passed `-mmin "+$MIN_AGE_MIN"` unconditionally. + Bash arithmetic note: `find -mmin +0` matches files modified MORE than + 0 minutes ago — i.e. it EXCLUDES anything touched in the last ~1 min. + End-of-session cleanup (`--min-age-min 0` per SKILL.md:1102-1115) is + specifically intended to catch the freshly-pushed final batch, so the + prior implementation silently kept the very leak it was supposed to + catch. Reviewers GPT-5.5 and Claude Opus 4.8 both flagged this. + + This test creates a registered worktree with a FRESH mtime (no + backdating via os.utime), then invokes --task-complete --min-age-min 0 + and asserts the dir is gone. + """ + base = tmp_path / "wt-base" + # old=False ⇒ mtime is "now", so any positive -mmin gate would skip it. + candidate, repo = self._real_registered_worktree( + base, ns="proj", name="dream-fresh", old=False, + ) + assert candidate.exists() + r = _run( + ["--task-complete", "--namespace", "proj", "--repo-root", str(repo), + "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 0, f"stderr: {r.stderr}" + assert not candidate.exists(), ( + "Fresh registered worktree MUST be swept with --min-age-min 0 — " + "otherwise end-of-session cleanup silently misses the final batch.\n" + f"stdout: {r.stdout}\nstderr: {r.stderr}" + ) + + def test_task_complete_refuses_bare_dot_namespace(self, tmp_path): + """Defense-in-depth: `--namespace .` is rejected at input validation. + + The depth-1 + dream-* filename filter + _worktree_safety.py gate + already prevent damage, but tightening the regex to require a + non-`.` first char makes the contract explicit (Opus 4.7-xhigh, + Opus 4.8, Gemini all flagged the loose regex as a nit). + """ + base = tmp_path / "wt-base" + (base / "proj").mkdir(parents=True) + r = _run( + ["--task-complete", "--namespace", ".", "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 2 + assert "namespace must match" in r.stderr.lower() + + def test_task_complete_refuses_double_dot_namespace(self, tmp_path): + """`--namespace ..` must be rejected — it would otherwise resolve the + walk root to `<base>/..` (the parent of the base).""" + base = tmp_path / "wt-base" + base.mkdir(parents=True) + r = _run( + ["--task-complete", "--namespace", "..", "--min-age-min", "0"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 2 + assert "namespace must match" in r.stderr.lower() + + def test_task_complete_rejects_non_integer_min_age_min(self, tmp_path): + """`--min-age-min` must be a non-negative integer. + + Previously this flowed straight to `find -mmin "+$MIN_AGE_MIN"` and + relied on find's error message. With the new `if [[ "$MIN_AGE_MIN" + -gt 0 ]]` branching for the -mmin +0 fix, an unvalidated string + would also break bash arithmetic. + """ + base = tmp_path / "wt-base" + (base / "proj").mkdir(parents=True) + r = _run( + ["--task-complete", "--namespace", "proj", "--min-age-min", "abc"], + env_extra={"DREAM_WORKTREE_BASE": str(base)}, + ) + assert r.returncode == 2 + assert "min-age-min" in r.stderr.lower() diff --git a/tests/skills/shadow_frog_dream/test_dream_reconcile.py b/tests/skills/shadow_frog_dream/test_dream_reconcile.py new file mode 100644 index 0000000..288d7fa --- /dev/null +++ b/tests/skills/shadow_frog_dream/test_dream_reconcile.py @@ -0,0 +1,3277 @@ +"""Tests for skills/shadow-frog-dream/dream-reconcile.py. + +Covers the high-leverage internal helpers (heading scan, dedup, bidirectional +back-pointers, manifest/index parsers, verification, top-index rebuild) and a +couple of CLI smoke paths via subprocess. Heavy integration paths +(`merge_discoveries`, `mirror_reports`, `update_index`, `update_state`, +`cleanup_branches` actual delete, `main`) are exercised indirectly through the +unit-level helpers they delegate to plus the CLI smoke tests. + +Regression coverage: +* B4 — `add_cross_reference_backpointer` idempotency + false-positive substring. +* B5 — `find_cross_references_heading` case-insensitive match. +* B6 — `rebuild_top_index` reproduces the committed coupon-demo header. +* PREFIX-FALSE-PASS — `verify_reconciliation` and `cleanup_branches` must not + confuse a short dream_id with a longer indexed one. +""" +import json +import os +import re +import subprocess +import sys +from pathlib import Path + +import pytest + + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent +SCRIPT = REPO_ROOT / "skills" / "shadow-frog-dream" / "dream-reconcile.py" + + +# --- Git env / helpers --------------------------------------------------- + +def _git_env(home: Path) -> dict: + return { + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "HOME": str(home), + "PATH": "/usr/bin:/bin:/usr/local/bin:/opt/homebrew/bin", + } + + +def _run(cmd, cwd, env=None, check=True): + return subprocess.run( + cmd, cwd=cwd, env=env, capture_output=True, text=True, check=check + ) + + +def _git(*args, cwd, env, check=True): + return _run(["git", *args], cwd=cwd, env=env, check=check) + + +def _seed_repo(repo: Path): + """Seed an empty initialized git repo with one commit and a bare origin.""" + env = _git_env(repo) + (repo / "README.md").write_text("# test\n") + _git("add", "-A", cwd=repo, env=env) + _git("commit", "-q", "-m", "init", cwd=repo, env=env) + return env + + +def _add_bare_remote(repo: Path, env: dict) -> Path: + """Create a bare repo alongside `repo` and wire it up as origin/main.""" + bare = repo.parent / f"{repo.name}.git" + _git("init", "--bare", "-q", str(bare), cwd=repo.parent, env=env) + _git("remote", "add", "origin", f"file://{bare}", cwd=repo, env=env) + _git("push", "-q", "-u", "origin", "main", cwd=repo, env=env) + # Mark origin/HEAD so cleanup_branches default-branch detection works. + _git("symbolic-ref", "refs/remotes/origin/HEAD", "refs/remotes/origin/main", + cwd=repo, env=env) + return bare + + +def make_dream_branch( + repo: Path, + env: dict, + dream_ns: str, + dream_id: str, + manifest: dict, + report: str | None = None, + patch: str | None = "diff --git a/x b/x\n", + extra_files: dict[str, str] | None = None, +) -> str: + """Create + push a dream branch from main with the given artifacts. + + Returns the short branch name. Leaves the repo checked out on its + original branch. + """ + branch = f"dream/{dream_ns}/{dream_id}" + original = _git("rev-parse", "--abbrev-ref", "HEAD", + cwd=repo, env=env).stdout.strip() + + _git("checkout", "-q", "-b", branch, cwd=repo, env=env) + + dream_dir = repo / ".shadow" / "_dreams" / dream_id + dream_dir.mkdir(parents=True, exist_ok=True) + (dream_dir / "manifest.json").write_text(json.dumps(manifest, indent=2)) + if report is not None: + (dream_dir / "report.md").write_text(report) + if patch is not None: + (dream_dir / "patch.diff").write_text(patch) + + if extra_files: + for rel, content in extra_files.items(): + target = repo / rel + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(content) + + _git("add", "-A", cwd=repo, env=env) + _git("commit", "-q", "-m", f"dream: {dream_id}", cwd=repo, env=env) + _git("push", "-q", "origin", branch, cwd=repo, env=env) + + _git("checkout", "-q", original, cwd=repo, env=env) + return branch + + +def _default_report(dream_id: str, base_commit: str = "abcdef1234567890") -> str: + return ( + "---\n" + f"dream_id: \"{dream_id}\"\n" + "category: bug hunting\n" + "verdict: useful\n" + f"base_commit: {base_commit}\n" + "---\n\n" + f"# {dream_id}\n\nBody.\n" + ) + + +def _default_manifest(dream_id: str, dream_ns: str = "proj") -> dict: + return { + "dream_id": dream_id, + "branch": f"dream/{dream_ns}/{dream_id}", + "parent_branch": "main", + "category": "bug hunting", + "verdict": "useful", + "title": f"Title for {dream_id}", + "discoveries": [], + "cross_cutting": [], + } + + +# =========================================================================== +# find_cross_references_heading — B5 case-insensitive regression +# =========================================================================== + +@pytest.mark.parametrize("heading", [ + "## Cross-References", + "## cross-references", + "## Cross-references", + "## CROSS-REFERENCES", +]) +def test_find_cross_references_heading_is_case_insensitive(dream_reconcile, heading): + lines = ["# Shadow\n", "\n", "## `foo`\n", "\n", "- discovery\n", "\n", heading + "\n"] + assert dream_reconcile.find_cross_references_heading(lines) == 6 + + +def test_find_cross_references_heading_missing_returns_negative_one(dream_reconcile): + lines = ["# Shadow\n", "## `foo`\n", "- d\n"] + assert dream_reconcile.find_cross_references_heading(lines) == -1 + + +def test_find_cross_references_heading_ignores_trailing_whitespace(dream_reconcile): + # The heading may have trailing whitespace before the newline. + lines = ["## Cross-References \n"] + assert dream_reconcile.find_cross_references_heading(lines) == 0 + + +# =========================================================================== +# is_duplicate_discovery +# =========================================================================== + +def test_is_duplicate_discovery_exact_match(dream_reconcile): + existing = ["- The cache silently fails on expired tokens\n"] + assert dream_reconcile.is_duplicate_discovery( + existing, "The cache silently fails on expired tokens" + ) is True + + +def test_is_duplicate_discovery_exact_match_ignores_whitespace_and_case(dream_reconcile): + existing = ["- The Cache Silently fails on Expired Tokens\n"] + assert dream_reconcile.is_duplicate_discovery( + existing, "the cache silently fails on expired tokens" + ) is True + + +def test_is_duplicate_discovery_short_distinct_keyword_not_merged(dream_reconcile): + """Short discoveries differing by ONE keyword must NOT be deduped.""" + existing = ["- returns None on expired tokens\n"] + new = "returns None on revoked tokens" + assert dream_reconcile.is_duplicate_discovery(existing, new) is False + + +def test_is_duplicate_discovery_long_high_overlap_is_merged(dream_reconcile): + """Fuzzy match: 20 words, 1 swapped → 19/20 = 0.95 → merge.""" + existing_words = ( + "alpha beta gamma delta epsilon zeta eta theta iota kappa " + "lambda mu nu xi omicron pi rho sigma tau upsilon" + ).split() + new_words = existing_words[:-1] + ["DIFFERENT_WORD"] + existing = ["- " + " ".join(existing_words) + "\n"] + new = " ".join(new_words) + # overlap = |intersection| / |new_set| = 19/20 = 0.95 ≥ DEDUP_THRESHOLD + assert dream_reconcile.is_duplicate_discovery(existing, new) is True + + +def test_is_duplicate_discovery_completely_different_text(dream_reconcile): + existing = ["- alpha beta gamma delta epsilon zeta eta theta iota kappa lambda mu nu\n"] + new = "totally unrelated content that shares no words at all here entirely now" + assert dream_reconcile.is_duplicate_discovery(existing, new) is False + + +def test_is_duplicate_discovery_skips_non_bullet_lines(dream_reconcile): + existing = ["## `foo`\n", " _(verified, source: exploration)_\n"] + assert dream_reconcile.is_duplicate_discovery(existing, "something new") is False + + +def test_is_duplicate_discovery_empty_new_text(dream_reconcile): + assert dream_reconcile.is_duplicate_discovery(["- whatever\n"], "") is False + + +# =========================================================================== +# _ensure_cross_references_section +# =========================================================================== + +def test_ensure_cross_references_appends_when_missing(dream_reconcile): + lines = ["# Shadow\n", "\n", "## `foo`\n", "- d\n"] + out = dream_reconcile._ensure_cross_references_section(list(lines)) + joined = "".join(out) + assert "## Cross-References" in joined + assert "_No cross-cutting discoveries yet._" in joined + + +def test_ensure_cross_references_is_idempotent(dream_reconcile): + lines = ["## `foo`\n", "## Cross-References\n", "\n", "_No cross-cutting discoveries yet._\n"] + out = dream_reconcile._ensure_cross_references_section(list(lines)) + # Section count should remain 1. + assert "".join(out).count("## Cross-References") == 1 + + +def test_ensure_cross_references_finds_lowercase_heading(dream_reconcile): + """A lowercase rewrite must NOT cause a duplicate section append.""" + lines = ["## `foo`\n", "## cross-references\n", "- [link](x.md)\n"] + out = dream_reconcile._ensure_cross_references_section(list(lines)) + assert "".join(out).count("## Cross-References") == 0 # not added + assert "".join(out).count("## cross-references") == 1 + + +# =========================================================================== +# add_cross_reference_backpointer — B4 regression +# =========================================================================== + +def test_add_cross_reference_backpointer_bootstraps_missing_shadow(dream_reconcile, tmp_path): + repo = tmp_path + (repo / ".shadow").mkdir() + + written = dream_reconcile.add_cross_reference_backpointer( + str(repo), "src/foo.py", "my-slug", "My Slug Title", "20260101-000000Z-x" + ) + assert written is True + + shadow = repo / ".shadow" / "src" / "foo.py.md" + assert shadow.is_file() + text = shadow.read_text() + assert "## File-Level" in text + assert "## Cross-References" in text + # depth=1 for src/foo.py → prefix "../" + assert "[My Slug Title](../_cross/my-slug.md)" in text + assert "dream: 20260101-000000Z-x" in text + + +def test_add_cross_reference_backpointer_is_idempotent(dream_reconcile, tmp_path): + """Calling twice must not duplicate the back-pointer line (B4).""" + repo = tmp_path + (repo / ".shadow").mkdir() + + dream_reconcile.add_cross_reference_backpointer( + str(repo), "foo.py", "slug-x", "Title X", "20260101-000000Z-x" + ) + written2 = dream_reconcile.add_cross_reference_backpointer( + str(repo), "foo.py", "slug-x", "Title X", "20260101-000000Z-x" + ) + assert written2 is False + + shadow = repo / ".shadow" / "foo.py.md" + body = shadow.read_text() + assert body.count("[Title X](_cross/slug-x.md)") == 1 + + +def test_add_cross_reference_backpointer_false_positive_substring(dream_reconcile, tmp_path): + """A discovery body that MENTIONS `_cross/foo.md` (no link) used to falsely + inhibit the back-pointer. The function must add the proper link. + """ + repo = tmp_path + (repo / ".shadow").mkdir() + shadow = repo / ".shadow" / "bar.py.md" + shadow.write_text( + "## `bar`\n" + "\n" + "- A discovery that talks about `_cross/foo.md` but is not a link.\n" + " _(verified, source: exploration)_\n" + "\n" + "## Cross-References\n" + "\n" + "_No cross-cutting discoveries yet._\n" + ) + + written = dream_reconcile.add_cross_reference_backpointer( + str(repo), "bar.py", "foo", "Foo Title", "20260101-000000Z-y" + ) + assert written is True + + body = shadow.read_text() + assert "[Foo Title](_cross/foo.md)" in body + # Discovery line preserved. + assert "talks about `_cross/foo.md`" in body + + +@pytest.mark.parametrize("file_part,expected_prefix,expected_dir", [ + ("foo.py", "", "."), + ("src/foo.py", "../", "src"), + ("a/b/c/foo.py", "../../../", "a/b/c"), +]) +def test_add_cross_reference_backpointer_depth_math( + dream_reconcile, tmp_path, file_part, expected_prefix, expected_dir +): + repo = tmp_path + (repo / ".shadow").mkdir() + dream_reconcile.add_cross_reference_backpointer( + str(repo), file_part, "slug", "T", "20260101-000000Z-z" + ) + shadow = repo / ".shadow" / (file_part + ".md") + assert shadow.is_file() + expected = f"[T]({expected_prefix}_cross/slug.md)" + assert expected in shadow.read_text() + + +def test_add_cross_reference_backpointer_replaces_placeholder( + dream_reconcile, tmp_path +): + repo = tmp_path + (repo / ".shadow").mkdir() + shadow = repo / ".shadow" / "foo.py.md" + shadow.write_text( + "## `foo`\n\n- d\n\n## Cross-References\n\n_No cross-cutting discoveries yet._\n" + ) + dream_reconcile.add_cross_reference_backpointer( + str(repo), "foo.py", "slug-y", "Y", "20260101-000000Z-q" + ) + body = shadow.read_text() + assert "_No cross-cutting discoveries yet._" not in body + assert "[Y](_cross/slug-y.md)" in body + + +# =========================================================================== +# merge_discovery_into_file +# =========================================================================== + +def test_merge_discovery_creates_new_shadow(dream_reconcile, tmp_path): + shadow = tmp_path / ".shadow" / "src" / "foo.py.md" + written = dream_reconcile.merge_discovery_into_file( + str(shadow), + "do_thing", + {"text": "Returns None on empty input.", + "status": "verified", "source": "exploration"}, + "20260101-000000Z-x", + ) + assert written is True + assert shadow.is_file() + body = shadow.read_text() + assert "## `do_thing`" in body + assert "- Returns None on empty input." in body + assert "_(verified, source: exploration)_" in body + assert "Dream report: `_dreams/20260101-000000Z-x/`" in body + assert "## Cross-References" in body + + +def test_merge_discovery_replaces_no_discoveries_placeholder(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n_No discoveries yet._\n\n## Cross-References\n\n_No cross-cutting discoveries yet._\n" + ) + dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + {"text": "Actual discovery.", "status": "verified", + "source": "exploration"}, + "20260101-000000Z-x", + ) + body = shadow.read_text() + assert "_No discoveries yet._" not in body + assert "- Actual discovery." in body + + +def test_merge_discovery_appends_after_existing_bullets(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n- existing discovery\n _(verified, source: exploration)_\n\n" + "## Cross-References\n\n_No cross-cutting discoveries yet._\n" + ) + dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + {"text": "Another discovery.", "status": "verified", + "source": "exploration"}, + "20260101-000000Z-x", + ) + body = shadow.read_text() + assert "- existing discovery" in body + assert "- Another discovery." in body + # Discovery order preserved. + assert body.index("existing discovery") < body.index("Another discovery") + + +def test_merge_discovery_skips_duplicate(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n- already here\n _(verified, source: exploration)_\n\n" + "## Cross-References\n\n_No cross-cutting discoveries yet._\n" + ) + written = dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + {"text": "already here", "status": "verified", "source": "exploration"}, + "20260101-000000Z-x", + ) + assert written is False + + +def test_merge_discovery_empty_text_returns_false(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + written = dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", {"text": " "}, "20260101-000000Z-x" + ) + assert written is False + assert not shadow.exists() + + +# =========================================================================== +# B15 — exact-text duplicate metadata merge (union labels, upgrade source +# trust, upgrade uncertain->verified; never touch refuted) +# =========================================================================== + +def _xref_footer(): + return "\n## Cross-References\n\n_No cross-cutting discoveries yet._\n" + + +def test_merge_discovery_unions_labels_on_exact_match(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n- claim text here\n _(verified, source: exploration, labels: [bug])_\n" + + _xref_footer() + ) + written = dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + {"text": "claim text here", "status": "verified", + "source": "exploration", "labels": ["security"]}, + "20260101-000000Z-x", + ) + assert written is True + body = shadow.read_text() + # Single merged discovery line (no duplicate appended). + assert body.count("- claim text here") == 1 + assert "labels: [bug, security]" in body + + +def test_merge_discovery_upgrades_source_trust(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n- some claim\n _(verified, source: exploration)_\n" + + _xref_footer() + ) + written = dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + {"text": "some claim", "status": "verified", "source": "user"}, + "20260101-000000Z-x", + ) + assert written is True + body = shadow.read_text() + assert "source: user" in body + assert "source: exploration" not in body + + +def test_merge_discovery_upgrades_uncertain_to_verified(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n- a claim\n _(uncertain, source: exploration)_\n" + + _xref_footer() + ) + written = dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + {"text": "a claim", "status": "verified", "source": "exploration"}, + "20260101-000000Z-x", + ) + assert written is True + body = shadow.read_text() + assert "_(verified, source: exploration)_" in body + + +def test_merge_discovery_never_downgrades_verified(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n- a claim\n _(verified, source: exploration)_\n" + + _xref_footer() + ) + written = dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + {"text": "a claim", "status": "uncertain", "source": "exploration"}, + "20260101-000000Z-x", + ) + # Nothing to upgrade → no write. + assert written is False + assert "_(verified, source: exploration)_" in shadow.read_text() + + +def test_merge_discovery_never_touches_refuted(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n- a claim\n _(refuted, source: exploration)_\n" + + _xref_footer() + ) + # New says verified — but refuted is a deliberate signal; status must hold. + written = dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + {"text": "a claim", "status": "verified", "source": "exploration", + "labels": ["bug"]}, + "20260101-000000Z-x", + ) + body = shadow.read_text() + # Status stays refuted; labels may still union. + assert "refuted" in body + assert "verified" not in body + + +def test_merge_discovery_fuzzy_match_still_skips(dream_reconcile, tmp_path): + """Fuzzy (non-exact) near-duplicates must NOT trigger a metadata merge — + they may be genuinely different claims.""" + existing = ("- the function returns none when the input list is " + "completely empty or missing entirely") + shadow = tmp_path / "foo.md" + shadow.write_text( + f"## `foo`\n\n{existing}\n _(uncertain, source: exploration)_\n" + + _xref_footer() + ) + # Same words minus one — high overlap but not exact. + new_text = ("the function returns none when the input list is " + "completely empty or missing") + written = dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + {"text": new_text, "status": "verified", "source": "exploration"}, + "20260101-000000Z-x", + ) + # Treated as fuzzy duplicate → skipped, metadata untouched. + assert written is False + assert "_(uncertain, source: exploration)_" in shadow.read_text() + + +class TestMetaMergeHelpers: + def test_parse_meta_line_full(self, dream_reconcile): + assert dream_reconcile._parse_meta_line( + " _(verified, source: user, labels: [bug, security])_\n" + ) == ("verified", "user", ["bug", "security"]) + + def test_parse_meta_line_no_labels(self, dream_reconcile): + assert dream_reconcile._parse_meta_line( + " _(uncertain, source: exploration)_" + ) == ("uncertain", "exploration", []) + + def test_parse_meta_line_non_meta_returns_none(self, dream_reconcile): + assert dream_reconcile._parse_meta_line("- not a meta line") is None + + def test_merge_meta_unions_and_upgrades(self, dream_reconcile): + status, source, labels, changed = dream_reconcile._merge_meta( + ("uncertain", "exploration", ["bug"]), "verified", "user", ["security"] + ) + assert (status, source, labels) == ("verified", "user", ["bug", "security"]) + assert changed is True + + def test_merge_meta_no_change(self, dream_reconcile): + status, source, labels, changed = dream_reconcile._merge_meta( + ("verified", "user", ["bug"]), "verified", "user", ["bug"] + ) + assert changed is False + + def test_merge_meta_refuted_untouched(self, dream_reconcile): + status, _, _, _ = dream_reconcile._merge_meta( + ("refuted", "exploration", []), "verified", "exploration", [] + ) + assert status == "refuted" + + +def test_merge_discovery_with_also_involves_and_labels(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + dream_reconcile.merge_discovery_into_file( + str(shadow), "foo", + { + "text": "Something with refs.", "status": "verified", + "source": "exploration", "labels": ["bug", "security"], + "also_involves": ["other.py::thing", "more.py::stuff"], + }, + "20260101-000000Z-x", + ) + body = shadow.read_text() + assert "labels: [bug, security]" in body + assert "Also involves: `other.py::thing`, `more.py::stuff`" in body + + +# =========================================================================== +# _read_indexed_dream_ids / _read_indexed_branches +# =========================================================================== + +INDEX_FIXTURE = """# Dream Experiments + +| dream_id | category | verdict | title | branch | parent | tip_commit | +|----------|----------|---------|-------|--------|--------|------------| +| 20260101-000000Z-alpha | bug hunting | useful | First | dream/p/20260101-000000Z-alpha | main | 1234567 | +| 20260102-000000Z-beta | investigation | useful | Second | dream/p/20260102-000000Z-beta | main | abcdef0 | +| 20260103-000000Z-gamma | feature design | dead_end | Third | dream/p/20260103-000000Z-gamma | main | 9876543 | +""" + + +def _write_index(repo: Path, content: str = INDEX_FIXTURE): + idx = repo / ".shadow" / "_dreams" / "_index.md" + idx.parent.mkdir(parents=True, exist_ok=True) + idx.write_text(content) + + +def test_read_indexed_dream_ids_parses_table(dream_reconcile, tmp_path): + _write_index(tmp_path) + ids = dream_reconcile._read_indexed_dream_ids(str(tmp_path)) + assert ids == { + "20260101-000000Z-alpha", + "20260102-000000Z-beta", + "20260103-000000Z-gamma", + } + + +def test_read_indexed_dream_ids_returns_empty_when_missing(dream_reconcile, tmp_path): + assert dream_reconcile._read_indexed_dream_ids(str(tmp_path)) == set() + + +def test_read_indexed_branches_parses_table(dream_reconcile, tmp_path): + _write_index(tmp_path) + rows = dream_reconcile._read_indexed_branches(str(tmp_path)) + assert ("dream/p/20260101-000000Z-alpha", "20260101-000000Z-alpha") in rows + assert ("dream/p/20260102-000000Z-beta", "20260102-000000Z-beta") in rows + assert len(rows) == 3 + + +# =========================================================================== +# verify_reconciliation — PREFIX FALSE-PASS regression +# =========================================================================== + +def _seed_dream_artifacts(repo: Path, dream_id: str): + d = repo / ".shadow" / "_dreams" / dream_id + d.mkdir(parents=True, exist_ok=True) + (d / "report.md").write_text(f"# {dream_id}\n") + (d / "manifest.json").write_text("{}") + (d / "patch.diff").write_text("diff\n") + + +def test_verify_reconciliation_passes_for_indexed_dream(dream_reconcile, tmp_path): + dream_id = "20260101-000000Z-alpha" + _write_index(tmp_path) + _seed_dream_artifacts(tmp_path, dream_id) + manifests = [(f"dream/p/{dream_id}", dream_id, {})] + assert dream_reconcile.verify_reconciliation(str(tmp_path), manifests) == [] + + +def test_verify_reconciliation_prefix_does_not_false_pass(dream_reconcile, tmp_path): + """A SHORTER manifest dream_id whose longer relative IS indexed + must still be reported as missing from the index. Naive substring + matching would have falsely passed this case. + """ + short_id = "20260420-1400-foo" + longer_id = "20260420-1400-foo-extended" # contains `short_id` as a substring + _write_index( + tmp_path, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {longer_id} | bug hunting | useful | Long | dream/p/{longer_id} | main | abc1234 |\n", + ) + _seed_dream_artifacts(tmp_path, short_id) # artifacts OK, but index lacks it + manifests = [(f"dream/p/{short_id}", short_id, {})] + failures = dream_reconcile.verify_reconciliation(str(tmp_path), manifests) + assert any("missing from _index.md" in f for f in failures), failures + + +def test_verify_reconciliation_reports_missing_artifacts(dream_reconcile, tmp_path): + dream_id = "20260101-000000Z-alpha" + _write_index(tmp_path) + # Only create report.md — manifest.json and patch.diff missing. + d = tmp_path / ".shadow" / "_dreams" / dream_id + d.mkdir(parents=True, exist_ok=True) + (d / "report.md").write_text("# r\n") + manifests = [(f"dream/p/{dream_id}", dream_id, {})] + failures = dream_reconcile.verify_reconciliation(str(tmp_path), manifests) + assert any("manifest.json" in f for f in failures) + assert any("patch.diff" in f for f in failures) + + +def test_verify_reconciliation_reports_missing_index(dream_reconcile, tmp_path): + dream_id = "20260101-000000Z-alpha" + _seed_dream_artifacts(tmp_path, dream_id) + manifests = [(f"dream/p/{dream_id}", dream_id, {})] + failures = dream_reconcile.verify_reconciliation(str(tmp_path), manifests) + assert any("_index.md does not exist" in f for f in failures) + + +# =========================================================================== +# cleanup_branches — PREFIX FALSE-PASS regression (data loss class) +# =========================================================================== + +@pytest.mark.slow +def test_cleanup_branches_keeps_prefix_only_branch_to_prevent_data_loss( + dream_reconcile, tmp_git_repo +): + """If the manifest dream_id is a PREFIX of an indexed ID (but not equal), + the branch must NOT be deleted — pre-fix substring check would have + falsely passed and destroyed the only copy of those discoveries. + """ + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + + short_id = "20260420-1400-foo" + longer_id = "20260420-1400-foo-extended" + + # Index contains only the LONGER id. + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {longer_id} | bug hunting | useful | Long | dream/proj/{longer_id} | main | abc1234 |\n", + ) + + # Mirror artifacts for SHORT id (so safety check 1 passes). + _seed_dream_artifacts(tmp_git_repo, short_id) + + manifests = [(f"dream/proj/{short_id}", short_id, {})] + + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), manifests, "proj", dry_run=True + ) + assert deleted == 0 + assert kept == 1 + + +@pytest.mark.slow +def test_cleanup_branches_keeps_when_artifacts_missing( + dream_reconcile, tmp_git_repo +): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + + dream_id = "20260420-141500Z-missing-artifacts" + _write_index(tmp_git_repo) # index doesn't matter — artifacts check fires first + + manifests = [(f"dream/proj/{dream_id}", dream_id, {})] + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), manifests, "proj", dry_run=True + ) + assert deleted == 0 + assert kept == 1 + + +@pytest.mark.slow +def test_cleanup_branches_respects_keep_branches_env( + dream_reconcile, tmp_git_repo, monkeypatch +): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + + monkeypatch.setenv("SHADOWFROG_KEEP_BRANCHES", "1") + manifests = [("dream/proj/xx", "xx", {})] + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), manifests, "proj", dry_run=False + ) + assert deleted == 0 + assert kept == 1 + + +@pytest.mark.slow +def test_cleanup_branches_refuses_with_uncommitted_shadow( + dream_reconcile, tmp_git_repo +): + """B16: the combined `reconcile --cleanup-branches` invocation merges + discoveries into the working tree but does NOT commit them. Cleanup must + refuse while .shadow/ is dirty — otherwise the ancestor check passes + against the stale (pre-reconcile) HEAD and the only durable copy of the + discoveries (the dream branch) gets deleted. + """ + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + + dream_id = "20260420-150000Z-dirty" + _write_index(tmp_git_repo) + _seed_dream_artifacts(tmp_git_repo, dream_id) # uncommitted .shadow/ changes + + manifests = [(f"dream/proj/{dream_id}", dream_id, {})] + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), manifests, "proj", dry_run=False + ) + assert deleted == 0 + assert kept == 1 + + +@pytest.mark.slow +def test_cleanup_branches_proceeds_when_shadow_committed_and_pushed( + dream_reconcile, tmp_git_repo +): + """Positive case: once the reconciliation is committed and pushed (clean + .shadow/, HEAD on origin/main), cleanup is allowed to delete the branch.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + + dream_id = "20260420-150500Z-clean" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + + # Mirror artifacts + index onto main, then commit and push. + _seed_dream_artifacts(tmp_git_repo, dream_id) + _write_index( + tmp_git_repo, + "# Dream Experiment Archive\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | {branch} | main | abc1234 |\n", + ) + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "reconcile", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", "main", cwd=tmp_git_repo, env=env) + + manifests = [(branch, dream_id, _default_manifest(dream_id))] + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), manifests, "proj", dry_run=False + ) + assert deleted == 1 + assert kept == 0 + + +# =========================================================================== +# _count_discoveries — excludes Cross-References back-pointers +# =========================================================================== + +def test_count_discoveries_excludes_cross_references_section(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## File-Level\n\n" + "- file-level discovery one\n" + "- file-level discovery two\n" + "\n" + "## `foo`\n\n" + "- symbol discovery one\n" + "- symbol discovery two\n" + "- symbol discovery three\n" + "\n" + "## Cross-References\n\n" + "- [back-pointer one](_cross/a.md)\n" + "- [back-pointer two](_cross/b.md)\n" + ) + assert dream_reconcile._count_discoveries(str(shadow)) == 5 + + +def test_count_discoveries_case_insensitive_xref_heading(dream_reconcile, tmp_path): + """A lowercase rewrite must still suppress the back-pointer bullets.""" + shadow = tmp_path / "foo.md" + shadow.write_text( + "## `foo`\n\n- real discovery\n\n" + "## cross-references\n\n- [bp](_cross/a.md)\n" + ) + assert dream_reconcile._count_discoveries(str(shadow)) == 1 + + +def test_count_discoveries_missing_file_returns_zero(dream_reconcile, tmp_path): + assert dream_reconcile._count_discoveries(str(tmp_path / "nope.md")) == 0 + + +def test_count_discoveries_matches_coupon_demo_index(dream_reconcile, coupon_demo): + """The committed _index.md says: cart=14, inventory=10, test_cart=9 → 33.""" + cart = coupon_demo / ".shadow" / "cart.py.md" + inv = coupon_demo / ".shadow" / "inventory.py.md" + tst = coupon_demo / ".shadow" / "test_cart.py.md" + assert dream_reconcile._count_discoveries(str(cart)) == 14 + assert dream_reconcile._count_discoveries(str(inv)) == 10 + assert dream_reconcile._count_discoveries(str(tst)) == 9 + + +# =========================================================================== +# _shadow_symbol_names / _shadow_language +# =========================================================================== + +def test_shadow_symbol_names_extracts_top_level_only(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "## File-Level\n\n- foo\n\n" + "## `bar`\n\n- d\n\n" + "### `bar.method`\n\n- nested d (not counted)\n\n" + "## `class Baz`\n\n- d\n\n" + "## `interface IQux`\n\n- d\n\n" + "## Cross-References\n\n" + ) + names = dream_reconcile._shadow_symbol_names(str(shadow)) + assert names == ["bar", "Baz", "IQux"] + + +def test_shadow_symbol_names_missing_file(dream_reconcile, tmp_path): + assert dream_reconcile._shadow_symbol_names(str(tmp_path / "missing.md")) == [] + + +def test_shadow_language_reads_header(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text( + "# Shadow: foo.py\n\n" + "**Language**: Python | **Lines**: 28 | **Last modified**: 2026-04-20\n\n" + "## `foo`\n" + ) + assert dream_reconcile._shadow_language(str(shadow)) == "Python" + + +def test_shadow_language_missing_header_returns_unknown(dream_reconcile, tmp_path): + shadow = tmp_path / "foo.md" + shadow.write_text("## `foo`\n\n- d\n") + assert dream_reconcile._shadow_language(str(shadow)) == "Unknown" + + +def test_shadow_language_reads_from_coupon_demo(dream_reconcile, coupon_demo): + cart = coupon_demo / ".shadow" / "cart.py.md" + assert dream_reconcile._shadow_language(str(cart)) == "Python" + + +# =========================================================================== +# rebuild_top_index — B6 regression +# =========================================================================== + +def test_rebuild_top_index_dry_run_does_not_write(dream_reconcile, coupon_demo): + index_path = coupon_demo / ".shadow" / "_index.md" + original = index_path.read_text() + dream_reconcile.rebuild_top_index(str(coupon_demo), dry_run=True) + assert index_path.read_text() == original + + +def test_rebuild_top_index_regenerates_coupon_demo_counts(dream_reconcile, coupon_demo): + """Regenerated header must report the same totals as the committed file.""" + index_path = coupon_demo / ".shadow" / "_index.md" + original = index_path.read_text() + + # Sanity-check the committed file states the expected counts. + assert "Total files: 3" in original + assert "Symbols: 9" in original + assert "Discoveries: 33" in original + assert "Cross-cutting: 3" in original + assert "Dream cycles: 3" in original + + dream_reconcile.rebuild_top_index(str(coupon_demo), dry_run=False) + new = index_path.read_text() + + assert "Total files: 3" in new + assert "Symbols: 9" in new + assert "Discoveries: 33" in new + assert "Cross-cutting: 3" in new + assert "Dream cycles: 3" in new + + # Per-file rows preserved (any order — counts should match). + assert re.search(r"\| cart\.py \| Python \| 4 .* \| 14 \|", new) + assert re.search(r"\| inventory\.py \| Python \| 2 .* \| 10 \|", new) + assert re.search(r"\| test_cart\.py \| Python \| 3 .* \| 9 \|", new) + + +def test_rebuild_top_index_preserves_init_provenance(dream_reconcile, coupon_demo): + index_path = coupon_demo / ".shadow" / "_index.md" + dream_reconcile.rebuild_top_index(str(coupon_demo), dry_run=False) + new = index_path.read_text() + assert "Initially generated by shadow-frog-init on 2026-04-20" in new + assert "Last updated by shadow-frog-dream on" in new + + +def test_rebuild_top_index_handles_missing_shadow_dir(dream_reconcile, tmp_path, capsys): + # No .shadow dir at all — should print and return without raising. + dream_reconcile.rebuild_top_index(str(tmp_path), dry_run=False) + out = capsys.readouterr().out + assert "No .shadow" in out + + +def test_rebuild_top_index_includes_underscore_prefixed_source_dirs( + dream_reconcile, tmp_path +): + """B19: a shadow nested under a `_`-prefixed source dir must appear in the + top index (only top-level _meta/_cross/_dreams are pruned).""" + shadow = tmp_path / ".shadow" + (shadow / "_meta").mkdir(parents=True) + (shadow / "src" / "_internal").mkdir(parents=True) + (shadow / "src" / "_internal" / "helper.py.md").write_text( + "# Shadow\n## `helper`\n\n- d1\n\n## Cross-References\n\n_No discoveries yet._\n" + ) + dream_reconcile.rebuild_top_index(str(tmp_path), dry_run=False) + new = (shadow / "_index.md").read_text() + assert "src/_internal/helper.py" in new + assert "Total files: 1" in new + + +# =========================================================================== +# load_manifests + discover_branches — integration via real branches +# =========================================================================== + +@pytest.mark.slow +def test_load_manifests_skips_invalid(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + + good_id = "20260420-100000Z-good" + bad_id = "20260420-100100Z-bad" + + make_dream_branch( + tmp_git_repo, env, "proj", good_id, + _default_manifest(good_id), + ) + make_dream_branch( + tmp_git_repo, env, "proj", bad_id, + {"dream_id": "OOPS", "category": "x", "verdict": "useful"}, + ) + + branches = [ + (f"dream/proj/{good_id}", good_id), + (f"dream/proj/{bad_id}", bad_id), + ] + manifests, skipped = dream_reconcile.load_manifests(str(tmp_git_repo), branches) + assert len(manifests) == 1 + assert manifests[0][1] == good_id + assert len(skipped) == 1 + assert skipped[0][1] == bad_id + + +@pytest.mark.slow +def test_discover_branches_filters_namespace_and_existing(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + + new_id = "20260420-110000Z-new" + indexed_id = "20260420-111000Z-indexed" + other_ns_id = "20260420-112000Z-other" + + make_dream_branch(tmp_git_repo, env, "proj", new_id, _default_manifest(new_id)) + make_dream_branch(tmp_git_repo, env, "proj", indexed_id, + _default_manifest(indexed_id)) + make_dream_branch(tmp_git_repo, env, "different", other_ns_id, + _default_manifest(other_ns_id, dream_ns="different")) + + # Mark indexed_id as already reconciled. + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {indexed_id} | bug hunting | useful | T | dream/proj/{indexed_id} | main | 1234567 |\n", + ) + + discovered = dream_reconcile.discover_branches(str(tmp_git_repo), "proj") + discovered_ids = {did for _, did in discovered} + assert new_id in discovered_ids + assert indexed_id not in discovered_ids # already in index + assert other_ns_id not in discovered_ids # wrong namespace + + +# =========================================================================== +# CLI smoke tests +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_help_exits_zero(): + result = subprocess.run( + [sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True, + ) + assert result.returncode == 0 + assert "Dream reconciliation" in result.stdout + assert "--dry-run" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_dry_run_with_no_dream_branches(tmp_git_repo): + """Run the full CLI against a repo with no dream branches → exits 0 + with the 'no new branches' messaging.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + + full_env = os.environ.copy() + full_env["DREAM_NAMESPACE"] = "proj" + + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), "--dry-run"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "No new branches" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_unknown_argument_exits_nonzero(): + result = subprocess.run( + [sys.executable, str(SCRIPT), "--bogus-flag"], + capture_output=True, text=True, + ) + assert result.returncode == 1 + assert "Unknown argument" in result.stderr + + +# =========================================================================== +# find_heading — symbol heading lookup +# =========================================================================== + +def test_find_heading_backtick_top_level(dream_reconcile): + lines = ["# Shadow\n", "\n", "## `do_thing`\n", "- d\n"] + assert dream_reconcile.find_heading(lines, "do_thing") == 2 + + +def test_find_heading_backtick_nested(dream_reconcile): + lines = ["## `class Foo`\n", "### `Foo.bar`\n", "- d\n"] + assert dream_reconcile.find_heading(lines, "Foo.bar") == 1 + + +def test_find_heading_bare_top_level(dream_reconcile): + lines = ["## do_thing\n", "- d\n"] + assert dream_reconcile.find_heading(lines, "do_thing") == 0 + + +def test_find_heading_bare_nested(dream_reconcile): + lines = ["### do_thing\n"] + assert dream_reconcile.find_heading(lines, "do_thing") == 0 + + +def test_find_heading_returns_negative_one_when_missing(dream_reconcile): + lines = ["## `other`\n", "## `also_other`\n"] + assert dream_reconcile.find_heading(lines, "missing") == -1 + + +def test_find_heading_returns_first_match(dream_reconcile): + """If the same heading appears twice (legacy malformed file), we want the + earliest line so subsequent inserts go into the correct section.""" + lines = ["## `foo`\n", "- d1\n", "## `foo`\n", "- d2\n"] + assert dream_reconcile.find_heading(lines, "foo") == 0 + + +def test_find_heading_ignores_trailing_whitespace(dream_reconcile): + lines = ["## `foo` \n"] + assert dream_reconcile.find_heading(lines, "foo") == 0 + + +def test_find_heading_does_not_match_partial(dream_reconcile): + """Substring-style false-positive guard.""" + lines = ["## `foobar`\n"] + assert dream_reconcile.find_heading(lines, "foo") == -1 + + +# =========================================================================== +# git / git_show helpers — subprocess wrappers +# =========================================================================== + +def test_git_returns_stdout_stripped(dream_reconcile, tmp_git_repo): + """git() should return stdout with surrounding whitespace stripped.""" + _seed_repo(tmp_git_repo) + out = dream_reconcile.git("rev-parse", "--abbrev-ref", "HEAD", + cwd=str(tmp_git_repo)) + assert out == "main" + + +def test_git_raises_runtime_error_on_failure(dream_reconcile, tmp_git_repo): + """check=True (default) raises RuntimeError on non-zero exit.""" + _seed_repo(tmp_git_repo) + with pytest.raises(RuntimeError, match="git "): + dream_reconcile.git("rev-parse", "no-such-ref-xyz", + cwd=str(tmp_git_repo)) + + +def test_git_check_false_swallows_error(dream_reconcile, tmp_git_repo): + """check=False returns whatever stdout came back (possibly empty).""" + _seed_repo(tmp_git_repo) + out = dream_reconcile.git("config", "--get", "nonexistent.shadowfrog.key", + cwd=str(tmp_git_repo), check=False) + assert out == "" # stdout empty when key missing + + +def test_git_show_returns_file_content(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + (tmp_git_repo / "hello.txt").write_text("hello world\n") + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "add hello", cwd=tmp_git_repo, env=env) + out = dream_reconcile.git_show("HEAD", "hello.txt", cwd=str(tmp_git_repo)) + assert out == "hello world\n" + + +def test_git_show_returns_none_for_missing_path(dream_reconcile, tmp_git_repo): + _seed_repo(tmp_git_repo) + out = dream_reconcile.git_show("HEAD", "no-such-file.txt", + cwd=str(tmp_git_repo)) + assert out is None + + +def test_git_show_returns_none_for_missing_ref(dream_reconcile, tmp_git_repo): + _seed_repo(tmp_git_repo) + out = dream_reconcile.git_show("no-such-ref", "README.md", + cwd=str(tmp_git_repo)) + assert out is None + + +# =========================================================================== +# _resolve_tip_commit +# =========================================================================== + +@pytest.mark.slow +def test_resolve_tip_commit_returns_short_sha(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-090000Z-tip" + branch = make_dream_branch( + tmp_git_repo, env, "proj", dream_id, _default_manifest(dream_id), + ) + tip = dream_reconcile._resolve_tip_commit(str(tmp_git_repo), branch) + assert re.fullmatch(r"[0-9a-f]{7}", tip), tip + + +@pytest.mark.slow +def test_resolve_tip_commit_returns_unknown_for_missing_branch( + dream_reconcile, tmp_git_repo +): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + tip = dream_reconcile._resolve_tip_commit( + str(tmp_git_repo), "dream/proj/never-existed" + ) + assert tip == "unknown" + + +@pytest.mark.slow +def test_resolve_tip_commit_returns_unknown_with_no_remote( + dream_reconcile, tmp_git_repo +): + """No origin remote at all — must not raise.""" + _seed_repo(tmp_git_repo) + tip = dream_reconcile._resolve_tip_commit( + str(tmp_git_repo), "dream/proj/any" + ) + assert tip == "unknown" + + +# =========================================================================== +# merge_discoveries +# =========================================================================== + +def _discovery(anchor, text, **kwargs): + out = {"anchor": anchor, "text": text} + out.update(kwargs) + return out + + +def _manifest_with(dream_id, discoveries=None, cross_cutting=None, dream_ns="proj"): + m = _default_manifest(dream_id, dream_ns=dream_ns) + m["discoveries"] = discoveries or [] + m["cross_cutting"] = cross_cutting or [] + return m + + +def test_merge_discoveries_creates_per_file_shadows(dream_reconcile, tmp_path): + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-000000Z-merge1" + manifest = _manifest_with(dream_id, discoveries=[ + _discovery("src/cart.py::add_item", "Cart silently drops negative qty.", + status="verified", source="exploration"), + _discovery("src/cart.py::checkout", "Checkout retries 3x on 5xx.", + status="verified", source="exploration"), + _discovery("lib/util.py::norm", "Norm strips zero-width space chars.", + status="verified", source="user"), + ]) + manifests = [(f"dream/proj/{dream_id}", dream_id, manifest)] + + merged, skipped = dream_reconcile.merge_discoveries( + str(repo), manifests, dry_run=False, + ) + assert merged == 3 + assert skipped == 0 + + cart = (repo / ".shadow" / "src" / "cart.py.md").read_text() + util = (repo / ".shadow" / "lib" / "util.py.md").read_text() + + assert "## `add_item`" in cart + assert "## `checkout`" in cart + assert "silently drops negative qty" in cart + assert "Dream report: `_dreams/20260420-000000Z-merge1/`" in cart + assert "## `norm`" in util + assert "source: user" in util + + +def test_merge_discoveries_adds_dream_report_marker_to_each( + dream_reconcile, tmp_path +): + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-000100Z-marker" + manifest = _manifest_with(dream_id, discoveries=[ + _discovery("a.py::x", "First discovery here for x."), + _discovery("a.py::y", "Second discovery here for y."), + ]) + dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + body = (repo / ".shadow" / "a.py.md").read_text() + assert body.count(f"Dream report: `_dreams/{dream_id}/`") == 2 + + +def test_merge_discoveries_dry_run_writes_no_files(dream_reconcile, tmp_path): + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-000200Z-dry" + manifest = _manifest_with(dream_id, discoveries=[ + _discovery("src/foo.py::bar", "Bar does the thing."), + ], cross_cutting=[ + {"slug": "abc", "title": "ABC", "text": "Crosses A,B,C.", + "refs": ["src/foo.py::bar", "lib/baz.py::qux"]}, + ]) + before = sorted(p.relative_to(repo) for p in repo.rglob("*")) + merged, _ = dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + dry_run=True, + ) + after = sorted(p.relative_to(repo) for p in repo.rglob("*")) + # Dry-run claims to have done all the work but writes nothing new. + assert merged == 2 + assert before == after + + +def test_merge_discoveries_skips_discoveries_without_anchor( + dream_reconcile, tmp_path +): + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-000300Z-noanc" + # Missing `::` → skipped, not crashing. + manifest = _manifest_with(dream_id, discoveries=[ + {"anchor": "no-double-colon-here", "text": "Skipped."}, + {"anchor": "", "text": "Also skipped."}, + _discovery("a.py::ok", "This one stays."), + ]) + merged, skipped = dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + assert merged == 1 + assert skipped == 2 + + +def test_merge_discoveries_normalizes_string_discovery(dream_reconcile, tmp_path): + """A bare string in `discoveries` becomes {'anchor': '', 'text': ...} + and is then skipped (no anchor) — must not crash.""" + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-000400Z-strnorm" + manifest = _manifest_with(dream_id, discoveries=[ + "raw string with no anchor", + ]) + merged, skipped = dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + assert merged == 0 + assert skipped == 1 + + +def test_merge_discoveries_creates_cross_cutting_file_with_back_pointers( + dream_reconcile, tmp_path +): + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-000500Z-cross" + manifest = _manifest_with(dream_id, cross_cutting=[ + { + "slug": "auth-lifecycle", + "title": "Auth Lifecycle", + "category": "behavior", + "refs": ["src/auth.py::login", "src/auth.py::logout", + "lib/session.py::Session"], + "text": "Login/logout flow shares state with Session.", + "status": "verified", + "source": "exploration", + }, + ]) + dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + + cross = repo / ".shadow" / "_cross" / "auth-lifecycle.md" + assert cross.is_file() + body = cross.read_text() + assert body.startswith("# Auth Lifecycle\n") + assert "**Category**: behavior" in body + assert "`src/auth.py::login`" in body + assert "`lib/session.py::Session`" in body + assert "_(verified, source: exploration)_" in body + + # Each referenced per-file shadow must have a back-pointer. + auth = (repo / ".shadow" / "src" / "auth.py.md").read_text() + sess = (repo / ".shadow" / "lib" / "session.py.md").read_text() + assert "[Auth Lifecycle](../_cross/auth-lifecycle.md)" in auth + assert "[Auth Lifecycle](../_cross/auth-lifecycle.md)" in sess + assert f"dream: {dream_id}" in auth + assert f"dream: {dream_id}" in sess + + +def test_merge_discoveries_normalizes_string_cross_cutting( + dream_reconcile, tmp_path +): + """A bare string in cross_cutting is normalized to a dict with a slug + derived from the text. With no refs and no anchor, it just creates a + cross-cutting file — must not crash.""" + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-000600Z-xcross" + manifest = _manifest_with(dream_id, cross_cutting=[ + "Some Behavior That Spans Files", + ]) + merged, _ = dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + # The slug is derived from the lowered-and-kebabbed text prefix. + cross_dir = repo / ".shadow" / "_cross" + assert cross_dir.is_dir() + files = list(cross_dir.glob("*.md")) + assert len(files) == 1 + assert "some-behavior-that-spans-files" in files[0].name + + +def test_merge_discoveries_skips_cross_cutting_without_slug( + dream_reconcile, tmp_path +): + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-000700Z-noslug" + manifest = _manifest_with(dream_id, cross_cutting=[ + {"slug": "", "title": "Empty", "refs": []}, + ]) + merged, skipped = dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + assert merged == 0 + assert not (repo / ".shadow" / "_cross").exists() + + +def test_merge_discoveries_skips_existing_cross_cutting_file( + dream_reconcile, tmp_path +): + """If `.shadow/_cross/<slug>.md` already exists, merging counts it as + skipped (no overwrite) but still heals back-pointers.""" + repo = tmp_path + (repo / ".shadow" / "_cross").mkdir(parents=True) + existing = repo / ".shadow" / "_cross" / "preexist.md" + existing.write_text("# Old Title\n\n**Refs**:\n- `x.py::y`\n") + + dream_id = "20260420-000800Z-preexist" + manifest = _manifest_with(dream_id, cross_cutting=[ + { + "slug": "preexist", "title": "Old Title", + "refs": ["x.py::y"], "text": "ignored", + }, + ]) + merged, skipped = dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + # No new cross file created — but the back-pointer to x.py is still added. + assert merged == 0 + assert skipped == 1 + assert "# Old Title" in existing.read_text() + # And the back-pointer landed. + assert "[Old Title](_cross/preexist.md)" in ( + repo / ".shadow" / "x.py.md" + ).read_text() + + +def test_merge_discoveries_skips_cross_ref_without_double_colon( + dream_reconcile, tmp_path +): + """Refs lacking `::` don't get back-pointers (no file is implied).""" + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-000900Z-bareref" + manifest = _manifest_with(dream_id, cross_cutting=[ + { + "slug": "behave", "title": "Behaviour", + "refs": ["just-a-symbol-name"], # no `::` + "text": "a thing", "status": "verified", "source": "exploration", + }, + ]) + dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + cross = repo / ".shadow" / "_cross" / "behave.md" + assert cross.is_file() + # No per-file shadow was synthesized. + other_md = [p for p in (repo / ".shadow").rglob("*.md") + if p.relative_to(repo / ".shadow").parts[0] != "_cross"] + assert other_md == [] + + +def test_merge_discoveries_dry_run_prints_planned_actions( + dream_reconcile, tmp_path, capsys +): + repo = tmp_path + (repo / ".shadow").mkdir() + dream_id = "20260420-001000Z-dryprint" + manifest = _manifest_with(dream_id, discoveries=[ + _discovery("a.py::z", "Discovery."), + ], cross_cutting=[ + {"slug": "xx", "title": "XX", + "refs": ["a.py::z"], "text": "t"}, + ]) + dream_reconcile.merge_discoveries( + str(repo), [(f"dream/proj/{dream_id}", dream_id, manifest)], + dry_run=True, + ) + out = capsys.readouterr().out + assert "Would merge: a.py::z" in out + assert "Would create cross-cutting: _cross/xx.md" in out + assert "+ back-pointer in .shadow/a.py.md" in out + + +def test_merge_discoveries_multiple_manifests(dream_reconcile, tmp_path): + repo = tmp_path + (repo / ".shadow").mkdir() + d1, d2 = "20260420-001100Z-d1", "20260420-001200Z-d2" + m1 = _manifest_with(d1, discoveries=[_discovery("a.py::x", "First.")]) + m2 = _manifest_with(d2, discoveries=[_discovery("a.py::y", "Second.")]) + merged, _ = dream_reconcile.merge_discoveries( + str(repo), + [ + (f"dream/proj/{d1}", d1, m1), + (f"dream/proj/{d2}", d2, m2), + ], + ) + assert merged == 2 + body = (repo / ".shadow" / "a.py.md").read_text() + assert "## `x`" in body + assert "## `y`" in body + assert f"_dreams/{d1}/" in body + assert f"_dreams/{d2}/" in body + + +# =========================================================================== +# mirror_reports +# =========================================================================== + +@pytest.mark.slow +def test_mirror_reports_copies_report_and_patch(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-020000Z-mirror" + manifest = _default_manifest(dream_id) + report = _default_report(dream_id) + patch = "diff --git a/foo b/foo\n--- a/foo\n+++ b/foo\n@@\n+changed\n" + make_dream_branch(tmp_git_repo, env, "proj", dream_id, + manifest, report=report, patch=patch) + + mirrored, corrupted = dream_reconcile.mirror_reports( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + assert mirrored == 1 + assert corrupted == [] + dream_dir = tmp_git_repo / ".shadow" / "_dreams" / dream_id + assert (dream_dir / "report.md").read_text() == report + assert (dream_dir / "patch.diff").read_text() == patch + # Manifest is rewritten from the in-memory dict. + saved_manifest = json.loads((dream_dir / "manifest.json").read_text()) + assert saved_manifest["dream_id"] == dream_id + + +@pytest.mark.slow +def test_mirror_reports_handles_missing_report_gracefully( + dream_reconcile, tmp_git_repo +): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-020100Z-noreport" + manifest = _default_manifest(dream_id) + # No report.md, but manifest + patch present. + make_dream_branch(tmp_git_repo, env, "proj", dream_id, manifest, + report=None, patch="diff --git a/x b/x\n") + mirrored, corrupted = dream_reconcile.mirror_reports( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + assert mirrored == 1 + assert corrupted == [] + dream_dir = tmp_git_repo / ".shadow" / "_dreams" / dream_id + # Manifest + patch mirrored; report.md NOT created on main side. + assert (dream_dir / "manifest.json").is_file() + assert (dream_dir / "patch.diff").is_file() + assert not (dream_dir / "report.md").is_file() + + +@pytest.mark.slow +def test_mirror_reports_handles_missing_patch(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-020200Z-nopatch" + manifest = _default_manifest(dream_id) + report = _default_report(dream_id) + make_dream_branch(tmp_git_repo, env, "proj", dream_id, manifest, + report=report, patch=None) + mirrored, _ = dream_reconcile.mirror_reports( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + assert mirrored == 1 + dream_dir = tmp_git_repo / ".shadow" / "_dreams" / dream_id + assert (dream_dir / "report.md").is_file() + assert not (dream_dir / "patch.diff").is_file() + + +@pytest.mark.slow +def test_mirror_reports_preserves_empty_patch(dream_reconcile, tmp_git_repo): + """A 0-byte patch.diff on the dream branch must be mirrored as a + 0-byte patch.diff on main — not skipped. + + Regression: a truthy `if patch:` check used to conflate + `git_show` returning None (file absent) with returning "" (file + present, 0 bytes). The empty-but-present case got silently dropped, + so verify_artifacts reported `missing patch.diff` even though the + file existed on origin, and the dream branch couldn't be cleaned up. + """ + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-020250Z-emptypatch" + manifest = _default_manifest(dream_id) + report = _default_report(dream_id) + make_dream_branch(tmp_git_repo, env, "proj", dream_id, manifest, + report=report, patch="") + mirrored, corrupted = dream_reconcile.mirror_reports( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + assert mirrored == 1 + assert corrupted == [] + dream_dir = tmp_git_repo / ".shadow" / "_dreams" / dream_id + patch_path = dream_dir / "patch.diff" + assert patch_path.is_file(), ( + "Empty patch.diff on the dream branch must still be mirrored to " + "main; otherwise verify_artifacts spuriously reports it missing." + ) + assert patch_path.stat().st_size == 0 + + +@pytest.mark.slow +def test_mirror_reports_detects_dream_id_mismatch(dream_reconcile, tmp_git_repo): + """If the report.md frontmatter declares a DIFFERENT dream_id, mirror + must write a 'Corrupted Report' placeholder and surface the mismatch — + but the manifest and patch are still mirrored so valid artifacts (and + the discoveries read from the manifest) are never lost to one bad line.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-020300Z-corrupt" + wrong_id = "20260420-020300Z-CORRUPTED-FROM-ELSEWHERE" + manifest = _default_manifest(dream_id) + # report references the WRONG id in frontmatter. + bad_report = _default_report(wrong_id) + make_dream_branch(tmp_git_repo, env, "proj", dream_id, manifest, + report=bad_report) + mirrored, corrupted = dream_reconcile.mirror_reports( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + # The mismatch is surfaced, but the dream's artifacts are still mirrored. + assert mirrored == 1 + assert corrupted == [(dream_id, wrong_id)] + dream_dir = tmp_git_repo / ".shadow" / "_dreams" / dream_id + body = (dream_dir / "report.md").read_text() + assert "Corrupted Report" in body + assert wrong_id in body + # The manifest must still be mirrored despite the corrupt report. + assert (dream_dir / "manifest.json").is_file() + + +@pytest.mark.slow +def test_mirror_reports_dry_run_writes_nothing(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-020400Z-dry" + manifest = _default_manifest(dream_id) + make_dream_branch(tmp_git_repo, env, "proj", dream_id, + manifest, report=_default_report(dream_id)) + before = sorted(p.relative_to(tmp_git_repo) + for p in tmp_git_repo.rglob("*") + if ".git" not in p.parts) + mirrored, corrupted = dream_reconcile.mirror_reports( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + dry_run=True, + ) + after = sorted(p.relative_to(tmp_git_repo) + for p in tmp_git_repo.rglob("*") + if ".git" not in p.parts) + assert mirrored == 1 + assert before == after + + +@pytest.mark.slow +def test_mirror_reports_multiple_manifests(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + d1 = "20260420-020500Z-m1" + d2 = "20260420-020600Z-m2" + make_dream_branch(tmp_git_repo, env, "proj", d1, + _default_manifest(d1), report=_default_report(d1)) + make_dream_branch(tmp_git_repo, env, "proj", d2, + _default_manifest(d2), report=_default_report(d2)) + mirrored, _ = dream_reconcile.mirror_reports( + str(tmp_git_repo), + [ + (f"dream/proj/{d1}", d1, _default_manifest(d1)), + (f"dream/proj/{d2}", d2, _default_manifest(d2)), + ], + ) + assert mirrored == 2 + for d in (d1, d2): + body = (tmp_git_repo / ".shadow" / "_dreams" / d / + "report.md").read_text() + assert f"# {d}" in body + + +# =========================================================================== +# update_index +# =========================================================================== + +@pytest.mark.slow +def test_update_index_bootstraps_when_missing(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-030000Z-boot" + make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + dream_reconcile.update_index( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, _default_manifest(dream_id))], + ) + idx = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + assert "# Dream Experiment Archive" in idx + assert "| dream_id | category | verdict |" in idx + assert f"| {dream_id} | bug hunting | useful |" in idx + assert f"dream/proj/{dream_id}" in idx + + +@pytest.mark.slow +def test_update_index_preserves_existing_rows(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + new_id = "20260420-030100Z-newer" + # Create the dream branch FIRST so the pre-seeded _index.md doesn't get + # carried into it (uncommitted files on main follow the checkout). + make_dream_branch(tmp_git_repo, env, "proj", new_id, + _default_manifest(new_id)) + _write_index(tmp_git_repo) # pre-seed with INDEX_FIXTURE (3 rows) + dream_reconcile.update_index( + str(tmp_git_repo), + [(f"dream/proj/{new_id}", new_id, _default_manifest(new_id))], + ) + body = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + assert "20260101-000000Z-alpha" in body + assert "20260102-000000Z-beta" in body + assert "20260103-000000Z-gamma" in body + assert new_id in body + + +@pytest.mark.slow +def test_update_index_records_real_tip_commit(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-030200Z-tip" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + expected_tip = dream_reconcile._resolve_tip_commit(str(tmp_git_repo), branch) + dream_reconcile.update_index( + str(tmp_git_repo), + [(branch, dream_id, _default_manifest(dream_id))], + ) + body = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + row = [l for l in body.splitlines() if dream_id in l][0] + assert f"| {expected_tip} |" in row + + +@pytest.mark.slow +def test_update_index_dry_run_does_not_write(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-030300Z-dry" + # Create dream branch FIRST so the pre-seeded file isn't pulled into it. + make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + idx_path = tmp_git_repo / ".shadow" / "_dreams" / "_index.md" + # Pre-seed the file so the bootstrap path doesn't fire. + idx_path.parent.mkdir(parents=True, exist_ok=True) + idx_path.write_text("# Dream Index\n\nold body\n") + before = idx_path.read_text() + dream_reconcile.update_index( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, _default_manifest(dream_id))], + dry_run=True, + ) + assert idx_path.read_text() == before + + +@pytest.mark.slow +def test_update_index_uses_report_heading_when_manifest_title_missing( + dream_reconcile, tmp_git_repo +): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-030400Z-fromreport" + manifest = _default_manifest(dream_id) + del manifest["title"] + # Report frontmatter then a markdown heading we want extracted. + report = ( + "---\n" + f"dream_id: \"{dream_id}\"\n" + "category: bug hunting\n" + "verdict: useful\n" + "base_commit: abcdef1234567890\n" + "---\n\n" + "# Beautiful Title From Report\n\nBody.\n" + ) + make_dream_branch(tmp_git_repo, env, "proj", dream_id, + manifest, report=report) + dream_reconcile.update_index( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + body = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + assert "Beautiful Title From Report" in body + + +@pytest.mark.slow +def test_update_index_falls_back_to_dream_id_when_no_title( + dream_reconcile, tmp_git_repo +): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-030500Z-noref" + manifest = _default_manifest(dream_id) + del manifest["title"] + # No report.md at all → fallback to "Dream <id>" + make_dream_branch(tmp_git_repo, env, "proj", dream_id, + manifest, report=None) + dream_reconcile.update_index( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + body = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + assert f"| Dream {dream_id} |" in body + + +@pytest.mark.slow +def test_update_index_strips_pipe_chars_from_title(dream_reconcile, tmp_git_repo): + """Title containing `|` would break markdown table parsing; must be neutralized.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-030600Z-pipe" + manifest = _default_manifest(dream_id) + manifest["title"] = "Bad | title | with pipes" + make_dream_branch(tmp_git_repo, env, "proj", dream_id, manifest) + dream_reconcile.update_index( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + body = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + row = [l for l in body.splitlines() if dream_id in l][0] + # Sanitized: pipes replaced with hyphens; row still has exactly the + # canonical 8 separators ('| ' + 7 columns + ' |'). + assert "Bad - title - with pipes" in row + assert row.count("|") == 8 + + +@pytest.mark.slow +def test_update_index_normalizes_category_with_parens( + dream_reconcile, tmp_git_repo +): + """Categories like `bug hunting (refresher)` must strip the trailing + parenthetical so they match the canonical category set.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-030700Z-cat" + manifest = _default_manifest(dream_id) + manifest["category"] = "Bug Hunting (notes from the field)" + make_dream_branch(tmp_git_repo, env, "proj", dream_id, manifest) + dream_reconcile.update_index( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + body = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + row = [l for l in body.splitlines() if dream_id in l][0] + assert "| bug hunting |" in row + assert "(notes" not in row + + +# =========================================================================== +# update_state +# =========================================================================== + +@pytest.mark.slow +def test_update_state_bootstraps_missing_state_json( + dream_reconcile, tmp_git_repo +): + env = _seed_repo(tmp_git_repo) + # No .shadow/_meta yet — must be created from scratch. + dream_id = "20260420-040000Z-init" + manifest = _default_manifest(dream_id) + dream_reconcile.update_state( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, manifest)], + ) + state_path = tmp_git_repo / ".shadow" / "_meta" / "state.json" + assert state_path.is_file() + state = json.loads(state_path.read_text()) + assert state["dream_cycles_completed"] == 1 + assert state["last_update_type"] == "dream" + assert "last_update_at" in state + assert "last_commit" in state and len(state["last_commit"]) == 40 + + +@pytest.mark.slow +def test_update_state_increments_dream_cycles(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + state_dir = tmp_git_repo / ".shadow" / "_meta" + state_dir.mkdir(parents=True) + (state_dir / "state.json").write_text(json.dumps({ + "version": 1, "dream_cycles_completed": 5, + "total_files": 0, "total_symbols": 0, "total_discoveries": 0, + })) + dream_id = "20260420-040100Z-inc" + dream_reconcile.update_state( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, _default_manifest(dream_id))], + ) + state = json.loads((state_dir / "state.json").read_text()) + assert state["dream_cycles_completed"] == 6 + assert state["last_update_type"] == "dream" + + +@pytest.mark.slow +def test_update_state_recomputes_totals_excludes_internal_dirs( + dream_reconcile, tmp_git_repo +): + """`_meta/`, `_cross/`, `_dreams/` files must NOT contribute to totals.""" + _seed_repo(tmp_git_repo) + shadow = tmp_git_repo / ".shadow" + # 1 real file with 2 symbols + 3 discoveries. + (shadow / "src").mkdir(parents=True) + (shadow / "src" / "real.py.md").write_text( + "# Shadow\n" + "## `foo`\n\n- d1\n- d2\n\n" + "## `bar`\n\n- d3\n\n" + "## Cross-References\n\n- [bp](../_cross/x.md)\n" + ) + # Cross-cutting (should NOT contribute). + (shadow / "_cross").mkdir() + (shadow / "_cross" / "x.md").write_text( + "# X\n## `should_not_count`\n- bogus\n" + ) + # Dream report (should NOT contribute). + (shadow / "_dreams" / "20260420-040200Z-x").mkdir(parents=True) + (shadow / "_dreams" / "20260420-040200Z-x" / "report.md").write_text( + "## `also_not_counted`\n- bogus\n" + ) + dream_id = "20260420-040200Z-totals" + dream_reconcile.update_state( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, _default_manifest(dream_id))], + ) + state = json.loads( + (shadow / "_meta" / "state.json").read_text() + ) + assert state["total_files"] == 1 + assert state["total_symbols"] == 2 # foo + bar + assert state["total_discoveries"] == 3 # d1, d2, d3 (back-pointer excluded) + + +@pytest.mark.slow +def test_update_state_counts_underscore_prefixed_source_dirs( + dream_reconcile, tmp_git_repo +): + """B19: `_`-prefixed *source* dirs (e.g. src/_internal/) are real shadows. + + Only top-level internal dirs (_meta/_cross/_dreams) are pruned; a mirrored + shadow nested under a `_`-prefixed source dir must still be counted. + """ + _seed_repo(tmp_git_repo) + shadow = tmp_git_repo / ".shadow" + (shadow / "src" / "_internal").mkdir(parents=True) + (shadow / "src" / "_internal" / "helper.py.md").write_text( + "# Shadow\n" + "## `helper`\n\n- d1\n- d2\n\n" + "## Cross-References\n\n_No discoveries yet._\n" + ) + dream_id = "20260420-040250Z-underscore" + dream_reconcile.update_state( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, _default_manifest(dream_id))], + ) + state = json.loads( + (shadow / "_meta" / "state.json").read_text() + ) + assert state["total_files"] == 1 + assert state["total_symbols"] == 1 + assert state["total_discoveries"] == 2 + + +@pytest.mark.slow +def test_update_state_dry_run_does_not_create_state( + dream_reconcile, tmp_git_repo, capsys +): + _seed_repo(tmp_git_repo) + dream_id = "20260420-040300Z-dry" + dream_reconcile.update_state( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, _default_manifest(dream_id))], + dry_run=True, + ) + assert not (tmp_git_repo / ".shadow" / "_meta" / "state.json").exists() + assert "Would create state.json" in capsys.readouterr().out + + +@pytest.mark.slow +def test_update_state_dry_run_does_not_modify_existing( + dream_reconcile, tmp_git_repo, capsys +): + _seed_repo(tmp_git_repo) + state_dir = tmp_git_repo / ".shadow" / "_meta" + state_dir.mkdir(parents=True) + payload = { + "version": 1, "dream_cycles_completed": 7, + "total_files": 9, "total_symbols": 99, "total_discoveries": 42, + } + (state_dir / "state.json").write_text(json.dumps(payload)) + original = (state_dir / "state.json").read_text() + dream_id = "20260420-040400Z-drye" + dream_reconcile.update_state( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, _default_manifest(dream_id))], + dry_run=True, + ) + assert (state_dir / "state.json").read_text() == original + assert "Would update state.json" in capsys.readouterr().out + + +@pytest.mark.slow +def test_update_state_records_last_commit_sha(dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + head_sha = _git("rev-parse", "HEAD", cwd=tmp_git_repo, env=env).stdout.strip() + dream_id = "20260420-040500Z-sha" + dream_reconcile.update_state( + str(tmp_git_repo), + [(f"dream/proj/{dream_id}", dream_id, _default_manifest(dream_id))], + ) + state = json.loads( + (tmp_git_repo / ".shadow" / "_meta" / "state.json").read_text() + ) + assert state["last_commit"] == head_sha + + +# =========================================================================== +# cleanup_branches — REAL delete path (non-dry-run) +# =========================================================================== + +@pytest.mark.slow +def test_cleanup_branches_actually_deletes_when_all_checks_pass( + dream_reconcile, tmp_git_repo +): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-050000Z-realdel" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + + # All safety conditions satisfied: + # 1. HEAD == origin/main → ancestor check OK. + # 2. All 3 artifacts on main. + _seed_dream_artifacts(tmp_git_repo, dream_id) + # 3. dream_id indexed. + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | {branch} | main | abc1234 |\n", + ) + + # Commit + push the reconciliation so .shadow/ is clean and HEAD is on + # origin/main (the safe two-phase flow the dirty-tree guard enforces). + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "reconcile", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", "main", cwd=tmp_git_repo, env=env) + + # Pre-check: branch exists on origin. + pre = _git("ls-remote", "--heads", "origin", branch, + cwd=tmp_git_repo, env=env).stdout + assert branch in pre + + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(branch, dream_id, _default_manifest(dream_id))], + "proj", + dry_run=False, + ) + assert deleted == 1 + assert kept == 0 + # Branch gone from origin. + post = _git("ls-remote", "--heads", "origin", branch, + cwd=tmp_git_repo, env=env).stdout + assert branch not in post + + +@pytest.mark.slow +def test_cleanup_branches_refuses_when_head_not_pushed( + dream_reconcile, tmp_git_repo +): + """If HEAD is ahead of origin/main, refuse to delete anything — would + risk losing the only copy of the discoveries.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-050100Z-notpushed" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + _seed_dream_artifacts(tmp_git_repo, dream_id) + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | {branch} | main | abc1234 |\n", + ) + # Make a new local commit on main that is NOT pushed. + (tmp_git_repo / "drift.txt").write_text("unpushed\n") + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "local-only", cwd=tmp_git_repo, env=env) + + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(branch, dream_id, _default_manifest(dream_id))], + "proj", + dry_run=False, + ) + assert deleted == 0 + assert kept == 1 + # Branch still on origin (NOT deleted). + post = _git("ls-remote", "--heads", "origin", branch, + cwd=tmp_git_repo, env=env).stdout + assert branch in post + + +@pytest.mark.slow +def test_cleanup_branches_keeps_when_descendant_branch_exists( + dream_reconcile, tmp_git_repo +): + """An un-reconciled child branch that lists `branch` as its parent must + inhibit deletion (would orphan the descendant's lineage).""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + parent_id = "20260420-050200Z-parent" + child_id = "20260420-050300Z-child" + + parent_branch = make_dream_branch( + tmp_git_repo, env, "proj", parent_id, _default_manifest(parent_id), + ) + + # Child manifest references `parent_branch` as its parent. + child_manifest = _default_manifest(child_id) + child_manifest["parent_branch"] = parent_branch + make_dream_branch(tmp_git_repo, env, "proj", child_id, child_manifest) + + # Parent fully reconciled; child NOT in index (still unreconciled). + _seed_dream_artifacts(tmp_git_repo, parent_id) + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {parent_id} | bug hunting | useful | T | {parent_branch} | main | abc1234 |\n", + ) + + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(parent_branch, parent_id, _default_manifest(parent_id))], + "proj", + dry_run=False, + ) + assert deleted == 0 + assert kept == 1 + # Parent branch is preserved on origin so the child still has its lineage. + post = _git("ls-remote", "--heads", "origin", parent_branch, + cwd=tmp_git_repo, env=env).stdout + assert parent_branch in post + + +@pytest.mark.slow +def test_cleanup_branches_keep_branches_env_overrides_real_delete( + dream_reconcile, tmp_git_repo, monkeypatch +): + """Even when ALL safety conditions pass, SHADOWFROG_KEEP_BRANCHES=1 + must inhibit deletion in the real-delete code path.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-050400Z-keepenv" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + _seed_dream_artifacts(tmp_git_repo, dream_id) + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | {branch} | main | abc1234 |\n", + ) + monkeypatch.setenv("SHADOWFROG_KEEP_BRANCHES", "1") + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(branch, dream_id, _default_manifest(dream_id))], + "proj", + dry_run=False, + ) + assert deleted == 0 + assert kept == 1 + # Branch still on origin. + post = _git("ls-remote", "--heads", "origin", branch, + cwd=tmp_git_repo, env=env).stdout + assert branch in post + + +@pytest.mark.parametrize("env_value", ["1", "true", "yes"]) +@pytest.mark.slow +def test_cleanup_branches_keep_branches_env_truthy_values( + dream_reconcile, tmp_git_repo, monkeypatch, env_value +): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + monkeypatch.setenv("SHADOWFROG_KEEP_BRANCHES", env_value) + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [("dream/proj/xx", "xx", {})], "proj", dry_run=False, + ) + assert deleted == 0 + assert kept == 1 + + +@pytest.mark.slow +def test_cleanup_branches_keep_branches_env_falsy_does_not_block( + dream_reconcile, tmp_git_repo, monkeypatch +): + """SHADOWFROG_KEEP_BRANCHES=0 (or empty) must NOT inhibit cleanup; + safety checks proceed normally.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-050500Z-zero" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + _seed_dream_artifacts(tmp_git_repo, dream_id) + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | {branch} | main | abc1234 |\n", + ) + monkeypatch.setenv("SHADOWFROG_KEEP_BRANCHES", "0") + # Commit + push so .shadow/ is clean (dirty-tree guard would otherwise fire). + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "reconcile", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", "main", cwd=tmp_git_repo, env=env) + deleted, kept = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(branch, dream_id, _default_manifest(dream_id))], + "proj", dry_run=False, + ) + # "0" is not a truthy value → cleanup proceeds → branch deleted. + assert deleted == 1 + assert kept == 0 + + +# =========================================================================== +# add_cross_reference_backpointer — coverage for append-after-existing path +# =========================================================================== + +def test_add_cross_reference_backpointer_appends_when_other_links_exist( + dream_reconcile, tmp_path +): + """If Cross-References already has a NON-matching back-pointer (no + placeholder), the new pointer must be appended after it.""" + repo = tmp_path + (repo / ".shadow").mkdir() + shadow = repo / ".shadow" / "foo.py.md" + shadow.write_text( + "## `foo`\n\n- discovery\n\n" + "## Cross-References\n\n" + "- [Existing](_cross/existing.md) (dream: 20260101-prev)\n" + ) + written = dream_reconcile.add_cross_reference_backpointer( + str(repo), "foo.py", "fresh", "Fresh Title", "20260420-091000Z-new", + ) + assert written is True + body = shadow.read_text() + assert "[Existing](_cross/existing.md)" in body + assert "[Fresh Title](_cross/fresh.md)" in body + # The pre-existing entry is before the new one. + assert body.index("Existing") < body.index("Fresh Title") + + +def test_add_cross_reference_backpointer_inserts_before_next_heading( + dream_reconcile, tmp_path +): + """File has another `## ...` heading after Cross-References → the + back-pointer must be inserted BEFORE that next heading.""" + repo = tmp_path + (repo / ".shadow").mkdir() + shadow = repo / ".shadow" / "z.py.md" + shadow.write_text( + "## `z`\n\n- d\n\n" + "## Cross-References\n\n" + "- [Pre](_cross/pre.md)\n\n" + "## Other Section\n\nstuff\n" + ) + dream_reconcile.add_cross_reference_backpointer( + str(repo), "z.py", "new", "New", "20260420-091500Z-y", + ) + body = shadow.read_text() + assert "[New](_cross/new.md)" in body + # Inserted before `## Other Section`. + assert body.index("[New]") < body.index("## Other Section") + + +# =========================================================================== +# main() — full CLI orchestrator integration tests +# =========================================================================== + +def _cli_env(extra=None): + """Realistic CLI env with isolated git config and optional overrides.""" + env = os.environ.copy() + env["GIT_CONFIG_GLOBAL"] = "/dev/null" + env["GIT_CONFIG_SYSTEM"] = "/dev/null" + if extra: + env.update(extra) + return env + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_namespace_requires_value(): + """`--namespace` with no following arg must exit 1 with a clear message.""" + result = subprocess.run( + [sys.executable, str(SCRIPT), "--namespace"], + capture_output=True, text=True, + ) + assert result.returncode == 1 + assert "--namespace requires a value" in result.stderr + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_full_happy_path_reconciles_two_dreams(tmp_git_repo): + """Two dream branches → reconcile → state.json, _index.md, _dreams/ + mirrors all updated; verify pass.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + d1 = "20260420-060000Z-cli1" + d2 = "20260420-060100Z-cli2" + m1 = _manifest_with(d1, discoveries=[ + _discovery("src/cart.py::add_item", "Cart drops negative qty silently."), + ], cross_cutting=[ + {"slug": "cart-flow", "title": "Cart Flow", + "refs": ["src/cart.py::add_item"], "text": "cross"}, + ]) + m2 = _manifest_with(d2, discoveries=[ + _discovery("src/cart.py::checkout", "Checkout retries 3x."), + ]) + make_dream_branch(tmp_git_repo, env, "proj", d1, m1, + report=_default_report(d1)) + make_dream_branch(tmp_git_repo, env, "proj", d2, m2, + report=_default_report(d2)) + + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo)], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "All 2 dreams verified" in result.stdout + # state.json updated. + state = json.loads( + (tmp_git_repo / ".shadow" / "_meta" / "state.json").read_text() + ) + assert state["dream_cycles_completed"] == 1 + assert state["last_update_type"] == "dream" + assert state["total_discoveries"] >= 2 + # _index.md has both dreams. + idx = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + assert d1 in idx + assert d2 in idx + # Per-dream mirror dirs. + for d in (d1, d2): + d_dir = tmp_git_repo / ".shadow" / "_dreams" / d + assert (d_dir / "manifest.json").is_file() + assert (d_dir / "report.md").is_file() + # Per-file shadow exists. + cart = (tmp_git_repo / ".shadow" / "src" / "cart.py.md").read_text() + assert "## `add_item`" in cart + assert "## `checkout`" in cart + # Cross-cutting file exists. + assert (tmp_git_repo / ".shadow" / "_cross" / "cart-flow.md").is_file() + # Top-level _index.md regenerated with dream cycles. + top = (tmp_git_repo / ".shadow" / "_index.md").read_text() + assert "Dream cycles: 1" in top + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_dry_run_makes_no_persistent_changes(tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + d1 = "20260420-061000Z-dry1" + m1 = _manifest_with(d1, discoveries=[ + _discovery("a.py::f", "Discovery for f."), + ]) + make_dream_branch(tmp_git_repo, env, "proj", d1, m1, + report=_default_report(d1)) + + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), "--dry-run"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "DRY RUN" in result.stdout + assert "would complete" in result.stdout + # No substantive artifacts: no per-file shadow, no state.json, + # no coverage.json, no mirrored dream dir. + assert not (tmp_git_repo / ".shadow" / "a.py.md").exists() + assert not (tmp_git_repo / ".shadow" / "_meta" / "state.json").exists() + assert not (tmp_git_repo / ".shadow" / "_meta" / "coverage.json").exists() + assert not (tmp_git_repo / ".shadow" / "_dreams" / d1).exists() + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_namespace_filters_branches(tmp_git_repo): + """`--namespace foo` ignores `dream/bar/...` branches entirely.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + in_id = "20260420-062000Z-inside" + out_id = "20260420-062100Z-outside" + make_dream_branch(tmp_git_repo, env, "inside", in_id, + _default_manifest(in_id, dream_ns="inside")) + make_dream_branch(tmp_git_repo, env, "outside", out_id, + _default_manifest(out_id, dream_ns="outside")) + full_env = _cli_env() + full_env.pop("DREAM_NAMESPACE", None) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), + "--namespace", "inside", "--dry-run"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert in_id in result.stdout + assert out_id not in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_skips_invalid_manifest_and_continues(tmp_git_repo): + """A dream with manifest-validation failures must NOT crash the run. + Other valid dreams continue to reconcile.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + good = "20260420-063000Z-good" + bad = "20260420-063100Z-bad" + make_dream_branch(tmp_git_repo, env, "proj", good, + _default_manifest(good), + report=_default_report(good)) + # Invalid manifest: dream_id mismatch. + bad_manifest = {"dream_id": "wrong-id", "category": "x", + "verdict": "useful", "discoveries": []} + make_dream_branch(tmp_git_repo, env, "proj", bad, bad_manifest, + report=_default_report(bad)) + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo)], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert good in result.stdout + # The bad one is mentioned as SKIP, not crashed. + assert "SKIP" in result.stdout and bad in result.stdout + # And only the good one made it into _index.md. + idx = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + assert good in idx + assert bad not in idx + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_no_dreams_exits_zero_with_message(tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo)], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0 + assert "No new branches" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_not_in_git_repo_exits_nonzero(tmp_path): + """Pointing at a non-git directory must error out.""" + not_a_repo = tmp_path / "not_a_repo" + not_a_repo.mkdir() + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(not_a_repo / "nope")], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 1 + assert "Not in a git repository" in result.stderr + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_no_positional_uses_git_toplevel(tmp_git_repo): + """No positional → git rev-parse --show-toplevel runs in CWD.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), "--dry-run"], + capture_output=True, text=True, env=full_env, cwd=str(tmp_git_repo), + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "No new branches" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_namespace_from_dotenv_file(tmp_git_repo): + """If DREAM_NAMESPACE is unset and a `.env` provides one, use it.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + (tmp_git_repo / ".env").write_text("DREAM_NAMESPACE=fromdotenv\n") + full_env = _cli_env() + full_env.pop("DREAM_NAMESPACE", None) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), "--dry-run"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "Namespace: fromdotenv" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_namespace_from_task_info_json(tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + (tmp_git_repo / "TASK_INFO.json").write_text( + json.dumps({"dream_namespace": "fromtaskinfo"}) + ) + full_env = _cli_env() + full_env.pop("DREAM_NAMESPACE", None) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), "--dry-run"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "Namespace: fromtaskinfo" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_namespace_falls_back_to_repo_basename(tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + full_env = _cli_env() + full_env.pop("DREAM_NAMESPACE", None) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), "--dry-run"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + # The fixture's repo dir is named "repo". + assert f"Namespace: {tmp_git_repo.name}" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_verify_only_empty_index_exits_zero(tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), "--verify-only"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0 + assert "No reconciled dreams" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_verify_only_passes_for_fully_reconciled_dream(tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-064000Z-vok" + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | dream/proj/{dream_id} | main | abc1234 |\n", + ) + _seed_dream_artifacts(tmp_git_repo, dream_id) + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), "--verify-only"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0 + assert "All 1 indexed dreams verified" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_verify_only_reports_missing_artifacts(tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-064100Z-broken" + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | dream/proj/{dream_id} | main | abc1234 |\n", + ) + # ONLY index, no dream artifacts. + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), "--verify-only"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 1 + assert "Verification FAILED" in result.stdout + assert "missing" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_cleanup_branches_after_reconcile(tmp_git_repo): + """Full reconcile + push + cleanup pipeline.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + d1 = "20260420-065000Z-cleanup" + m1 = _manifest_with(d1, discoveries=[ + _discovery("a.py::f", "Discovery for cleanup test."), + ]) + branch = make_dream_branch(tmp_git_repo, env, "proj", d1, m1, + report=_default_report(d1)) + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + + # Reconcile + commit + push so the cleanup ancestor check passes. + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo)], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "reconcile", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", "main", cwd=tmp_git_repo, env=env) + + # Re-run with --cleanup-branches; no new dreams → falls through. + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), + "--cleanup-branches"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "Deleted: 1" in result.stdout + # Branch gone from origin. + post = _git("ls-remote", "--heads", "origin", branch, + cwd=tmp_git_repo, env=env).stdout + assert branch not in post + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_cleanup_no_new_no_indexed_exits_zero(tmp_git_repo): + """`--cleanup-branches` with no new branches AND empty index exits 0.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo), + "--cleanup-branches"], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0 + assert "Nothing in _index.md to clean up" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_corrupted_manifest_json_skipped(tmp_git_repo): + """A branch with malformed JSON in manifest.json must be reported as + SKIP without crashing the run.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + bad_id = "20260420-066000Z-badjson" + branch = f"dream/proj/{bad_id}" + original = _git("rev-parse", "--abbrev-ref", "HEAD", + cwd=tmp_git_repo, env=env).stdout.strip() + _git("checkout", "-q", "-b", branch, cwd=tmp_git_repo, env=env) + dream_dir = tmp_git_repo / ".shadow" / "_dreams" / bad_id + dream_dir.mkdir(parents=True) + (dream_dir / "manifest.json").write_text("{this is not json") + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "bad json", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", branch, cwd=tmp_git_repo, env=env) + _git("checkout", "-q", original, cwd=tmp_git_repo, env=env) + + full_env = _cli_env({"DREAM_NAMESPACE": "proj"}) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(tmp_git_repo)], + capture_output=True, text=True, env=full_env, + ) + assert result.returncode == 0 + assert "SKIP" in result.stdout + assert "invalid JSON" in result.stdout + + +# --- Regression: update_index dry-run must not write to disk --- +# Bug surfaced by Phase-3 audit: prior update_index bootstrapped +# _dreams/_index.md (creating directory + file) BEFORE the dry_run check, +# so --dry-run silently materialized an empty index file. Dry-run must +# be fully read-only. + +class TestUpdateIndexDryRunIsReadOnly: + def test_dry_run_does_not_create_index_when_missing( + self, dream_reconcile, tmp_git_repo + ): + shadow = tmp_git_repo / ".shadow" + manifests = [ + ("dream/proj/xx", "20260420-1500Z-feat", + {"category": "feature", "verdict": "useful"}), + ] + dream_reconcile.update_index(str(tmp_git_repo), manifests, dry_run=True) + + assert not (shadow / "_dreams" / "_index.md").exists() + assert not (shadow / "_dreams").exists() + + def test_dry_run_does_not_create_dreams_dir_when_shadow_exists( + self, dream_reconcile, tmp_git_repo + ): + shadow = tmp_git_repo / ".shadow" + shadow.mkdir() + manifests = [ + ("dream/proj/xx", "20260420-1500Z-feat", + {"category": "feature", "verdict": "useful"}), + ] + dream_reconcile.update_index(str(tmp_git_repo), manifests, dry_run=True) + + assert not (shadow / "_dreams").exists() + assert not (shadow / "_dreams" / "_index.md").exists() + + def test_dry_run_does_not_append_when_index_exists( + self, dream_reconcile, tmp_git_repo + ): + shadow_dreams = tmp_git_repo / ".shadow" / "_dreams" + shadow_dreams.mkdir(parents=True) + index_path = shadow_dreams / "_index.md" + original = ( + "# Dream Index\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + ) + index_path.write_text(original) + + manifests = [ + ("dream/proj/xx", "20260420-1500Z-feat", + {"category": "feature", "verdict": "useful"}), + ] + dream_reconcile.update_index(str(tmp_git_repo), manifests, dry_run=True) + + assert index_path.read_text() == original + + def test_dry_run_still_prints_preview( + self, dream_reconcile, tmp_git_repo, capsys + ): + manifests = [ + ("dream/proj/xx", "20260420-1500Z-foo", + {"category": "feature", "verdict": "useful"}), + ("dream/proj/yy", "20260420-1600Z-bar", + {"category": "bug", "verdict": "dead_end"}), + ] + dream_reconcile.update_index(str(tmp_git_repo), manifests, dry_run=True) + out = capsys.readouterr().out + assert "Would index: 20260420-1500Z-foo" in out + assert "Would index: 20260420-1600Z-bar" in out + + +# =========================================================================== +# Canonical header helpers — B4 regression +# (reconciler-created shadows must match init-created ones) +# =========================================================================== + +class TestCanonicalHeaderHelpers: + @pytest.mark.parametrize("rel,lang", [ + ("src/auth.py", "Python"), + ("lib/http.go", "Go"), + ("app/main.ts", "TypeScript"), + ("weird.unknownext", "Unknown"), + ]) + def test_detect_language(self, dream_reconcile, rel, lang): + assert dream_reconcile._detect_language(rel) == lang + + def test_detect_language_basename(self, dream_reconcile): + # Basename map (e.g. Dockerfile/Makefile) must win over extension. + assert dream_reconcile._detect_language("ops/Dockerfile") != "Unknown" + + @pytest.mark.parametrize("shadow,expected", [ + ("/repo/.shadow/src/auth.py.md", "src/auth.py"), + ("/repo/.shadow/main.go.md", "main.go"), + ("/repo/nested/.shadow/a/b/c.ts.md", "a/b/c.ts"), + ]) + def test_source_rel_from_shadow(self, dream_reconcile, shadow, expected): + assert dream_reconcile._source_rel_from_shadow(shadow) == expected + + def test_canonical_header_lines_match_init_template(self, dream_reconcile): + lines = dream_reconcile._canonical_header_lines( + "/repo/.shadow/src/auth.py.md" + ) + text = "".join(lines) + assert text.startswith("# Shadow: src/auth.py\n") + assert "**Language**: Python\n" in text + assert "## File-Level\n" in text + assert "_No discoveries yet._\n" in text + + def test_canonical_header_unknown_language(self, dream_reconcile): + text = "".join( + dream_reconcile._canonical_header_lines("/r/.shadow/x.qqq.md") + ) + assert "**Language**: Unknown\n" in text + + +# =========================================================================== +# _merge_refs_into_cross_file — B14 regression (ref union, not drop) +# =========================================================================== + +class TestMergeRefsIntoCrossFile: + def _write_cross(self, tmp_path, refs): + p = tmp_path / "slug.md" + body = ["# Title", "", "**Category**: pattern", "**Refs**:"] + body += [f"- `{r}`" for r in refs] + body += ["", "**Discovery**: something", ""] + p.write_text("\n".join(body)) + return p + + def test_unions_new_refs(self, dream_reconcile, tmp_path): + p = self._write_cross(tmp_path, ["src/a.py::f"]) + changed = dream_reconcile._merge_refs_into_cross_file( + str(p), ["src/b.py::g", "src/a.py::f"] + ) + assert changed is True + text = p.read_text() + assert "- `src/a.py::f`" in text + assert "- `src/b.py::g`" in text + # No duplication of the already-present ref. + assert text.count("- `src/a.py::f`") == 1 + + def test_no_change_when_all_present(self, dream_reconcile, tmp_path): + p = self._write_cross(tmp_path, ["src/a.py::f"]) + before = p.read_text() + changed = dream_reconcile._merge_refs_into_cross_file( + str(p), ["src/a.py::f"] + ) + assert changed is False + assert p.read_text() == before + + def test_missing_file_returns_false(self, dream_reconcile, tmp_path): + assert dream_reconcile._merge_refs_into_cross_file( + str(tmp_path / "nope.md"), ["x::y"] + ) is False + + def test_no_refs_block_returns_false(self, dream_reconcile, tmp_path): + p = tmp_path / "norefs.md" + p.write_text("# Title\n\nNo refs section here.\n") + assert dream_reconcile._merge_refs_into_cross_file( + str(p), ["x::y"] + ) is False + # File untouched. + assert "No refs section here." in p.read_text() + + +# =========================================================================== +# _resolve_tip_commit — B10 regression (no 7-char truncation, hex-validated) +# =========================================================================== + +class TestResolveTipCommit: + def test_unknown_when_branch_absent(self, dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + assert dream_reconcile._resolve_tip_commit( + str(tmp_git_repo), "does-not-exist" + ) == "unknown" + + def test_resolves_real_branch_to_hex_sha(self, dream_reconcile, tmp_git_repo): + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-030000Z-boot" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + tip = dream_reconcile._resolve_tip_commit(str(tmp_git_repo), branch) + assert tip != "unknown" + assert re.fullmatch(r"[0-9a-fA-F]{7,40}", tip) + # Matches the branch's actual full SHA prefix (not truncated to 7). + full = _git("rev-parse", f"origin/{branch}", + cwd=tmp_git_repo, env=env).stdout.strip() + assert full.startswith(tip) + + +# =========================================================================== +# Worktree GC — Fix 3 from bug-worktree-leak.md +# +# After cleanup_branches successfully deletes a dream branch, the matching +# worktree directory at $DREAM_WORKTREE_BASE/<ns>/dream-<slug>/ must also +# be removed. Pre-fix: directories accumulated forever (538/540 leaked on +# matplotlib; 691/693 on sphinx per the bug report). +# =========================================================================== + +class TestSlugFromDreamId: + """Direct tests for the slug-derivation helper. Critical: dream_ids + contain '-' (the date itself has one), so a naive partition('-') yields + the timestamp tail, NOT the slug. Worktree GC would target the wrong + directory.""" + + def test_extracts_simple_slug(self, dream_reconcile): + assert dream_reconcile._slug_from_dream_id( + "20260420-050000Z-foo" + ) == "foo" + + def test_extracts_multipart_slug_with_dashes(self, dream_reconcile): + """Slug like 't01-csv-fuzzer' — must NOT lose internal dashes.""" + assert dream_reconcile._slug_from_dream_id( + "20260420-050000Z-t01-csv-fuzzer" + ) == "t01-csv-fuzzer" + + def test_extracts_slug_with_dots_and_underscores(self, dream_reconcile): + assert dream_reconcile._slug_from_dream_id( + "20260420-050000Z-v1.2.3_beta" + ) == "v1.2.3_beta" + + @pytest.mark.parametrize("bad", [ + "", # empty + None, # not a string + "garbage", # no timestamp + "2026-04-20-foo", # wrong date shape + "20260420050000Z-foo", # missing dash between date and time + "20260420-050000-foo", # missing Z + ]) + def test_returns_none_for_malformed_id(self, dream_reconcile, bad): + assert dream_reconcile._slug_from_dream_id(bad) is None + + +@pytest.mark.slow +def test_cleanup_branches_also_removes_worktree( + dream_reconcile, tmp_git_repo, tmp_path, monkeypatch +): + """Full integration: after cleanup_branches deletes the branch, the + worktree directory at ${DREAM_WORKTREE_BASE}/<ns>/dream-<slug>/ must + also be gone. Pre-fix this directory leaked forever.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-070000Z-gctest" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + _seed_dream_artifacts(tmp_git_repo, dream_id) + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | {branch} | main | abc1234 |\n", + ) + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "reconcile", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", "main", cwd=tmp_git_repo, env=env) + + # Build the worktree dir the reconciler will look for. + base = tmp_path / "wt-base" + worktree_dir = base / "proj" / "dream-gctest" + worktree_dir.parent.mkdir(parents=True) + _git("worktree", "add", "-q", str(worktree_dir), branch, + cwd=tmp_git_repo, env=env) + assert worktree_dir.is_dir() + + monkeypatch.setenv("DREAM_WORKTREE_BASE", str(base)) + + deleted, _ = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(branch, dream_id, _default_manifest(dream_id))], + "proj", + dry_run=False, + ) + assert deleted == 1 + assert not worktree_dir.exists(), ( + "worktree GC didn't fire — bug-worktree-leak.md regression" + ) + + +@pytest.mark.slow +def test_cleanup_branches_worktree_gc_falls_back_on_dead_gitdir( + dream_reconcile, tmp_git_repo, tmp_path, monkeypatch +): + """The whole reason this fix exists: git worktree remove fails silently + when the gitdir pointer is broken. The reconciler's GC must fall back + to rm -rf (safety-gated) so the directory does NOT leak.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-080000Z-dead" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + _seed_dream_artifacts(tmp_git_repo, dream_id) + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | {branch} | main | abc1234 |\n", + ) + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "reconcile", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", "main", cwd=tmp_git_repo, env=env) + + # Manually build a DEAD worktree directory at the expected location + # (broken .git pointer — git worktree remove will refuse to clean it). + base = tmp_path / "wt-base" + worktree_dir = base / "proj" / "dream-dead" + worktree_dir.mkdir(parents=True) + (worktree_dir / ".git").write_text("gitdir: /nonexistent/wt-dir\n") + (worktree_dir / "leaked.pyc").write_text("# leak me\n") + + monkeypatch.setenv("DREAM_WORKTREE_BASE", str(base)) + + deleted, _ = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(branch, dream_id, _default_manifest(dream_id))], + "proj", + dry_run=False, + ) + assert deleted == 1 + assert not worktree_dir.exists(), ( + "fallback rm -rf must clean dead worktrees" + ) + + +@pytest.mark.slow +def test_cleanup_branches_worktree_gc_refuses_unsafe_base( + dream_reconcile, tmp_git_repo, tmp_path, monkeypatch, capsys +): + """If $DREAM_WORKTREE_BASE is set to something sensitive (/, /tmp, $HOME), + the safety gate must refuse the rm even though the branch delete itself + succeeded. The branch should still be deleted (it's already been + successfully pushed and verified).""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + dream_id = "20260420-090000Z-unsafe" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + _seed_dream_artifacts(tmp_git_repo, dream_id) + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | {branch} | main | abc1234 |\n", + ) + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "reconcile", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", "main", cwd=tmp_git_repo, env=env) + + # Decoy at /tmp/proj/dream-unsafe — must NOT be touched. + decoy = tmp_path / "should-not-be-deleted" + decoy.mkdir() + (decoy / "important.txt").write_text("keep me\n") + + # Point DREAM_WORKTREE_BASE at /tmp — gate must refuse. + monkeypatch.setenv("DREAM_WORKTREE_BASE", "/tmp") + + deleted, _ = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(branch, dream_id, _default_manifest(dream_id))], + "proj", + dry_run=False, + ) + # Branch delete itself MUST still succeed — GC failure is non-fatal. + assert deleted == 1 + captured = capsys.readouterr() + assert "Skipping worktree GC" in captured.out or "sensitive root" in captured.out + # Decoy is untouched. + assert decoy.is_dir() + assert (decoy / "important.txt").read_text() == "keep me\n" + + +@pytest.mark.slow +def test_cleanup_branches_worktree_gc_skips_unparseable_dream_id( + dream_reconcile, tmp_git_repo, tmp_path, monkeypatch, capsys +): + """A malformed dream_id (manifest schema drift, future migration, …) + must NOT crash cleanup_branches. The GC bails silently and the branch + is still deleted.""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + # Branch uses a malformed dream_id (no Z, wrong shape). + dream_id = "not-a-real-dream-id" + branch = make_dream_branch(tmp_git_repo, env, "proj", dream_id, + _default_manifest(dream_id)) + _seed_dream_artifacts(tmp_git_repo, dream_id) + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_id} | bug hunting | useful | T | {branch} | main | abc1234 |\n", + ) + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "reconcile", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", "main", cwd=tmp_git_repo, env=env) + + monkeypatch.setenv("DREAM_WORKTREE_BASE", str(tmp_path / "wt-base")) + + deleted, _ = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(branch, dream_id, _default_manifest(dream_id))], + "proj", + dry_run=False, + ) + assert deleted == 1 # branch still deleted + + +# =========================================================================== +# Cross-deletion guard for slug-collision (S2 from 5-model review panel) +# =========================================================================== +# Worktree paths are slug-only (see dream-setup.sh:165), but dream_ids +# include a timestamp. So two dreams with the same slug at different +# times share a worktree path. If dream A is reconciled AFTER dream B has +# reclaimed the shared path, A's GC must NOT delete B's live worktree. + + +class TestRegisteredWorktreeBranch: + """Unit-level coverage of _registered_worktree_branch — the helper + that lets the GC distinguish 'our worktree' from 'someone else's + worktree at the same shared path'.""" + + @pytest.mark.slow + def test_returns_branch_for_registered_path( + self, dream_reconcile, tmp_git_repo, tmp_path + ): + env = _seed_repo(tmp_git_repo) + wt = tmp_path / "wt" + _git("worktree", "add", "-q", "-b", "feature-x", str(wt), + cwd=tmp_git_repo, env=env) + branch = dream_reconcile._registered_worktree_branch( + str(tmp_git_repo), str(wt) + ) + assert branch == "feature-x" + + @pytest.mark.slow + def test_returns_none_for_unregistered_path( + self, dream_reconcile, tmp_git_repo, tmp_path + ): + _seed_repo(tmp_git_repo) + # A path that's not registered as a worktree at all. + branch = dream_reconcile._registered_worktree_branch( + str(tmp_git_repo), str(tmp_path / "nope") + ) + assert branch is None + + @pytest.mark.slow + def test_matches_through_symlink( + self, dream_reconcile, tmp_git_repo, tmp_path + ): + """macOS /tmp ↔ /private/tmp scenario: the path we query may + differ from the path git recorded, but realpath unifies them.""" + env = _seed_repo(tmp_git_repo) + real_wt = tmp_path / "real-wt" + link_wt = tmp_path / "link-wt" + _git("worktree", "add", "-q", "-b", "feature-y", str(real_wt), + cwd=tmp_git_repo, env=env) + link_wt.symlink_to(real_wt) + branch_via_link = dream_reconcile._registered_worktree_branch( + str(tmp_git_repo), str(link_wt) + ) + assert branch_via_link == "feature-y" + + +@pytest.mark.slow +def test_cleanup_branches_does_not_clobber_concurrent_slug_collision( + dream_reconcile, tmp_git_repo, tmp_path, monkeypatch, capsys +): + """S2 from review panel — REPRODUCES the cross-deletion bug. + + Setup: two dreams share slug='same' at different timestamps. + Dream A: 20260420-100000Z-same + Dream B: 20260420-110000Z-same (created after A) + Dream B has reclaimed the shared worktree path; A's branch exists + but has no worktree of its own. Reconciling A must NOT touch B's + live worktree (where uncommitted work would otherwise be lost).""" + env = _seed_repo(tmp_git_repo) + _add_bare_remote(tmp_git_repo, env) + base = tmp_path / "wt-base" + base.mkdir() + monkeypatch.setenv("DREAM_WORKTREE_BASE", str(base)) + + # Build dream A's branch (no worktree — simulating the leak case + # where the path was reclaimed by a later dream). + dream_a_id = "20260420-100000Z-same" + dream_b_id = "20260420-110000Z-same" + branch_a = make_dream_branch(tmp_git_repo, env, "proj", dream_a_id, + _default_manifest(dream_a_id)) + branch_b = make_dream_branch(tmp_git_repo, env, "proj", dream_b_id, + _default_manifest(dream_b_id)) + _seed_dream_artifacts(tmp_git_repo, dream_a_id) + _write_index( + tmp_git_repo, + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + f"| {dream_a_id} | bug hunting | useful | A | {branch_a} | main | abc1234 |\n", + ) + _git("add", "-A", cwd=tmp_git_repo, env=env) + _git("commit", "-q", "-m", "reconcile A", cwd=tmp_git_repo, env=env) + _git("push", "-q", "origin", "main", cwd=tmp_git_repo, env=env) + + # Dream B claims the shared path. (In production, dream-setup.sh's + # idempotent pre-clean would have already wiped any stale A worktree.) + shared_path = base / "proj" / "dream-same" + shared_path.parent.mkdir(parents=True) + _git("worktree", "add", "-q", str(shared_path), branch_b, + cwd=tmp_git_repo, env=env) + assert shared_path.is_dir() + # Add a "user file" — proxy for uncommitted work that must survive. + (shared_path / "uncommitted-work.txt").write_text("PRECIOUS WORK\n") + + # Reconcile A — its GC must NOT take down B's worktree. + deleted, _ = dream_reconcile.cleanup_branches( + str(tmp_git_repo), + [(branch_a, dream_a_id, _default_manifest(dream_a_id))], + "proj", + dry_run=False, + ) + assert deleted == 1, "A's branch should have been cleaned up" + + # The critical assertion: B's worktree must survive intact. + assert shared_path.is_dir(), ( + "B's live worktree was DESTROYED by A's reconciler GC — " + "cross-deletion regression (S2)" + ) + assert (shared_path / "uncommitted-work.txt").read_text() == "PRECIOUS WORK\n", ( + "user's uncommitted work in B's worktree was lost" + ) + # B's branch must still exist too. + branches_out = _git("branch", cwd=tmp_git_repo, env=env).stdout + assert branch_b in branches_out, f"B's branch is gone: {branches_out}" + + # The skip should have been LOGGED so operators can audit. + captured = capsys.readouterr() + assert "now belongs to" in captured.out, ( + f"cross-deletion skip was silent — operators won't know:\n{captured.out}" + ) diff --git a/tests/skills/shadow_frog_dream/test_dream_setup_sh.py b/tests/skills/shadow_frog_dream/test_dream_setup_sh.py new file mode 100644 index 0000000..ef02238 --- /dev/null +++ b/tests/skills/shadow_frog_dream/test_dream_setup_sh.py @@ -0,0 +1,557 @@ +"""Tests for skills/shadow-frog-dream/dream-setup.sh — Dream worktree setup. + +Exercises: --help, happy path worktree+branch creation, RUN_PREFIX detection, +namespace override, slug validation, and dry-run mode. +""" +import json +import os +import subprocess +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent +DREAM_SETUP = REPO_ROOT / "skills" / "shadow-frog-dream" / "dream-setup.sh" + + +def _base_env(cwd: Path, extras: dict | None = None) -> dict: + env = { + "PATH": os.environ.get("PATH", "/usr/bin:/bin:/usr/local/bin"), + "HOME": str(cwd), + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "LANG": "en_US.UTF-8", + } + if extras: + env.update(extras) + return env + + +def _make_git_repo(path: Path, branch: str = "main") -> None: + """Create a git repo with an initial commit and origin/main ref.""" + env = _base_env(path) + subprocess.run(["git", "init", "-q", "-b", branch], cwd=path, check=True, env=env) + subprocess.run(["git", "config", "user.email", "test@test.invalid"], cwd=path, check=True, env=env) + subprocess.run(["git", "config", "user.name", "Test"], cwd=path, check=True, env=env) + subprocess.run(["git", "config", "commit.gpgsign", "false"], cwd=path, check=True, env=env) + (path / "README.md").write_text("# test\n") + subprocess.run(["git", "add", "-A"], cwd=path, check=True, env=env) + subprocess.run(["git", "commit", "-q", "-m", "init"], cwd=path, check=True, env=env) + # Create a fake origin remote pointing to self for origin/main ref + subprocess.run(["git", "remote", "add", "origin", str(path)], cwd=path, check=True, env=env) + subprocess.run(["git", "fetch", "-q", "origin"], cwd=path, check=True, env=env) + + +def run_dream_setup( + args: list[str], cwd: Path, env_extra: dict | None = None +) -> subprocess.CompletedProcess: + """Run dream-setup.sh with given args.""" + env = _base_env(cwd, env_extra) + return subprocess.run( + ["bash", str(DREAM_SETUP), *args], + capture_output=True, + text=True, + cwd=cwd, + env=env, + ) + + +@pytest.mark.slow +@pytest.mark.integration +class TestDreamSetupHelp: + def test_help_exits_zero(self, tmp_path): + # --help should work even outside a git repo (it just prints and exits) + result = run_dream_setup(["--help"], cwd=tmp_path) + assert result.returncode == 0 + assert "slug" in result.stdout.lower() or "slug" in result.stderr.lower() or "Usage" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +class TestDreamSetupHappyPath: + """Creates worktree and branch correctly.""" + + def test_creates_worktree_and_branch(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "worktrees" + + result = run_dream_setup( + ["--slug", "t01-test", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={"DREAM_WORKTREE_BASE": str(worktree_base)}, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + data = json.loads(result.stdout) + + assert "dream_ns" in data + assert "branch_name" in data + assert "worktree_dir" in data + assert data["slug"] == "t01-test" + assert "dream/" in data["branch_name"] + assert "t01-test" in data["dream_id"] + + # Verify worktree exists + wt_dir = Path(data["worktree_dir"]) + assert wt_dir.is_dir() + + # Verify branch exists in repo + env = _base_env(repo) + branches = subprocess.run( + ["git", "branch", "--list", data["branch_name"]], + cwd=repo, capture_output=True, text=True, env=env, + ) + # Branch may be in worktree, check via worktree list + wt_list = subprocess.run( + ["git", "worktree", "list"], cwd=repo, + capture_output=True, text=True, env=env, + ) + assert str(wt_dir) in wt_list.stdout + + def test_worktree_has_same_head_as_base(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "worktrees" + env = _base_env(repo) + + # Get main HEAD + main_head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=repo, + capture_output=True, text=True, check=True, env=env, + ).stdout.strip() + + result = run_dream_setup( + ["--slug", "t02-head", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={"DREAM_WORKTREE_BASE": str(worktree_base)}, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + data = json.loads(result.stdout) + assert data["base_commit"] == main_head + + +@pytest.mark.slow +@pytest.mark.integration +class TestDreamSetupRunPrefix: + """RUN_PREFIX detection based on lock files.""" + + def test_no_lock_files_empty_prefix(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "wt" + + result = run_dream_setup( + ["--slug", "t03-nolock", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={"DREAM_WORKTREE_BASE": str(worktree_base)}, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + data = json.loads(result.stdout) + assert data["run_prefix"] == "" + + def test_uv_lock_gives_uv_run_prefix(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + # Add uv.lock + (repo / "uv.lock").write_text("") + worktree_base = tmp_path / "wt" + + result = run_dream_setup( + ["--slug", "t04-uvlock", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={"DREAM_WORKTREE_BASE": str(worktree_base)}, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + data = json.loads(result.stdout) + assert data["run_prefix"] == "uv run" + + def test_package_lock_gives_npx_prefix(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + (repo / "package-lock.json").write_text("{}") + worktree_base = tmp_path / "wt" + + result = run_dream_setup( + ["--slug", "t05-npm", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={"DREAM_WORKTREE_BASE": str(worktree_base)}, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + data = json.loads(result.stdout) + assert data["run_prefix"] == "npx" + + +@pytest.mark.slow +@pytest.mark.integration +class TestDreamSetupIdempotent: + """Re-running with same slug cleans and recreates (idempotent).""" + + def test_same_slug_twice_succeeds(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "wt" + + args = ["--slug", "t06-idem", "--repo-root", str(repo), "--print-json"] + extras = {"DREAM_WORKTREE_BASE": str(worktree_base)} + + r1 = run_dream_setup(args, cwd=repo, env_extra=extras) + assert r1.returncode == 0, f"stderr: {r1.stderr}" + + r2 = run_dream_setup(args, cwd=repo, env_extra=extras) + assert r2.returncode == 0, f"stderr: {r2.stderr}" + # Both should produce valid JSON with same worktree dir + d1 = json.loads(r1.stdout) + d2 = json.loads(r2.stdout) + assert d1["worktree_dir"] == d2["worktree_dir"] + + +@pytest.mark.slow +@pytest.mark.integration +class TestDreamSetupNamespace: + """DREAM_NAMESPACE / --namespace honored in branch name.""" + + def test_namespace_override_in_branch(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "wt" + + result = run_dream_setup( + ["--slug", "t07-ns", "--namespace", "my-custom-ns", + "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={"DREAM_WORKTREE_BASE": str(worktree_base)}, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + data = json.loads(result.stdout) + assert data["dream_ns"] == "my-custom-ns" + assert "dream/my-custom-ns/" in data["branch_name"] + + def test_env_namespace_used(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "wt" + + result = run_dream_setup( + ["--slug", "t08-envns", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={ + "DREAM_WORKTREE_BASE": str(worktree_base), + "DREAM_NAMESPACE": "env-ns-test", + }, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + data = json.loads(result.stdout) + assert data["dream_ns"] == "env-ns-test" + + +@pytest.mark.slow +@pytest.mark.integration +class TestDreamSetupValidation: + """Input validation prevents bad slugs.""" + + def test_missing_slug_fails(self, tmp_path): + result = run_dream_setup([], cwd=tmp_path) + assert result.returncode != 0 + assert "slug" in result.stderr.lower() + + def test_invalid_slug_rejected(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + result = run_dream_setup( + ["--slug", "bad slug!!", "--repo-root", str(repo)], + cwd=repo, + ) + assert result.returncode != 0 + assert "must match" in result.stderr + + +@pytest.mark.slow +@pytest.mark.integration +class TestDreamSetupGitignoreGuard: + """dream requires .shadow/ to be git-tracked, not gitignored.""" + + def test_refuses_when_shadow_gitignored(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + (repo / ".gitignore").write_text(".shadow/\n") + (repo / ".shadow").mkdir() + result = run_dream_setup( + ["--slug", "t10-ignored", "--repo-root", str(repo), + "--dry-run", "--print-json"], + cwd=repo, + ) + assert result.returncode != 0 + assert ".shadow/ is gitignored" in result.stderr + + def test_proceeds_when_shadow_tracked(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + # .gitignore present but does NOT ignore .shadow/ + (repo / ".gitignore").write_text("build/\n__pycache__/\n") + (repo / ".shadow").mkdir() + worktree_base = tmp_path / "wt" + result = run_dream_setup( + ["--slug", "t10-tracked", "--repo-root", str(repo), + "--dry-run", "--print-json"], + cwd=repo, + env_extra={"DREAM_WORKTREE_BASE": str(worktree_base)}, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + assert "gitignored" not in result.stderr + + def test_refuses_when_shadow_gitignored_but_already_tracked(self, tmp_path): + """Edge case: .shadow/ is gitignored AND has previously-committed + content. `git check-ignore .shadow` reports not-ignored (tracked wins), + but `git add -A` still drops NEW children — so the guard must probe a + child path and still refuse.""" + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + # Commit some .shadow content FIRST, then add the gitignore rule. + shadow_meta = repo / ".shadow" / "_meta" + shadow_meta.mkdir(parents=True) + (shadow_meta / "state.json").write_text("{}\n") + env = _base_env(repo) + subprocess.run(["git", "add", "-A"], cwd=repo, check=True, env=env) + subprocess.run(["git", "commit", "-qm", "track shadow"], + cwd=repo, check=True, env=env) + (repo / ".gitignore").write_text(".shadow/\n") + subprocess.run(["git", "add", ".gitignore"], cwd=repo, check=True, env=env) + subprocess.run(["git", "commit", "-qm", "ignore shadow"], + cwd=repo, check=True, env=env) + + result = run_dream_setup( + ["--slug", "t10-tracked-ignored", "--repo-root", str(repo), + "--dry-run", "--print-json"], + cwd=repo, + ) + assert result.returncode != 0 + assert ".shadow/ is gitignored" in result.stderr + + +@pytest.mark.slow +@pytest.mark.integration +class TestDreamSetupDryRun: + """--dry-run computes values without creating worktree.""" + + def test_dry_run_no_worktree_created(self, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "wt" + + result = run_dream_setup( + ["--slug", "t09-dry", "--repo-root", str(repo), + "--dry-run", "--print-json"], + cwd=repo, + env_extra={"DREAM_WORKTREE_BASE": str(worktree_base)}, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + data = json.loads(result.stdout) + # Worktree dir should NOT exist + assert not Path(data["worktree_dir"]).exists() + assert data["slug"] == "t09-dry" + + +# =========================================================================== +# Auto-GC throttle (Bug A fix from bug-cleanup-gaps.md) +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +class TestDreamSetupAutoGC: + """`dream-setup.sh` triggers `dream-gc.sh` periodically. + + Bug A from bug-cleanup-gaps.md: dream-gc.sh existed but had no caller + in the skill flow, so long-running fleets accumulated orphans + indefinitely. dream-setup.sh now invokes it at the start of each new + dream, throttled to once per DREAM_GC_INTERVAL_MIN (default 60). + """ + + def _orphan(self, base: Path, ns: str, name: str = "dream-orphan") -> Path: + """Plant an orphan worktree under the namespace dir.""" + d = base / ns / name + d.mkdir(parents=True) + (d / ".git").write_text("gitdir: /nonexistent/path\n") + (d / "leftover.txt").write_text("orphaned\n") + ancient = 946684800 + os.utime(d, (ancient, ancient)) + return d + + def test_auto_gc_runs_when_no_tombstone(self, tmp_path): + """First invocation sweeps orphans (no tombstone yet).""" + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "worktrees" + ns = repo.name + + # Plant an orphan that the auto-GC should clean up. + orphan = self._orphan(worktree_base, ns) + assert orphan.exists() + + result = run_dream_setup( + ["--slug", "t01-gc", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={ + "DREAM_WORKTREE_BASE": str(worktree_base), + # Force min-age-min=0 so the ancient orphan is in the find window + "DREAM_GC_AGE_MIN": "0", + }, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + # Orphan must be gone — auto-GC ran. + assert not orphan.exists(), ( + f"Auto-GC should have swept the orphan\nstderr: {result.stderr}" + ) + # Tombstone created. + tombstone = worktree_base / ns / ".last-gc" + assert tombstone.exists() + + def test_auto_gc_throttled_by_recent_tombstone(self, tmp_path): + """Fresh tombstone (< DREAM_GC_INTERVAL_MIN) suppresses the trigger.""" + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "worktrees" + ns = repo.name + + # Pre-create the tombstone with current mtime (fresh). + (worktree_base / ns).mkdir(parents=True) + tombstone = worktree_base / ns / ".last-gc" + tombstone.touch() + + orphan = self._orphan(worktree_base, ns) + + result = run_dream_setup( + ["--slug", "t02-throttle", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={ + "DREAM_WORKTREE_BASE": str(worktree_base), + "DREAM_GC_INTERVAL_MIN": "60", # tombstone is fresh, won't trigger + "DREAM_GC_AGE_MIN": "0", + }, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + # Orphan must STILL exist — auto-GC was throttled. + assert orphan.exists(), ( + f"Fresh tombstone should suppress auto-GC\nstderr: {result.stderr}" + ) + + def test_auto_gc_disabled_via_env(self, tmp_path): + """`DREAM_GC_AUTO=0` opts out of the auto-trigger entirely.""" + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "worktrees" + ns = repo.name + + orphan = self._orphan(worktree_base, ns) + + result = run_dream_setup( + ["--slug", "t03-disabled", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={ + "DREAM_WORKTREE_BASE": str(worktree_base), + "DREAM_GC_AUTO": "0", + "DREAM_GC_AGE_MIN": "0", + }, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + # Orphan must STILL exist — GC was opt'd out. + assert orphan.exists(), ( + f"DREAM_GC_AUTO=0 should disable auto-GC\nstderr: {result.stderr}" + ) + # Tombstone NOT created. + assert not (worktree_base / ns / ".last-gc").exists() + + def test_auto_gc_invalid_env_warns_and_continues(self, tmp_path): + """Non-integer interval/age must not break dream setup.""" + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "worktrees" + + result = run_dream_setup( + ["--slug", "t04-badenv", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={ + "DREAM_WORKTREE_BASE": str(worktree_base), + "DREAM_GC_INTERVAL_MIN": "not-a-number", + }, + ) + # Dream setup must still succeed — auto-GC is best-effort. + assert result.returncode == 0, f"stderr: {result.stderr}" + data = json.loads(result.stdout) + # Worktree was still created. + assert Path(data["worktree_dir"]).is_dir() + + def test_auto_gc_does_not_pollute_eval_stdout(self, tmp_path): + """Auto-GC output MUST go to stderr to preserve the `eval` contract.""" + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "worktrees" + ns = repo.name + + # Plant an orphan so the GC actually has work to log about. + self._orphan(worktree_base, ns) + + # Use --print-env (NOT --print-json) — this is the eval-consumed path. + result = run_dream_setup( + ["--slug", "t05-stdout", "--repo-root", str(repo)], # default = --print-env + cwd=repo, + env_extra={ + "DREAM_WORKTREE_BASE": str(worktree_base), + "DREAM_GC_AGE_MIN": "0", + }, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + # Stdout must contain ONLY export lines — nothing from the GC. + for line in result.stdout.splitlines(): + stripped = line.strip() + if not stripped: + continue + assert stripped.startswith("export "), ( + f"Non-export line on stdout would break `eval`: {line!r}\n" + f"Full stdout:\n{result.stdout}" + ) + + def test_auto_gc_sweeps_other_namespace_orphans_too(self, tmp_path): + """The auto-trigger sweeps the whole base, not just its own ns. + + That's deliberate — leaks in any ns count, and the find walk is cheap. + """ + repo = tmp_path / "repo" + repo.mkdir() + _make_git_repo(repo) + worktree_base = tmp_path / "worktrees" + + # Orphan under a DIFFERENT namespace + other_orphan = self._orphan(worktree_base, ns="other-repo") + assert other_orphan.exists() + + result = run_dream_setup( + ["--slug", "t06-cross", "--repo-root", str(repo), "--print-json"], + cwd=repo, + env_extra={ + "DREAM_WORKTREE_BASE": str(worktree_base), + "DREAM_GC_AGE_MIN": "0", + }, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + # The cross-namespace orphan was swept too. + assert not other_orphan.exists(), ( + f"Auto-GC sweeps the whole base\nstderr: {result.stderr}" + ) diff --git a/tests/skills/shadow_frog_dream/test_dream_validate.py b/tests/skills/shadow_frog_dream/test_dream_validate.py new file mode 100644 index 0000000..2ede881 --- /dev/null +++ b/tests/skills/shadow_frog_dream/test_dream_validate.py @@ -0,0 +1,582 @@ +"""Tests for skills/shadow-frog-dream/dream-validate.py. + +dream-validate.py is the pre-commit hard gate: it inspects +`.shadow/_dreams/<dream_id>/` artifacts (report.md, manifest.json, +patch.diff) and exits 1 on any structural or semantic violation. It +also runs `git diff` against the report's `base_commit` to verify the +agent mirrored discoveries into per-file shadows. + +Tests construct a real dream tree in a tmp git repo and invoke the +script via subprocess so the exit-code contract is exercised +end-to-end. Most cases are marked `slow` because they shell out. +""" +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent +SCRIPT = REPO_ROOT / "skills" / "shadow-frog-dream" / "dream-validate.py" + + +# --- Helpers --------------------------------------------------------------- + +def _git_env(home: Path) -> dict: + return { + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "HOME": str(home), + "PATH": "/usr/bin:/bin:/usr/local/bin:/opt/homebrew/bin", + } + + +def _run(cmd, cwd, env=None, check=True): + return subprocess.run( + cmd, cwd=cwd, env=env, capture_output=True, text=True, check=check + ) + + +def _commit_base(repo: Path) -> str: + """Seed initial commit. Returns the base SHA (full).""" + env = _git_env(repo) + (repo / "README.md").write_text("# base\n") + _run(["git", "add", "-A"], cwd=repo, env=env) + _run(["git", "commit", "-q", "-m", "base"], cwd=repo, env=env) + return _run(["git", "rev-parse", "HEAD"], cwd=repo, env=env).stdout.strip() + + +def _run_validate(dream_id: str, worktree: Path) -> subprocess.CompletedProcess: + """Invoke dream-validate.py as a subprocess; never raises on non-zero.""" + return subprocess.run( + [sys.executable, str(SCRIPT), dream_id, str(worktree)], + capture_output=True, text=True, + ) + + +def _default_manifest(dream_id: str, **overrides) -> dict: + m = { + "dream_id": dream_id, + "branch": f"dream/proj/{dream_id}", + "parent_branch": "main", + "category": "bug hunting", + "verdict": "useful", + "title": "Test dream", + "discoveries": [], + "cross_cutting": [], + } + m.update(overrides) + return m + + +def _default_report(dream_id: str, base_commit: str, **fm_overrides) -> str: + fm = { + "dream_id": f'"{dream_id}"', + "category": "bug hunting", + "verdict": "useful", + "base_commit": base_commit, + "branch": f'"dream/proj/{dream_id}"', + "parent_branch": '"main"', + "remote": '"origin"', + } + fm.update(fm_overrides) + fm_lines = "\n".join(f"{k}: {v}" for k, v in fm.items()) + return f"---\n{fm_lines}\n---\n\n# {dream_id}\n\nBody.\n" + + +def _write_dream( + worktree: Path, + dream_id: str, + *, + manifest: dict | None = None, + report: str | None = None, + patch: str = "diff --git a/x b/x\n--- /dev/null\n+++ b/x\n@@\n+x\n", +): + d = worktree / ".shadow" / "_dreams" / dream_id + d.mkdir(parents=True, exist_ok=True) + if manifest is not None: + (d / "manifest.json").write_text(json.dumps(manifest, indent=2)) + if report is not None: + (d / "report.md").write_text(report) + (d / "patch.diff").write_text(patch) + return d + + +# =========================================================================== +# Help / arg parsing +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_help_exits_zero(): + result = subprocess.run( + [sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True, + ) + assert result.returncode == 0 + assert "Validate dream artifacts" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_no_args_exits_nonzero_and_prints_usage(): + result = subprocess.run( + [sys.executable, str(SCRIPT)], + capture_output=True, text=True, + ) + assert result.returncode == 1 + assert "Usage" in result.stdout or "Validate dream artifacts" in result.stdout + + +# =========================================================================== +# Missing directory / missing files +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_missing_dream_directory_fails(tmp_git_repo): + result = _run_validate("20260101-000000Z-nope", tmp_git_repo) + assert result.returncode == 1 + assert "Missing dream subdirectory" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_missing_manifest_and_report_fails(tmp_git_repo): + dream_id = "20260101-000000Z-missing" + d = tmp_git_repo / ".shadow" / "_dreams" / dream_id + d.mkdir(parents=True, exist_ok=True) + (d / "patch.diff").write_text("diff\n") + # No report.md, no manifest.json. + + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "Missing" in result.stdout + assert "report.md" in result.stdout + assert "manifest.json" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_flat_file_fails(tmp_git_repo): + """Old flat-file layout (.shadow/_dreams/<id>.md) is rejected.""" + dream_id = "20260101-000000Z-flat" + dreams = tmp_git_repo / ".shadow" / "_dreams" + dreams.mkdir(parents=True, exist_ok=True) + (dreams / f"{dream_id}.md").write_text("# flat\n") # forbidden + + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "Flat file found" in result.stdout + assert "MUST use subdirectory" in result.stdout + + +# =========================================================================== +# Empty patch.diff +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_empty_patch_diff_fails(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-emptypatch" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest(dream_id), + report=_default_report(dream_id, base), + patch="", # empty + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "patch.diff is empty" in result.stdout + assert "completion criterion #1" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_whitespace_only_patch_diff_fails(tmp_git_repo): + """Non-empty but content-free patch.diff (no diff markers) is rejected.""" + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-blankpatch" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest(dream_id), + report=_default_report(dream_id, base), + patch=" \n\n \n", # whitespace only — no diff markers + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "no unified-diff markers" in result.stdout + + +# =========================================================================== +# Frontmatter / manifest dream_id mismatch +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_report_dream_id_mismatch_fails(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-good" + wrong_id = "20260101-000000Z-WRONG" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest(dream_id), + # report says a different dream_id + report=_default_report(wrong_id, base).replace( + f"# {wrong_id}", f"# {dream_id}" + ), + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "dream_id mismatch" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_manifest_dream_id_mismatch_fails(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-good" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest("20260101-000000Z-DIFFERENT"), + report=_default_report(dream_id, base), + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "manifest.json dream_id mismatch" in result.stdout + + +# =========================================================================== +# Required fields / invalid category / invalid verdict +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_missing_required_field_fails(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-missingfield" + m = _default_manifest(dream_id) + del m["title"] + _write_dream( + tmp_git_repo, dream_id, + manifest=m, + report=_default_report(dream_id, base), + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "missing required fields" in result.stdout + assert "title" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_invalid_category_fails(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-badcat" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest(dream_id, category="navel-gazing"), + report=_default_report(dream_id, base), + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "Invalid category" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_invalid_verdict_fails(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-badverdict" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest(dream_id, verdict="maybe"), + report=_default_report(dream_id, base), + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "Invalid verdict" in result.stdout + + +# =========================================================================== +# Discovery op validation (update/refute not yet supported) +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +@pytest.mark.parametrize("bad_op", ["update", "refute"]) +def test_unsupported_discovery_op_fails(tmp_git_repo, bad_op): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-badop" + # We don't actually need shadow mirroring for the op check to fire (it + # runs before the mirror check). Use an empty discoveries list shaped + # in a way that the op error blocks first. + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest( + dream_id, + discoveries=[{ + "op": bad_op, + "anchor": "x.py::y", + "text": "Something here.", + }], + ), + report=_default_report(dream_id, base), + ) + # Add a mirror so the OTHER hard-error (mirror check) doesn't drown + # out our op message — though both can co-occur. + _mirror_shadow(tmp_git_repo, "x.py", base) + + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert f'op = "{bad_op}"' in result.stdout + assert 'only "add" is supported' in result.stdout + + +def _mirror_shadow(repo: Path, source_rel: str, base_sha: str): + """Add a .shadow/<source>.md file as an uncommitted change, so the + `git diff base..HEAD` + `git status` mirror check sees it. + """ + shadow = repo / ".shadow" / (source_rel + ".md") + shadow.parent.mkdir(parents=True, exist_ok=True) + shadow.write_text("## `y`\n\n- mirrored discovery\n") + + +# =========================================================================== +# Discoveries-not-mirrored check (the "LOST at merge time" message — +# discoveries ARE merged via the manifest, but PR reviewers can't see them +# in context without per-file shadow updates). +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_discoveries_not_mirrored_to_shadows_fails(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-nomirror" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest( + dream_id, + discoveries=[{ + "op": "add", + "anchor": "src/auth.py::validate", + "text": "Always returns True on empty input.", + }], + ), + report=_default_report(dream_id, base), + ) + # Note: no .shadow/src/auth.py.md created → mirror check should fail. + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + # The actionable wording mentions PR reviewers / mirror. + assert "NO .shadow/*.md files outside _dreams/" in result.stdout + assert "PR reviewers" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_missing_base_commit_in_report_fails_when_discoveries_present(tmp_git_repo): + _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-nobase" + # Build a report WITHOUT base_commit by leaving it out of frontmatter. + report = ( + "---\n" + f'dream_id: "{dream_id}"\n' + "category: bug hunting\n" + "verdict: useful\n" + "---\n\n" + f"# {dream_id}\n" + ) + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest( + dream_id, + discoveries=[{ + "op": "add", "anchor": "x.py::y", "text": "claim" + }], + ), + report=report, + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 1 + assert "missing `base_commit`" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_unresolvable_base_commit_warns_not_errors(tmp_git_repo): + """A base_commit that git can't resolve should downgrade the mirror + check to a warning (return 0), not a misleading hard 'no shadows' error.""" + _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-badbase" + bogus = "deadbeefdeadbeefdeadbeefdeadbeefdeadbeef" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest( + dream_id, + discoveries=[{ + "op": "add", "anchor": "x.py::y", "text": "claim" + }], + ), + report=_default_report(dream_id, bogus), + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 0, result.stdout + result.stderr + assert "could not resolve base_commit" in result.stdout + # The misleading hard-error wording must NOT appear. + assert "NO .shadow/*.md files outside _dreams/" not in result.stdout + + +# =========================================================================== +# Label-triage WARNINGS (non-blocking) +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_label_triage_emits_warning_but_does_not_fail(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-warn" + # Trigger the 'bug' label signal: phrase "silently fails" matches. + # Provide labels: [] so the warning fires. + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest( + dream_id, + discoveries=[{ + "op": "add", + "anchor": "src/foo.py::bar", + "text": "The cache silently fails on expired tokens.", + "labels": [], + }], + ), + report=_default_report(dream_id, base), + ) + _mirror_shadow(tmp_git_repo, "src/foo.py", base) + + result = _run_validate(dream_id, tmp_git_repo) + # Happy: succeeds (warning is non-blocking). + assert result.returncode == 0, result.stdout + result.stderr + assert "WARNING" in result.stdout + assert "'bug' signal" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_label_triage_silent_when_label_already_set(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-labelled" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest( + dream_id, + discoveries=[{ + "op": "add", + "anchor": "src/foo.py::bar", + "text": "The cache silently fails on expired tokens.", + "labels": ["bug"], + }], + ), + report=_default_report(dream_id, base), + ) + _mirror_shadow(tmp_git_repo, "src/foo.py", base) + + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 0, result.stdout + result.stderr + assert "'bug' signal" not in result.stdout + + +# =========================================================================== +# Happy path +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_happy_path_passes(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-happy" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest( + dream_id, + discoveries=[{ + "op": "add", + "anchor": "src/foo.py::bar", + "text": "Returns 0 for empty list.", + "labels": [], + }], + ), + report=_default_report(dream_id, base), + ) + _mirror_shadow(tmp_git_repo, "src/foo.py", base) + + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 0, result.stdout + result.stderr + assert "Validation passed" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_happy_path_with_no_discoveries_skips_mirror_check(tmp_git_repo): + """A manifest with zero discoveries doesn't need to mirror any shadow.""" + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-empty" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest(dream_id, discoveries=[]), + report=_default_report(dream_id, base), + ) + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 0, result.stdout + result.stderr + + +# =========================================================================== +# B1 regression — bare-string discoveries must not crash validation. +# The reconciler normalizes `"some text"` to `{"text": "..."}`; validate +# must accept exactly what the reconciler does instead of raising on str. +# =========================================================================== + +@pytest.mark.slow +@pytest.mark.integration +def test_bare_string_discovery_is_accepted(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-strdisc" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest( + dream_id, + discoveries=["Silently returns None on expired tokens."], + ), + report=_default_report(dream_id, base), + ) + _mirror_shadow(tmp_git_repo, "src/foo.py", base) + + result = _run_validate(dream_id, tmp_git_repo) + assert result.returncode == 0, result.stdout + result.stderr + # Must not blow up with an attribute/type error on the str entry. + assert "Traceback" not in result.stderr + assert "Validation passed" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_non_dict_non_str_discovery_is_reported_not_crashed(tmp_git_repo): + base = _commit_base(tmp_git_repo) + dream_id = "20260101-000000Z-baddisc" + _write_dream( + tmp_git_repo, dream_id, + manifest=_default_manifest( + dream_id, + discoveries=[12345], + ), + report=_default_report(dream_id, base), + ) + _mirror_shadow(tmp_git_repo, "src/foo.py", base) + + result = _run_validate(dream_id, tmp_git_repo) + assert "Traceback" not in result.stderr + assert result.returncode == 1 + assert "must be a string or object" in result.stdout diff --git a/tests/skills/shadow_frog_dream/test_worktree_safety.py b/tests/skills/shadow_frog_dream/test_worktree_safety.py new file mode 100644 index 0000000..f621027 --- /dev/null +++ b/tests/skills/shadow_frog_dream/test_worktree_safety.py @@ -0,0 +1,305 @@ +"""Tests for skills/shadow-frog-dream/_worktree_safety.py. + +The safety module is the *only* line of defense between a misconfigured +`$DREAM_WORKTREE_BASE` (or a buggy slug-derivation in the reconciler) and +`rm -rf`ing something we shouldn't. These tests are deliberately paranoid: +every adversarial input that could escape the gate gets its own assertion. +""" +import os +import sys +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent +SKILL_DIR = REPO_ROOT / "skills" / "shadow-frog-dream" + +sys.path.insert(0, str(SKILL_DIR)) +from _worktree_safety import ( # noqa: E402 + UnsafePath, _FORBIDDEN_BASES, _strip_macos_private, safe_worktree_path +) +sys.path.pop(0) + + +# =========================================================================== +# Happy path — legitimate dream worktrees pass +# =========================================================================== + +class TestHappyPath: + def test_canonical_shape_under_default_base(self): + p = safe_worktree_path( + "/tmp/shadowfrog-dreams/proj/dream-foo", + "/tmp/shadowfrog-dreams", + ) + # On macOS the path may resolve through /private — both are fine. + assert str(p).endswith("/proj/dream-foo") + + def test_custom_base_under_tmpdir(self, tmp_path): + base = tmp_path / "my-dreams" + target = base / "myproject" / "dream-t01-fuzzer" + target.mkdir(parents=True) + p = safe_worktree_path(str(target), str(base)) + assert p.exists() + + def test_slug_with_dots_and_underscores(self, tmp_path): + # SAFE_RE allows [A-Za-z0-9._-], so v1.2.3 and snake_case are fine. + base = tmp_path / "b" + target = base / "ns_one" / "dream-v1.2.3" + target.mkdir(parents=True) + assert safe_worktree_path(str(target), str(base)) == target.resolve() + + +# =========================================================================== +# Rule 1: non-empty inputs +# =========================================================================== + +class TestEmptyInputs: + @pytest.mark.parametrize("path", ["", " ", "\t\n"]) + def test_rejects_empty_path(self, path): + with pytest.raises(UnsafePath, match="empty"): + safe_worktree_path(path, "/tmp/shadowfrog-dreams") + + @pytest.mark.parametrize("base", ["", " ", "\t\n"]) + def test_rejects_empty_base(self, base): + with pytest.raises(UnsafePath, match="empty"): + safe_worktree_path("/tmp/x/proj/dream-foo", base) + + @pytest.mark.parametrize("path", [None, 42, ["/tmp/x"]]) + def test_rejects_non_string_path(self, path): + with pytest.raises(UnsafePath): + safe_worktree_path(path, "/tmp/shadowfrog-dreams") + + +# =========================================================================== +# Rule 2: absolute paths only +# =========================================================================== + +class TestRelativePaths: + @pytest.mark.parametrize("path", [ + "tmp/shadowfrog-dreams/proj/dream-foo", + "./proj/dream-foo", + "../escape", + "proj/dream-foo", + ]) + def test_rejects_relative_path(self, path): + with pytest.raises(UnsafePath, match="not absolute"): + safe_worktree_path(path, "/tmp/shadowfrog-dreams") + + def test_rejects_relative_base(self): + with pytest.raises(UnsafePath, match="not absolute"): + safe_worktree_path( + "/tmp/shadowfrog-dreams/proj/dream-foo", + "tmp/shadowfrog-dreams", + ) + + +# =========================================================================== +# Rule 3: no ".." traversal in literal input +# =========================================================================== + +class TestTraversal: + @pytest.mark.parametrize("path", [ + "/tmp/shadowfrog-dreams/../etc/passwd", + "/tmp/shadowfrog-dreams/proj/dream-foo/../../../etc", + "/tmp/shadowfrog-dreams/../shadowfrog-dreams/proj/dream-foo", + ]) + def test_rejects_traversal_in_path(self, path): + with pytest.raises(UnsafePath, match=r"'\.\.'"): + safe_worktree_path(path, "/tmp/shadowfrog-dreams") + + def test_rejects_traversal_in_base(self): + with pytest.raises(UnsafePath, match=r"'\.\.'"): + safe_worktree_path( + "/tmp/shadowfrog-dreams/proj/dream-foo", + "/tmp/shadowfrog-dreams/../shadowfrog-dreams", + ) + + +# =========================================================================== +# Rule 4: sensitive bases — refused regardless of path-shape validity +# =========================================================================== + +class TestSensitiveBases: + @pytest.mark.parametrize("base", [ + "/", "/tmp", "/var", "/var/folders", "/var/tmp", "/etc", "/home", + "/Users", "/root", "/bin", "/sbin", "/dev", "/proc", "/sys", + "/usr", "/lib", "/Library", "/System", + # macOS /private prefixed forms — must ALSO refuse. + "/private/tmp", "/private/etc", "/private/var", + ]) + def test_rejects_sensitive_base(self, base): + # Use a path shape that would pass shape-check, so only Rule 4 can fail. + with pytest.raises(UnsafePath, match="sensitive root"): + safe_worktree_path(f"{base}/proj/dream-foo", base) + + def test_rejects_home_as_base(self): + home = os.path.expanduser("~") + if not home or home == "~": + pytest.skip("no HOME set") + with pytest.raises(UnsafePath, match="sensitive root"): + safe_worktree_path(f"{home}/proj/dream-foo", home) + + +# =========================================================================== +# Rule 5: strictly under base (no escape via symlinks or absolute paths) +# =========================================================================== + +class TestUnderBase: + def test_rejects_base_itself(self, tmp_path): + base = tmp_path / "b" + base.mkdir() + with pytest.raises(UnsafePath, match="strictly under base"): + safe_worktree_path(str(base), str(base)) + + def test_rejects_path_above_base(self, tmp_path): + base = tmp_path / "b" + base.mkdir() + # The base is /<tmp>/b; this path is /<tmp>/other + other = tmp_path / "other" / "proj" / "dream-foo" + with pytest.raises(UnsafePath, match="strictly under base"): + safe_worktree_path(str(other), str(base)) + + def test_rejects_symlinked_leaf_escaping_base(self, tmp_path): + # base/ns/dream-evil → tmp_path/escape-target (outside base) + base = tmp_path / "b" + ns = base / "ns" + ns.mkdir(parents=True) + escape = tmp_path / "escape-target" + escape.mkdir() + link = ns / "dream-evil" + link.symlink_to(escape) + with pytest.raises(UnsafePath, match="strictly under base"): + safe_worktree_path(str(link), str(base)) + + def test_rejects_symlinked_parent_escaping_base(self, tmp_path): + # base/escape-ns is a symlink to /etc. base/escape-ns/dream-x must + # be refused even though the LITERAL input looks valid. + base = tmp_path / "b" + base.mkdir() + (base / "escape-ns").symlink_to("/etc") + with pytest.raises(UnsafePath, match="strictly under base"): + safe_worktree_path(str(base / "escape-ns" / "dream-x"), str(base)) + + +# =========================================================================== +# Rule 6: exact `<base>/<ns>/dream-<slug>` shape +# =========================================================================== + +class TestShape: + def test_rejects_too_shallow_one_level(self, tmp_path): + base = tmp_path / "b" + base.mkdir() + with pytest.raises(UnsafePath, match="2 levels under"): + safe_worktree_path(str(base / "dream-foo"), str(base)) + + def test_rejects_too_deep_three_levels(self, tmp_path): + base = tmp_path / "b" + base.mkdir() + with pytest.raises(UnsafePath, match="2 levels under"): + safe_worktree_path( + str(base / "ns" / "dream-foo" / "extra"), str(base), + ) + + def test_rejects_leaf_without_dream_prefix(self, tmp_path): + base = tmp_path / "b" + base.mkdir() + with pytest.raises(UnsafePath, match="dream-"): + safe_worktree_path(str(base / "ns" / "notdream-foo"), str(base)) + + def test_rejects_empty_slug(self, tmp_path): + base = tmp_path / "b" + base.mkdir() + with pytest.raises(UnsafePath, match="slug"): + safe_worktree_path(str(base / "ns" / "dream-"), str(base)) + + @pytest.mark.parametrize("bad_slug", [ + "foo bar", # space + "foo;rm -rf /", # shell metachar + "foo$evil", # shell metachar + "foo/bar", # subdir embedded — would land at depth 3 anyway + "foo\nbar", # newline + ]) + def test_rejects_unsafe_slug(self, tmp_path, bad_slug): + base = tmp_path / "b" + base.mkdir() + with pytest.raises(UnsafePath): + safe_worktree_path( + str(base / "ns" / f"dream-{bad_slug}"), str(base), + ) + + @pytest.mark.parametrize("bad_ns", [ + "n s", "n;s", "n$s", "n\ts", "n/s", + ]) + def test_rejects_unsafe_ns(self, tmp_path, bad_ns): + base = tmp_path / "b" + base.mkdir() + with pytest.raises(UnsafePath): + safe_worktree_path( + str(base / bad_ns / "dream-foo"), str(base), + ) + + +# =========================================================================== +# CLI exit codes (used by dream-cleanup.sh and dream-gc.sh) +# =========================================================================== + +class TestCLI: + def _run(self, *args): + import subprocess + return subprocess.run( + [sys.executable, str(SKILL_DIR / "_worktree_safety.py"), *args], + capture_output=True, text=True, + ) + + def test_exit_0_when_safe_and_exists(self, tmp_path): + base = tmp_path / "b" + target = base / "ns" / "dream-foo" + target.mkdir(parents=True) + r = self._run(str(target), str(base)) + assert r.returncode == 0 + + def test_exit_2_when_safe_and_missing(self, tmp_path): + base = tmp_path / "b" + base.mkdir() + r = self._run(str(base / "ns" / "dream-foo"), str(base)) + assert r.returncode == 2 + + def test_exit_1_when_unsafe(self): + r = self._run("/tmp/proj/dream-foo", "/tmp") + assert r.returncode == 1 + assert "ERROR" in r.stderr + + def test_exit_1_on_usage_error(self): + r = self._run() + assert r.returncode == 1 + + +# =========================================================================== +# Internal helper: _strip_macos_private +# =========================================================================== + +class TestMacosPrivateStripping: + @pytest.mark.parametrize("inp,expected", [ + ("/private/tmp", "/tmp"), + ("/private/var/folders", "/var/folders"), + ("/private/etc", "/etc"), + ("/private", "/private"), # standalone — don't strip + ("/var/private/folders", "/var/private/folders"), # not a prefix + ("/tmp", "/tmp"), # no-op + ]) + def test_strips_only_leading_private_segment(self, inp, expected): + assert _strip_macos_private(inp) == expected + + +# =========================================================================== +# Regression guard: the forbidden-base list must include the macOS /private +# variants for /tmp, /var, /etc — the original implementation missed these. +# =========================================================================== + +class TestForbiddenBaseListInvariants: + def test_tmp_var_etc_are_forbidden(self): + for must_be_listed in ("/tmp", "/var", "/etc", "/home", "/Users"): + assert must_be_listed in _FORBIDDEN_BASES, ( + f"{must_be_listed} must stay in _FORBIDDEN_BASES to keep " + f"the safety gate trustworthy" + ) diff --git a/tests/skills/shadow_frog_init/__init__.py b/tests/skills/shadow_frog_init/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/skills/shadow_frog_init/test_shadow_init.py b/tests/skills/shadow_frog_init/test_shadow_init.py new file mode 100644 index 0000000..ca6bda8 --- /dev/null +++ b/tests/skills/shadow_frog_init/test_shadow_init.py @@ -0,0 +1,1604 @@ +"""Tests for `skills/shadow-frog-init/shadow-init.py`. + +Philosophy: USE REAL FILES, REAL GIT, REAL SUBPROCESSES (per +`minimal-mocking-tests`). The script is loaded as a module via the +`shadow_init` fixture (importlib loader path) for in-process tests, and +exercised end-to-end via subprocess for the CLI argparse dispatcher. + +Categories: + * Pure-function tests (no I/O): Symbol formatting, language detection, + path filters, language extractors. Parametrized aggressively. + * Filesystem tests using `tmp_path` / `tmp_git_repo`: walk, shadowignore + loading, end-to-end file discovery. + * Scaffolding tests: build_state_json (B2 regression), build_shadow_content, + build_index. + * CLI integration tests (marked slow + integration): full subprocess + invocations covering --dry-run, --reset, and the empty-repo B2 case. + +B2 regression: `last_commit` must NEVER be `""`. It should always be either +a 40-char SHA or the sentinel string `"none"` (downstream hooks rely on +that distinction). +""" +import json +import re +import subprocess +import sys +from pathlib import Path + +import pytest + + +REPO_ROOT = Path(__file__).resolve().parents[3] +SCRIPT = REPO_ROOT / "skills/shadow-frog-init/shadow-init.py" + + +def _to_tuples(symbols): + """Flatten a list of Symbol objects to (name, kind, parent) tuples.""" + return [(s.name, s.kind, s.parent) for s in symbols] + + +def _run_shadow_init(repo_root, *args, timeout=60): + """Invoke shadow-init.py as a subprocess, returning the CompletedProcess.""" + return subprocess.run( + [sys.executable, str(SCRIPT), "--root", str(repo_root), *args], + capture_output=True, text=True, timeout=timeout, cwd=str(repo_root), + ) + + +# --------------------------------------------------------------------------- +# Symbol class +# --------------------------------------------------------------------------- + +def test_symbol_display_name_top_level(shadow_init): + s = shadow_init.Symbol("foo", "function") + assert s.display_name == "foo" + assert s.heading_text == "foo" + assert s.is_container is False + + +def test_symbol_display_name_method_has_parent_dot(shadow_init): + s = shadow_init.Symbol("validate", "method", parent="UserAuth") + assert s.display_name == "UserAuth.validate" + assert s.heading_text == "UserAuth.validate" + + +@pytest.mark.parametrize("kind,name,expected_heading,container", [ + ("class", "Foo", "class Foo", True), + ("interface", "IBar", "interface IBar", True), + ("enum", "Color", "enum Color", True), + ("trait", "Greet", "trait Greet", True), + ("struct", "Point", "struct Point", True), + ("protocol", "Encodable", "protocol Encodable", True), + ("module", "Utils", "module Utils", True), + ("function", "main", "main", False), +]) +def test_symbol_heading_and_container_per_kind(shadow_init, kind, name, expected_heading, container): + s = shadow_init.Symbol(name, kind) + assert s.heading_text == expected_heading + assert s.is_container is container + + +# --------------------------------------------------------------------------- +# detect_language +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize("rel_path,expected", [ + ("foo.py", "Python"), + ("src/main.py", "Python"), + ("foo.js", "JavaScript"), + ("foo.jsx", "JavaScript"), + ("foo.mjs", "JavaScript"), + ("foo.cjs", "JavaScript"), + ("foo.ts", "TypeScript"), + ("foo.tsx", "TypeScript"), + ("Foo.java", "Java"), + ("Foo.kt", "Kotlin"), + ("Foo.kts", "Kotlin"), + ("Foo.scala", "Scala"), + ("foo.go", "Go"), + ("foo.rs", "Rust"), + ("foo.rb", "Ruby"), + ("foo.c", "C"), + ("foo.h", "C"), + ("foo.cpp", "C++"), + ("foo.hpp", "C++"), + ("foo.cs", "C#"), + ("foo.php", "PHP"), + ("foo.sh", "Shell"), + ("foo.bash", "Shell"), + ("foo.zsh", "Shell"), + ("foo.swift", "Swift"), + ("foo.yaml", "YAML"), + ("foo.yml", "YAML"), + ("foo.toml", "TOML"), + ("foo.json", "JSON"), +]) +def test_detect_language_by_extension(shadow_init, rel_path, expected): + assert shadow_init.detect_language(rel_path) == expected + + +@pytest.mark.parametrize("basename,expected", [ + ("Makefile", "Makefile"), + ("Dockerfile", "Dockerfile"), + ("Containerfile", "Dockerfile"), + ("Rakefile", "Ruby"), + ("Gemfile", "Ruby"), +]) +def test_detect_language_by_basename(shadow_init, basename, expected): + # Basename works both bare and in a subdirectory + assert shadow_init.detect_language(basename) == expected + assert shadow_init.detect_language(f"path/to/{basename}") == expected + + +@pytest.mark.parametrize("rel_path", [ + "foo.xyz", + "README", + "weird.exe", + "no_extension_file", +]) +def test_detect_language_unknown_returns_unknown(shadow_init, rel_path): + assert shadow_init.detect_language(rel_path) == "Unknown" + + +# --------------------------------------------------------------------------- +# _is_excluded_path +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize("path,is_excluded", [ + # Excluded directory components + ("node_modules/foo/bar.js", True), + ("foo/node_modules/bar.js", True), + ("vendor/lib.js", True), + ("__pycache__/foo.pyc", True), + ("dist/bundle.js", True), + ("build/output.txt", True), + ("target/release/app", True), + ("out/main.js", True), + (".venv/lib/site-packages/foo.py", True), + ("venv/bin/python", True), + (".shadow/foo.md", True), + # Excluded suffixes + ("foo.min.js", True), + ("path/foo.min.css", True), + ("foo.map", True), + ("Pipfile.lock", True), + # ShadowFrog's own install artifacts (project install copies these in) + (".github/skills/shadow-frog/SKILL.md", True), + (".github/skills/shadow-frog-init/shadow-init.py", True), + (".github/hooks/scripts/shadow-frog-pre-tool.sh", True), + (".claude/skills/shadow-frog/SKILL.md", True), + (".claude/skills/shadow-frog-viewer/shadow-viewer.py", True), + (".claude/hooks/scripts/shadow-frog-check-init.sh", True), + # NOT excluded + ("src/main.py", False), + ("foo/bar.js", False), + ("Makefile", False), + ("package.json", False), # ".lock" suffix, not "lock" — package.json should pass + (".github/workflows/ci.yml", False), # user's own .github content is shadowed + (".github/skills/my-other-skill/SKILL.md", False), # only shadow-frog* is excluded + (".claude/skills/my-other-skill/SKILL.md", False), # only shadow-frog* is excluded +]) +def test_is_excluded_path(shadow_init, path, is_excluded): + assert shadow_init._is_excluded_path(path) is is_excluded + + +# --------------------------------------------------------------------------- +# _is_source_file +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize("path,expected", [ + ("foo.py", True), + ("foo.PY", True), # case insensitive + ("foo.js", True), + ("foo.ts", True), + ("Makefile", True), + ("path/to/Dockerfile", True), + ("path/to/Rakefile", True), + # Not source + ("foo.txt", False), + ("foo.png", False), + ("foo.exe", False), + ("README", False), + ("notes.md", False), +]) +def test_is_source_file(shadow_init, path, expected): + assert shadow_init._is_source_file(path) is expected + + +# --------------------------------------------------------------------------- +# _walk_files +# --------------------------------------------------------------------------- + +def test_walk_files_empty_repo(shadow_init, tmp_path): + empty = tmp_path / "empty" + empty.mkdir() + assert shadow_init._walk_files(str(empty)) == [] + + +def test_walk_files_one_file(shadow_init, tmp_path): + repo = tmp_path / "one_file" + repo.mkdir() + (repo / "foo.py").write_text("x = 1\n") + assert shadow_init._walk_files(str(repo)) == ["foo.py"] + + +def test_walk_files_skips_excluded_and_dotfile_dirs(shadow_init, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + (repo / "foo.py").write_text("x = 1\n") + (repo / "src").mkdir() + (repo / "src" / "main.py").write_text("y = 2\n") + # Excluded dir + (repo / "node_modules").mkdir() + (repo / "node_modules" / "bundle.js").write_text("z = 3\n") + # Hidden dir (skipped by _walk_files's leading-dot filter) + (repo / ".git").mkdir() + (repo / ".git" / "config").write_text("[core]\n") + + result = sorted(shadow_init._walk_files(str(repo))) + assert "foo.py" in result + assert "src/main.py" in result + assert not any("node_modules" in p for p in result) + assert not any(".git" in p for p in result) + + +# --------------------------------------------------------------------------- +# _load_shadowignore +# --------------------------------------------------------------------------- + +def test_load_shadowignore_no_file(shadow_init, tmp_path): + matcher = shadow_init._load_shadowignore(tmp_path) + assert matcher("anything.py") is False + + +def test_load_shadowignore_empty_file(shadow_init, tmp_path): + (tmp_path / ".shadowignore").write_text("") + matcher = shadow_init._load_shadowignore(tmp_path) + assert matcher("anything.py") is False + + +def test_load_shadowignore_comments_only(shadow_init, tmp_path): + (tmp_path / ".shadowignore").write_text("# header comment\n\n# another\n") + matcher = shadow_init._load_shadowignore(tmp_path) + assert matcher("anything.py") is False + + +def test_load_shadowignore_pattern_matches(shadow_init, tmp_path): + (tmp_path / ".shadowignore").write_text("*.log\nsecret/\n") + matcher = shadow_init._load_shadowignore(tmp_path) + assert matcher("foo.log") is True + assert matcher("nested/path/error.log") is True + assert matcher("foo.py") is False + + +def test_load_shadowignore_negation_does_not_crash(shadow_init, tmp_path): + # Negation is valid gitwildmatch syntax — the matcher must not raise. + (tmp_path / ".shadowignore").write_text("*.log\n!keep.log\n") + matcher = shadow_init._load_shadowignore(tmp_path) + # Both calls must succeed (we don't assert specific values because + # pathspec respects negation but the fnmatch fallback does not). + matcher("foo.log") + matcher("keep.log") + + +# --------------------------------------------------------------------------- +# discover_files (end-to-end against tmp_git_repo) +# --------------------------------------------------------------------------- + +@pytest.mark.slow +def test_discover_files_end_to_end(shadow_init, tmp_git_repo): + (tmp_git_repo / "foo.py").write_text("def hello(): pass\n") + (tmp_git_repo / "bar.js").write_text("function world() {}\n") + (tmp_git_repo / "ignore.txt").write_text("not source\n") + sub = tmp_git_repo / "src" + sub.mkdir() + (sub / "baz.py").write_text("x = 1\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "add files"], + cwd=tmp_git_repo, check=True) + + shadow_dir = tmp_git_repo / ".shadow" # may or may not exist; fine either way + result = shadow_init.discover_files(str(tmp_git_repo), shadow_dir) + + assert "foo.py" in result + assert "bar.js" in result + assert "src/baz.py" in result + assert "ignore.txt" not in result # not a recognized source ext + # discover_files returns a sorted, deduped list + assert result == sorted(set(result)) + + +@pytest.mark.slow +def test_discover_files_honors_shadowignore(shadow_init, tmp_git_repo): + (tmp_git_repo / "foo.py").write_text("def a(): pass\n") + (tmp_git_repo / "bar.py").write_text("def b(): pass\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "init"], + cwd=tmp_git_repo, check=True) + + shadow_dir = tmp_git_repo / ".shadow" + shadow_dir.mkdir() + (shadow_dir / ".shadowignore").write_text("bar.py\n") + + result = shadow_init.discover_files(str(tmp_git_repo), shadow_dir) + assert "foo.py" in result + assert "bar.py" not in result + + +# --------------------------------------------------------------------------- +# extract_symbols dispatcher +# --------------------------------------------------------------------------- + +def test_extract_symbols_dispatcher_python(shadow_init): + syms = shadow_init.extract_symbols("def foo(): pass\n", "Python", "foo.py") + assert len(syms) == 1 + assert syms[0].name == "foo" + assert syms[0].kind == "function" + + +def test_extract_symbols_dispatcher_javascript(shadow_init): + syms = shadow_init.extract_symbols("function foo() {}\n", "JavaScript", "foo.js") + assert len(syms) == 1 + assert syms[0].name == "foo" + + +def test_extract_symbols_dispatcher_typescript_uses_js_extractor(shadow_init): + # TypeScript shares the JS extractor via the EXTRACTORS dispatch table. + syms = shadow_init.extract_symbols( + "export function bar(): number { return 1; }\n", + "TypeScript", "bar.ts", + ) + assert any(s.name == "bar" for s in syms) + + +@pytest.mark.parametrize("unknown_lang", ["Unknown", "YAML", "JSON", "TOML", "Makefile", "Dockerfile"]) +def test_extract_symbols_unknown_language_returns_empty(shadow_init, unknown_lang): + """Languages without a registered extractor return [] (not None, not error).""" + assert shadow_init.extract_symbols("anything", unknown_lang, "x") == [] + + +def test_extract_symbols_never_raises_on_garbage(shadow_init): + """Per docstring: never raises — returns [] on any failure.""" + result = shadow_init.extract_symbols( + "\x00\x01\x02 ~`!@#$%^&*()", "Python", "weird.py", + ) + assert isinstance(result, list) + + +# --------------------------------------------------------------------------- +# Per-language extractors +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize("source,expected", [ + # simple top-level function + ("def foo():\n pass\n", + [("foo", "function", None)]), + # async function + ("async def fetch():\n pass\n", + [("fetch", "function", None)]), + # class with method + ("class Foo:\n def bar(self):\n pass\n", + [("Foo", "class", None), ("bar", "method", "Foo")]), + # function with default args + ("def add(x=1, y=2):\n return x + y\n", + [("add", "function", None)]), + # decorator does not interfere + ("@staticmethod\ndef foo():\n pass\n", + [("foo", "function", None)]), + # comment-only / empty + ("# just comments\n# nothing here\n", []), + ("", []), + # module-level ALL_CAPS constant + ("MAX_RETRIES = 3\n", + [("MAX_RETRIES", "constant", None)]), + # annotated module-level constant + ("TIMEOUT: int = 30\n", + [("TIMEOUT", "constant", None)]), + # lowercase module var is NOT captured + ("config = {}\n", []), + # comparison / augmented assignment are NOT captured + ("COUNT += 1\n", []), + # constant alongside a function + ("CACHE = {}\ndef get(code):\n return CACHE.get(code)\n", + [("CACHE", "constant", None), ("get", "function", None)]), +]) +def test_extract_python(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_python(source)) == expected + + +def test_extract_python_class_scoped_constant_not_captured(shadow_init): + """ALL_CAPS names inside a class body are class attributes, not + module-level constants, so they are NOT emitted as constant symbols.""" + source = ( + "class Foo:\n" + " CLASS_CONST = 5\n" + " def bar(self):\n" + " LOCAL = 2\n" + " return LOCAL\n" + "GLOBAL_AFTER = {}\n" + ) + syms = _to_tuples(shadow_init._extract_python(source)) + assert ("CLASS_CONST", "constant", None) not in syms + assert ("Foo", "class", None) in syms + assert ("bar", "method", "Foo") in syms + # A module-level constant after the class IS captured (scope reset). + assert ("GLOBAL_AFTER", "constant", None) in syms + + +@pytest.mark.parametrize("source,expected", [ + ("function foo() {}\n", + [("foo", "function", None)]), + ("export function bar() {}\n", + [("bar", "function", None)]), + ("async function fetchData() {}\n", + [("fetchData", "function", None)]), + # const arrow function + ("const baz = () => { return 1; };\n", + [("baz", "function", None)]), + # class with method + ("class Foo {\n bar() { return 1; }\n}\n", + [("Foo", "class", None), ("bar", "method", "Foo")]), + # comment-only + ("// just a comment\n", []), +]) +def test_extract_javascript(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_javascript(source)) == expected + + +@pytest.mark.parametrize("source,expected", [ + ("public class Foo {\n public void bar() {}\n}\n", + [("Foo", "class", None), ("bar", "method", "Foo")]), + ("public interface IBar {\n}\n", + [("IBar", "interface", None)]), + ("enum Color {\n RED, BLUE\n}\n", + [("Color", "enum", None)]), + # Kotlin `fun` inside class + ("class Greeter {\n fun hello() {}\n}\n", + [("Greeter", "class", None), ("hello", "method", "Greeter")]), + # comment-only + ("// nothing\n", []), +]) +def test_extract_java_like(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_java_like(source)) == expected + + +@pytest.mark.parametrize("source,expected", [ + ("func Foo() {}\n", + [("Foo", "function", None)]), + ("type Server struct {\n addr string\n}\n", + [("Server", "struct", None)]), + ("type Handler interface {\n Handle()\n}\n", + [("Handler", "interface", None)]), + # method on pointer receiver + ("func (s *Server) Start() {}\n", + [("Start", "method", "Server")]), + # method on value receiver + ("func (s Server) Stop() {}\n", + [("Stop", "method", "Server")]), + # package + comment only + ("package main\n// nothing\n", []), +]) +def test_extract_go(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_go(source)) == expected + + +@pytest.mark.parametrize("source,expected", [ + ("fn main() {}\n", + [("main", "function", None)]), + ("pub fn add(x: i32, y: i32) -> i32 { x + y }\n", + [("add", "function", None)]), + ("struct Point {\n x: f64,\n y: f64,\n}\n", + [("Point", "struct", None)]), + ("enum Color { Red, Blue }\n", + [("Color", "enum", None)]), + # impl block: fn inside an impl becomes a method + ("struct Foo;\nimpl Foo {\n fn new() -> Self { Foo }\n}\n", + [("Foo", "struct", None), ("new", "method", "Foo")]), + # comment-only + ("// just a comment\n", []), +]) +def test_extract_rust(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_rust(source)) == expected + + +@pytest.mark.parametrize("source,expected", [ + # top-level def + ("def hello\n puts 'hi'\nend\n", + [("hello", "function", None)]), + # class with def -> method + ("class Foo\n def bar\n 1\n end\nend\n", + [("Foo", "class", None), ("bar", "method", "Foo")]), + # module: methods inside are NOT scoped to a class (Ruby extractor only + # sets current_class for `class`, not `module`) + ("module M\n def fn\n 1\n end\nend\n", + [("M", "module", None), ("fn", "function", None)]), + # comment-only + ("# nothing\n", []), +]) +def test_extract_ruby(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_ruby(source)) == expected + + +@pytest.mark.parametrize("source,expected", [ + # plain function + ("int add(int a, int b) {\n return a + b;\n}\n", + [("add", "function", None)]), + # class with declared method body + ("class Foo {\npublic:\n void bar();\n};\n", + [("Foo", "class", None), ("bar", "method", "Foo")]), + # preprocessor lines (start with #) are skipped + ("#include <stdio.h>\nint main() { return 0; }\n", + [("main", "function", None)]), + # commented-out class is skipped; real fn extracted + ("// fake class Foo {};\nint real() { return 0; }\n", + [("real", "function", None)]), +]) +def test_extract_c_cpp(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_c_cpp(source)) == expected + + +@pytest.mark.parametrize("source,expected", [ + ("<?php\nfunction foo() {}\n", + [("foo", "function", None)]), + ("<?php\nclass Foo {\n public function bar() {}\n}\n", + [("Foo", "class", None), ("bar", "method", "Foo")]), + ("<?php\nabstract class Base {\n abstract function init();\n}\n", + [("Base", "class", None), ("init", "method", "Base")]), + ("<?php\n// just a comment\n", []), +]) +def test_extract_php(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_php(source)) == expected + + +@pytest.mark.parametrize("source,expected", [ + # `function NAME {` form + ("function foo {\n echo hi\n}\n", + [("foo", "function", None)]), + # `NAME() {` form + ("bar() {\n echo bar\n}\n", + [("bar", "function", None)]), + # both forms in one file, shebang ignored + ("#!/bin/bash\nfunction a {\n :\n}\nb() {\n :\n}\n", + [("a", "function", None), ("b", "function", None)]), + # comment-only + ("# comment only\n", []), +]) +def test_extract_shell(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_shell(source)) == expected + + +@pytest.mark.parametrize("source,expected", [ + # top-level func + ("func hello() -> String {\n return \"hi\"\n}\n", + [("hello", "function", None)]), + # class with method + ("class Greeter {\n func hi() {}\n}\n", + [("Greeter", "class", None), ("hi", "method", "Greeter")]), + # protocol with method + ("protocol Greeter {\n func hi()\n}\n", + [("Greeter", "protocol", None), ("hi", "method", "Greeter")]), + # struct (no methods) + ("struct Point {\n var x: Double\n}\n", + [("Point", "struct", None)]), + # comment-only + ("// nothing\n", []), +]) +def test_extract_swift(shadow_init, source, expected): + assert _to_tuples(shadow_init._extract_swift(source)) == expected + + +# --------------------------------------------------------------------------- +# build_state_json — B2 REGRESSION +# --------------------------------------------------------------------------- + +def test_build_state_json_empty_string_becomes_none_sentinel(shadow_init): + """B2 regression: empty string last_commit MUST become 'none', not ''.""" + state = shadow_init.build_state_json(0, 0, "") + assert state["last_commit"] == "none", ( + f"B2 regression: expected sentinel 'none', got {state['last_commit']!r}. " + "Downstream hooks rely on 'none' to distinguish missing-commit from empty." + ) + + +def test_build_state_json_none_becomes_none_sentinel(shadow_init): + """B2 regression: None last_commit MUST become 'none'.""" + state = shadow_init.build_state_json(0, 0, None) + assert state["last_commit"] == "none" + + +def test_build_state_json_valid_sha_passes_through(shadow_init): + """B2 regression: a real 40-char SHA must NOT be replaced by the sentinel.""" + sha = "abcd" * 10 # exactly 40 hex chars + assert len(sha) == 40, "test setup: SHA must be 40 chars" + state = shadow_init.build_state_json(0, 0, sha) + assert state["last_commit"] == sha + + +def test_build_state_json_required_fields(shadow_init): + state = shadow_init.build_state_json(5, 17, "deadbeef" * 5) + assert state["version"] == 1 + assert state["last_update_type"] == "init" + assert state["total_files"] == 5 + assert state["total_symbols"] == 17 + assert state["total_discoveries"] == 0 + assert state["dream_cycles_completed"] == 0 + # ISO-ish timestamp (or "unknown" if the clock call failed — unlikely) + assert "T" in state["initialized_at"] or state["initialized_at"] == "unknown" + assert state["initialized_at"] == state["last_update_at"] + + +# --------------------------------------------------------------------------- +# build_shadow_content +# --------------------------------------------------------------------------- + +def test_build_shadow_content_no_symbols(shadow_init): + content = shadow_init.build_shadow_content( + "src/foo.py", "Python", 42, "2024-01-15", [], + ) + assert content.startswith("# Shadow: src/foo.py") + assert "**Language**: Python" in content + assert "**Lines**: 42" in content + assert "**Last modified**: 2024-01-15" in content + assert "## File-Level" in content + assert "## Cross-References" in content + assert "_No discoveries yet._" in content + assert "_No cross-cutting discoveries yet._" in content + + +def test_build_shadow_content_with_class_and_method_and_function(shadow_init): + Symbol = shadow_init.Symbol + syms = [ + Symbol("Foo", "class"), + Symbol("bar", "method", parent="Foo"), + Symbol("standalone", "function"), + ] + content = shadow_init.build_shadow_content( + "foo.py", "Python", 10, "2024-01-15", syms, + ) + # Class container heading with prefix and child as ### + assert "## `class Foo`" in content + assert "### `Foo.bar`" in content + # Standalone function as ## + assert "## `standalone`" in content + # Cross-References is the last section + assert content.rfind("## Cross-References") > content.rfind("## `standalone`") + + +def test_build_shadow_content_empty_container_gets_placeholder(shadow_init): + Symbol = shadow_init.Symbol + content = shadow_init.build_shadow_content( + "foo.java", "Java", 5, "2024-01-15", + [Symbol("Empty", "class")], + ) + assert "## `class Empty`" in content + # The class with no children still gets a placeholder + placeholder_after_heading = content.split("## `class Empty`", 1)[1] + assert "_No discoveries yet._" in placeholder_after_heading + + +# --------------------------------------------------------------------------- +# build_index +# --------------------------------------------------------------------------- + +def test_build_index_empty(shadow_init): + content = shadow_init.build_index([], 0) + assert content.startswith("# Shadow Index") + assert "Total files: 0" in content + assert "Symbols: 0" in content + assert "| File | Language | Symbols | Discoveries |" in content + assert "|------|----------|---------|-------------|" in content + + +def test_build_index_with_records(shadow_init): + Symbol = shadow_init.Symbol + records = [ + ("src/foo.py", "Python", [Symbol("hello", "function"), Symbol("Foo", "class")]), + ("src/bar.js", "JavaScript", []), + ] + content = shadow_init.build_index(records, 2) + assert "Total files: 2" in content + assert "src/foo.py" in content + assert "src/bar.js" in content + assert "Python" in content + assert "JavaScript" in content + # Top-level symbol names should appear in the symbols column + assert "hello" in content + assert "Foo" in content + # Empty record renders symbol count of 0 + assert "| src/bar.js | JavaScript | 0 |" in content + + +def test_build_index_truncates_long_symbol_lists(shadow_init): + Symbol = shadow_init.Symbol + syms = [Symbol(f"f{i}", "function") for i in range(5)] + records = [("foo.py", "Python", syms)] + content = shadow_init.build_index(records, 5) + assert "..." in content # truncated tail + assert "5 (" in content # explicit count + + +# --------------------------------------------------------------------------- +# CLI integration (subprocess path) +# --------------------------------------------------------------------------- + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_dry_run_does_not_write_shadow(tmp_git_repo): + """--dry-run must report what *would* happen but never touch the FS.""" + result = _run_shadow_init(tmp_git_repo, "--reset", "--dry-run") + assert result.returncode == 0, ( + f"stderr: {result.stderr}\nstdout: {result.stdout}" + ) + assert not (tmp_git_repo / ".shadow").exists() + combined = (result.stdout + result.stderr).lower() + assert "dry-run" in combined + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_creates_shadow_for_committed_file(tmp_git_repo): + """End-to-end: a committed `foo.py` should yield a complete .shadow/ tree.""" + (tmp_git_repo / "foo.py").write_text( + "def hello():\n return 'world'\n\nclass Greeter:\n def hi(self):\n return 'hi'\n" + ) + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "add foo"], + cwd=tmp_git_repo, check=True) + + result = _run_shadow_init(tmp_git_repo, "--reset") + assert result.returncode == 0, ( + f"stderr: {result.stderr}\nstdout: {result.stdout}" + ) + + shadow = tmp_git_repo / ".shadow" + assert shadow.is_dir() + assert (shadow / "foo.py.md").is_file() + assert (shadow / "_index.md").is_file() + assert (shadow / "_prefs.md").is_file() + assert (shadow / ".shadowignore").is_file() + assert (shadow / "_meta" / "state.json").is_file() + assert (shadow / "_cross").is_dir() + assert (shadow / "_dreams").is_dir() + + state = json.loads((shadow / "_meta" / "state.json").read_text()) + assert isinstance(state["last_commit"], str) + assert re.fullmatch(r"[0-9a-f]{40}", state["last_commit"]), ( + f"Expected 40-char hex SHA, got: {state['last_commit']!r}" + ) + assert state["total_files"] == 1 + assert state["total_symbols"] >= 2 # hello + Greeter (at least) + assert state["last_update_type"] == "init" + assert state["version"] == 1 + + shadow_content = (shadow / "foo.py.md").read_text() + assert "hello" in shadow_content + assert "Greeter" in shadow_content + assert "Python" in shadow_content + assert "## Cross-References" in shadow_content + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_b2_no_commits_uses_none_sentinel(tmp_git_repo): + """B2 regression: an empty repo (no HEAD) must write last_commit='none'. + + git rev-parse HEAD fails in a fresh repo with no commits — the script + must fall back to the 'none' sentinel rather than writing an empty string. + """ + result = _run_shadow_init(tmp_git_repo, "--reset") + assert result.returncode == 0, ( + f"stderr: {result.stderr}\nstdout: {result.stdout}" + ) + + state_path = tmp_git_repo / ".shadow" / "_meta" / "state.json" + state = json.loads(state_path.read_text()) + assert state["last_commit"] == "none", ( + f"B2 regression: expected sentinel 'none', got {state['last_commit']!r}. " + "An empty string here breaks downstream hooks that distinguish " + "missing-commit from empty-commit." + ) + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_refuses_to_overwrite_without_reset(tmp_git_repo): + """If .shadow/ already exists, the script must refuse without --reset.""" + (tmp_git_repo / ".shadow").mkdir() + (tmp_git_repo / ".shadow" / "marker").write_text("preserve me\n") + + result = _run_shadow_init(tmp_git_repo) # no --reset + assert result.returncode == 1 + # The pre-existing marker must NOT have been deleted + assert (tmp_git_repo / ".shadow" / "marker").is_file() + assert "already exists" in result.stderr.lower() + + +# =========================================================================== +# APPENDED TESTS — main() CLI dispatcher, init_shadow integration, +# remaining extractor edge branches, and helper-function direct coverage. +# +# These tests target the uncovered-line ranges in the shadow-init coverage +# report so the script reaches ~75% line coverage. They follow the same +# zero-mock philosophy as the tests above: real subprocess, real git, +# real filesystems. monkeypatch is used only for sys.argv / cwd plumbing +# required to exercise main() in-process (so coverage is recorded). +# =========================================================================== + +import os + + +@pytest.fixture +def reset_diagnostics(shadow_init): + """Snapshot/restore the module-level warning/error counters. + + `shadow_init` is session-scoped so counters accumulate across tests. + Tests that assert on counter contents should request this fixture + so they get a clean slate and don't perturb later tests. + """ + saved_w = shadow_init._warning_count + saved_e = shadow_init._error_count + saved_d = list(shadow_init._diagnostics) + shadow_init._warning_count = 0 + shadow_init._error_count = 0 + shadow_init._diagnostics.clear() + try: + yield shadow_init + finally: + shadow_init._warning_count = saved_w + shadow_init._error_count = saved_e + shadow_init._diagnostics[:] = saved_d + + +# --------------------------------------------------------------------------- +# warn() / error() — direct invocation (lines 41-43, 49-51) +# --------------------------------------------------------------------------- + +def test_warn_increments_counter_and_logs(reset_diagnostics, capsys): + si = reset_diagnostics + si.warn("synthetic warning") + assert si._warning_count == 1 + assert ("warning", "synthetic warning") in si._diagnostics + err = capsys.readouterr().err + assert "[shadow-init warning] synthetic warning" in err + + +def test_error_increments_counter_and_logs(reset_diagnostics, capsys): + si = reset_diagnostics + si.error("synthetic error") + assert si._error_count == 1 + assert ("error", "synthetic error") in si._diagnostics + err = capsys.readouterr().err + assert "[shadow-init error] synthetic error" in err + + +# --------------------------------------------------------------------------- +# run_git() failure branches (lines 69-81) +# --------------------------------------------------------------------------- + +def test_run_git_returns_none_outside_repo(shadow_init, tmp_path): + """A failing git command (non-zero exit) yields None + warn.""" + out = shadow_init.run_git(["rev-parse", "HEAD"], cwd=str(tmp_path)) + assert out is None + + +def test_run_git_returns_stdout_on_success(shadow_init, tmp_git_repo): + out = shadow_init.run_git(["rev-parse", "--is-inside-work-tree"], + cwd=str(tmp_git_repo)) + assert out is not None + assert out.strip() == "true" + + +def test_run_git_unknown_subcommand_returns_none(shadow_init, tmp_git_repo): + """`git this-is-not-real` returns non-zero → run_git returns None.""" + out = shadow_init.run_git(["this-is-not-a-real-git-command"], + cwd=str(tmp_git_repo)) + assert out is None + + +# --------------------------------------------------------------------------- +# find_repo_root() (lines 88-128) +# --------------------------------------------------------------------------- + +def test_find_repo_root_returns_root_inside_git(shadow_init, tmp_git_repo, + monkeypatch, capsys): + monkeypatch.chdir(tmp_git_repo) + root = shadow_init.find_repo_root() + assert root is not None + assert Path(root).resolve() == tmp_git_repo.resolve() + # Diagnostic line printed to stderr + assert "Detected" in capsys.readouterr().err + + +def test_find_repo_root_outside_repo_returns_none(shadow_init, tmp_path, + monkeypatch): + monkeypatch.chdir(tmp_path) + assert shadow_init.find_repo_root() is None + + +def test_find_repo_root_inside_subdirectory(shadow_init, tmp_git_repo, + monkeypatch): + """find_repo_root should walk up to repo root from a subdir.""" + sub = tmp_git_repo / "deep" / "nested" + sub.mkdir(parents=True) + monkeypatch.chdir(sub) + root = shadow_init.find_repo_root() + assert root is not None + assert Path(root).resolve() == tmp_git_repo.resolve() + + +# --------------------------------------------------------------------------- +# _walk_files: additional coverage +# --------------------------------------------------------------------------- + +def test_walk_files_lists_all_non_excluded(shadow_init, tmp_path): + (tmp_path / "a.py").write_text("x") + (tmp_path / "sub").mkdir() + (tmp_path / "sub" / "b.txt").write_text("y") + result = shadow_init._walk_files(tmp_path) + # Note: walk does NOT filter by extension — that's discover_files's job + assert "a.py" in result + assert os.path.join("sub", "b.txt") in result + + +# --------------------------------------------------------------------------- +# _load_shadowignore: unreadable file (lines 225-227) +# --------------------------------------------------------------------------- + +@pytest.mark.skipif(hasattr(os, "geteuid") and os.geteuid() == 0, + reason="root can read 0o000 files") +def test_load_shadowignore_unreadable_file_warns(shadow_init, tmp_path, + reset_diagnostics): + ignore = tmp_path / ".shadowignore" + ignore.write_text("*.log\n") + ignore.chmod(0o000) + try: + matcher = reset_diagnostics._load_shadowignore(tmp_path) + # Falls back to a permissive matcher + assert matcher("anything.log") is False + msgs = [m for lvl, m in reset_diagnostics._diagnostics + if lvl == "warning"] + assert any("Could not read .shadowignore" in m for m in msgs) + finally: + ignore.chmod(0o600) + + +# --------------------------------------------------------------------------- +# discover_files: walk fallback in non-git dir (lines 282-284) +# --------------------------------------------------------------------------- + +def test_discover_files_falls_back_to_walk_in_non_git_dir(shadow_init, + tmp_path): + """ls-files fails outside git → walk fallback finds files anyway.""" + (tmp_path / "foo.py").write_text("def x(): pass\n") + (tmp_path / "bar.js").write_text("function y(){}\n") + (tmp_path / "ignored.txt").write_text("not source\n") + result = shadow_init.discover_files(str(tmp_path), tmp_path / ".shadow") + assert "foo.py" in result + assert "bar.js" in result + assert "ignored.txt" not in result # filtered by _is_source_file + + +# --------------------------------------------------------------------------- +# Python extractor edge: top-level def after a class exits the class scope +# (lines 414-415) +# --------------------------------------------------------------------------- + +def test_extract_python_top_level_def_after_class_exits_scope(shadow_init): + source = ( + "class Foo:\n" + " def a(self):\n" + " pass\n" + "\n" + "def b():\n" + " pass\n" + ) + syms = _to_tuples(shadow_init._extract_python(source)) + assert ("Foo", "class", None) in syms + assert ("a", "method", "Foo") in syms + assert ("b", "function", None) in syms + + +def test_extract_python_def_at_class_indent_becomes_function(shadow_init): + """A `def` at the *same* indent as the class is a top-level function + (triggers the secondary scope-exit branch in the def-handler).""" + # Using \t-equivalent: two classes at column 0, with a def at column 0 + # immediately after (no blank line) so the line-after-method handler + # reaches the indent==class_indent comparison on the def path. + source = "class Foo:\n def m(self):\n pass\ndef top():\n pass\n" + syms = _to_tuples(shadow_init._extract_python(source)) + assert ("top", "function", None) in syms + + +# --------------------------------------------------------------------------- +# JavaScript extractor edges: function inside class (470), brace clamp (493) +# --------------------------------------------------------------------------- + +def test_extract_javascript_function_keyword_inside_class_is_method(shadow_init): + """A `function name()` line while we're inside a class body should + attribute to the class (line 470).""" + source = ( + "class C {\n" + "function inner() {}\n" + "}\n" + ) + syms = _to_tuples(shadow_init._extract_javascript(source)) + assert ("C", "class", None) in syms + assert ("inner", "method", "C") in syms + + +def test_extract_javascript_clamps_negative_brace_depth(shadow_init): + """Stray close-braces must not crash; depth should clamp to 0 (line 493).""" + source = "}}\n}}\nfunction good() {}\n" + syms = _to_tuples(shadow_init._extract_javascript(source)) + assert ("good", "function", None) in syms + + +# --------------------------------------------------------------------------- +# Java-like extractor: Kotlin-specific `fun <T>` (lines 553-555) + brace +# clamp (lines 558-559) +# --------------------------------------------------------------------------- + +def test_extract_java_like_kotlin_generic_fun_inside_class(shadow_init): + """`fun <T> name(...)` doesn't match the primary regex (because of + the generic between `fun` and the name) — falls through to the + Kotlin-specific fallback regex.""" + source = ( + "class Box {\n" + " fun <T> tag(item: T): T { return item }\n" + "}\n" + ) + syms = _to_tuples(shadow_init._extract_java_like(source)) + assert ("Box", "class", None) in syms + assert ("tag", "method", "Box") in syms + + +def test_extract_java_like_clamps_negative_brace_depth(shadow_init): + source = "}}}\nclass Z {}\n" + syms = _to_tuples(shadow_init._extract_java_like(source)) + assert ("Z", "class", None) in syms + + +# --------------------------------------------------------------------------- +# Rust / Ruby / C++ / PHP / Swift — brace/depth clamp branches +# --------------------------------------------------------------------------- + +def test_extract_rust_clamps_negative_brace_depth(shadow_init): + source = "}}}\npub fn ok() {}\n" + syms = _to_tuples(shadow_init._extract_rust(source)) + assert ("ok", "function", None) in syms + + +def test_extract_ruby_clamps_negative_depth(shadow_init): + source = "end\nend\ndef ok\nend\n" + syms = _to_tuples(shadow_init._extract_ruby(source)) + assert ("ok", "function", None) in syms + + +def test_extract_c_cpp_scoped_function_attributed_to_scope(shadow_init): + """`Foo::bar()` syntax → method of `Foo` via the scope capture group + (line 740).""" + source = "int Foo::bar() { return 1; }\n" + syms = _to_tuples(shadow_init._extract_c_cpp(source)) + assert ("bar", "method", "Foo") in syms + + +def test_extract_c_cpp_skips_blacklisted_control_keywords(shadow_init): + """Lines like `int if(x)` would otherwise be reported as a function + named `if` — the blacklist branch (lines 736-738) prevents this.""" + source = ( + "int if() { return 1; }\n" + "int real_one() { return 2; }\n" + ) + syms = _to_tuples(shadow_init._extract_c_cpp(source)) + assert not any(name == "if" for name, *_ in syms) + assert ("real_one", "function", None) in syms + + +def test_extract_c_cpp_clamps_negative_brace_depth(shadow_init): + source = "}}}\nint ok() { return 0; }\n" + syms = _to_tuples(shadow_init._extract_c_cpp(source)) + assert ("ok", "function", None) in syms + + +def test_extract_php_clamps_negative_brace_depth(shadow_init): + source = "}}}\n<?php\nfunction ok() {}\n" + syms = _to_tuples(shadow_init._extract_php(source)) + assert any(name == "ok" for name, *_ in syms) + + +def test_extract_swift_clamps_negative_brace_depth(shadow_init): + source = "}}}\nfunc ok() {}\n" + syms = _to_tuples(shadow_init._extract_swift(source)) + assert ("ok", "function", None) in syms + + +# --------------------------------------------------------------------------- +# get_last_modified / count_lines (lines 945-956, 960) +# --------------------------------------------------------------------------- + +def test_get_last_modified_returns_iso_date_for_committed_file(shadow_init, + tmp_git_repo): + (tmp_git_repo / "foo.py").write_text("x\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "m"], + cwd=tmp_git_repo, check=True) + date = shadow_init.get_last_modified("foo.py", str(tmp_git_repo)) + assert re.fullmatch(r"\d{4}-\d{2}-\d{2}", date), \ + f"expected ISO date, got: {date!r}" + + +def test_get_last_modified_returns_unknown_in_non_git_dir(shadow_init, + tmp_path): + (tmp_path / "foo.py").write_text("x\n") + assert shadow_init.get_last_modified("foo.py", str(tmp_path)) == "unknown" + + +def test_get_last_modified_returns_unknown_for_uncommitted(shadow_init, + tmp_git_repo): + (tmp_git_repo / "fresh.py").write_text("x\n") # not committed + assert shadow_init.get_last_modified("fresh.py", + str(tmp_git_repo)) == "unknown" + + +@pytest.mark.parametrize("source,expected", [ + ("", 0), + ("a", 1), + ("a\nb\n", 2), + ("a\nb\nc", 3), + ("\n\n\n", 3), +]) +def test_count_lines(shadow_init, source, expected): + assert shadow_init.count_lines(source) == expected + + +# --------------------------------------------------------------------------- +# build_shadow_content: orphan nested symbol (lines 1009-1012) +# --------------------------------------------------------------------------- + +def test_build_shadow_content_orphan_method_renders_as_h3(shadow_init): + """A symbol with parent set but no preceding container of that name + falls into the orphan branch — rendered as a ### heading using the + `Parent.name` display name.""" + Symbol = shadow_init.Symbol + content = shadow_init.build_shadow_content( + "foo.py", "Python", 1, "2024-01-01", + [Symbol("lost", "method", parent="Ghost")], + ) + assert "### `Ghost.lost`" in content + + +def test_build_shadow_content_orphan_after_real_class(shadow_init): + """Orphan and proper-class scenarios coexist in one file without crashing.""" + Symbol = shadow_init.Symbol + syms = [ + Symbol("Real", "class"), + Symbol("inner", "method", parent="Real"), + Symbol("escaped", "method", parent="Vanished"), # orphan + ] + content = shadow_init.build_shadow_content( + "x.py", "Python", 10, "2024-01-01", syms, + ) + assert "## `class Real`" in content + assert "### `Real.inner`" in content + assert "### `Vanished.escaped`" in content + + +# --------------------------------------------------------------------------- +# build_index: empty-top-level names fallback (1143), malformed row (1150-1152) +# --------------------------------------------------------------------------- + +def test_build_index_no_top_level_falls_back_to_display_names(shadow_init): + """When a file has only nested symbols (no top-level), the index row + shows `Parent.name` display names instead of an empty list.""" + Symbol = shadow_init.Symbol + records = [("foo.py", "Python", + [Symbol("m1", "method", parent="X"), + Symbol("m2", "method", parent="X")])] + content = shadow_init.build_index(records, 2) + assert "X.m1" in content + assert "X.m2" in content + + +def test_build_index_malformed_record_renders_fallback_row(shadow_init, + reset_diagnostics): + """A row where `symbols` is None blows up len()/iteration → caught, + fallback row with `?` in the symbols column is emitted (line 1152).""" + bad_records = [("foo.py", "Python", None)] + content = reset_diagnostics.build_index(bad_records, 0) + assert "| foo.py | Python | ? | 0 |" in content + msgs = [m for lvl, m in reset_diagnostics._diagnostics if lvl == "warning"] + assert any("Failed to build index row" in m for m in msgs) + + +# --------------------------------------------------------------------------- +# init_shadow direct invocation (lines 1168-1400) +# --------------------------------------------------------------------------- + +def test_init_shadow_empty_git_repo_writes_none_sentinel(shadow_init, + tmp_git_repo): + """Empty repo (no commits): last_commit must be the 'none' sentinel + (B2 regression) and all scaffold artifacts must exist.""" + ok = shadow_init.init_shadow(str(tmp_git_repo)) + assert ok is True + shadow = tmp_git_repo / ".shadow" + assert (shadow / "_meta" / "state.json").is_file() + assert (shadow / "_cross").is_dir() + assert (shadow / "_dreams").is_dir() + assert (shadow / "_dreams" / "_index.md").is_file() + assert (shadow / ".shadowignore").is_file() + assert (shadow / "_index.md").is_file() + assert (shadow / "_prefs.md").is_file() + + state = json.loads((shadow / "_meta" / "state.json").read_text()) + assert state["last_commit"] == "none" + assert state["total_files"] == 0 + assert state["total_symbols"] == 0 + assert state["version"] == 1 + assert state["last_update_type"] == "init" + assert state["total_discoveries"] == 0 + assert state["dream_cycles_completed"] == 0 + + +def test_init_shadow_dreams_index_has_seven_column_header(shadow_init, + tmp_git_repo): + """The _dreams/_index.md scaffold must use the 7-column schema that + dream-reconcile.py / dream-lineage.py validate against.""" + shadow_init.init_shadow(str(tmp_git_repo)) + idx = (tmp_git_repo / ".shadow" / "_dreams" / "_index.md").read_text() + assert "dream_id" in idx + assert "category" in idx + assert "verdict" in idx + assert "title" in idx + assert "branch" in idx + assert "parent" in idx + assert "tip_commit" in idx + + +def test_init_shadow_default_shadowignore_has_expected_content(shadow_init, + tmp_git_repo): + shadow_init.init_shadow(str(tmp_git_repo)) + sig = (tmp_git_repo / ".shadow" / ".shadowignore").read_text() + assert "node_modules/" in sig + assert "*.min.js" in sig + assert ".shadow/" in sig + assert "*.png" in sig + + +def test_init_shadow_prefs_default_content(shadow_init, tmp_git_repo): + shadow_init.init_shadow(str(tmp_git_repo)) + prefs = (tmp_git_repo / ".shadow" / "_prefs.md").read_text() + assert "# Preferences" in prefs + assert "_No preferences recorded yet._" in prefs + + +def test_init_shadow_mixed_language_project(shadow_init, tmp_git_repo): + """Multi-language repo: every recognized source file gets a shadow.""" + (tmp_git_repo / "main.py").write_text("def main(): pass\n") + (tmp_git_repo / "app.js").write_text("function start() {}\n") + (tmp_git_repo / "lib.go").write_text("package x\nfunc Hello() {}\n") + (tmp_git_repo / "mod.rb").write_text("class Foo\n def bar\n end\nend\n") + (tmp_git_repo / "thing.rs").write_text("pub fn run() {}\n") + (tmp_git_repo / "shell.sh").write_text("greet() { echo hi; }\n") + sub = tmp_git_repo / "src" + sub.mkdir() + (sub / "deep.ts").write_text("export function compute() {}\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "init"], + cwd=tmp_git_repo, check=True) + + ok = shadow_init.init_shadow(str(tmp_git_repo)) + assert ok is True + shadow = tmp_git_repo / ".shadow" + for path in ("main.py.md", "app.js.md", "lib.go.md", "mod.rb.md", + "thing.rs.md", "shell.sh.md", "src/deep.ts.md"): + assert (shadow / path).is_file(), f"missing shadow: {path}" + + state = json.loads((shadow / "_meta" / "state.json").read_text()) + assert state["total_files"] == 7 + assert state["total_symbols"] >= 7 # each file contributes >=1 symbol + assert re.fullmatch(r"[0-9a-f]{40}", state["last_commit"]) + + index = (shadow / "_index.md").read_text() + for p in ("main.py", "app.js", "lib.go", "mod.rb", + "thing.rs", "shell.sh", "src/deep.ts"): + assert p in index + + +def test_init_shadow_state_counts_match_disk(shadow_init, tmp_git_repo): + """state.json totals must match actual files/symbols on disk.""" + (tmp_git_repo / "a.py").write_text( + "def f1(): pass\ndef f2(): pass\nclass C:\n def m(self): pass\n" + ) + (tmp_git_repo / "b.py").write_text("def only(): pass\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "m"], + cwd=tmp_git_repo, check=True) + + shadow_init.init_shadow(str(tmp_git_repo)) + state = json.loads( + (tmp_git_repo / ".shadow" / "_meta" / "state.json").read_text() + ) + # Two source files + assert state["total_files"] == 2 + # 4 syms in a.py (f1, f2, C, C.m) + 1 in b.py + assert state["total_symbols"] == 5 + + +def test_init_shadow_refuses_when_shadow_exists_without_reset(shadow_init, + tmp_git_repo, + reset_diagnostics): + (tmp_git_repo / ".shadow").mkdir() + (tmp_git_repo / ".shadow" / "marker").write_text("preserve\n") + ok = reset_diagnostics.init_shadow(str(tmp_git_repo), reset=False) + assert ok is False + assert (tmp_git_repo / ".shadow" / "marker").read_text() == "preserve\n" + msgs = [m for lvl, m in reset_diagnostics._diagnostics if lvl == "error"] + assert any("already exists" in m for m in msgs) + + +def test_init_shadow_reset_replaces_existing(shadow_init, tmp_git_repo): + (tmp_git_repo / ".shadow").mkdir() + (tmp_git_repo / ".shadow" / "stale.md").write_text("old\n") + (tmp_git_repo / "foo.py").write_text("def x(): pass\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "m"], + cwd=tmp_git_repo, check=True) + + ok = shadow_init.init_shadow(str(tmp_git_repo), reset=True) + assert ok is True + assert not (tmp_git_repo / ".shadow" / "stale.md").exists() + assert (tmp_git_repo / ".shadow" / "foo.py.md").is_file() + + +def test_init_shadow_dry_run_creates_no_files(shadow_init, tmp_git_repo, + capsys): + (tmp_git_repo / "foo.py").write_text("def x(): pass\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "m"], + cwd=tmp_git_repo, check=True) + + ok = shadow_init.init_shadow(str(tmp_git_repo), dry_run=True) + assert ok is True + assert not (tmp_git_repo / ".shadow").exists() + out = capsys.readouterr().out.lower() + assert "dry-run" in out + + +def test_init_shadow_dry_run_with_existing_shadow_and_reset(shadow_init, + tmp_git_repo, + capsys): + """dry_run + reset must report 'Would delete' without actually deleting.""" + (tmp_git_repo / ".shadow").mkdir() + (tmp_git_repo / ".shadow" / "keep.txt").write_text("preserved\n") + ok = shadow_init.init_shadow(str(tmp_git_repo), reset=True, dry_run=True) + assert ok is True + # Original .shadow contents untouched + assert (tmp_git_repo / ".shadow" / "keep.txt").read_text() == "preserved\n" + out = capsys.readouterr().out.lower() + assert "would delete" in out + + +def test_init_shadow_skips_excluded_paths_and_basenames(shadow_init, + tmp_git_repo): + """node_modules/, *.lock, *.min.js are filtered by built-in rules.""" + (tmp_git_repo / "real.py").write_text("def x(): pass\n") + nm = tmp_git_repo / "node_modules" + nm.mkdir() + (nm / "junk.js").write_text("function junk(){}\n") + (tmp_git_repo / "huge.lock").write_text("{}\n") + (tmp_git_repo / "min.min.js").write_text("var a=1;\n") + # Non-source file (no recognized extension) + (tmp_git_repo / "README.txt").write_text("docs\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "m"], + cwd=tmp_git_repo, check=True) + + ok = shadow_init.init_shadow(str(tmp_git_repo)) + assert ok is True + shadow = tmp_git_repo / ".shadow" + assert (shadow / "real.py.md").is_file() + assert not (shadow / "node_modules").exists() + assert not (shadow / "huge.lock.md").exists() + assert not (shadow / "min.min.js.md").exists() + assert not (shadow / "README.txt.md").exists() + + state = json.loads((shadow / "_meta" / "state.json").read_text()) + assert state["total_files"] == 1 + + +def test_init_shadow_special_basenames_detected(shadow_init, tmp_git_repo): + """Dockerfile and Makefile (basename-keyed languages) get shadows too.""" + (tmp_git_repo / "Dockerfile").write_text("FROM scratch\n") + (tmp_git_repo / "Makefile").write_text("all:\n\techo hi\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "m"], + cwd=tmp_git_repo, check=True) + + shadow_init.init_shadow(str(tmp_git_repo)) + shadow = tmp_git_repo / ".shadow" + assert (shadow / "Dockerfile.md").is_file() + assert (shadow / "Makefile.md").is_file() + docker_md = (shadow / "Dockerfile.md").read_text() + assert "Dockerfile" in docker_md + + +def test_init_shadow_handles_unknown_language_file(shadow_init, tmp_git_repo): + """A YAML file is recognized as a source file but has no extractor — + a shadow is still created (with File-Level only, no symbols).""" + (tmp_git_repo / "config.yaml").write_text("key: value\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "m"], + cwd=tmp_git_repo, check=True) + + shadow_init.init_shadow(str(tmp_git_repo)) + content = (tmp_git_repo / ".shadow" / "config.yaml.md").read_text() + assert "YAML" in content + assert "## File-Level" in content + assert "## Cross-References" in content + + +def test_init_shadow_returns_true_with_no_source_files(shadow_init, + tmp_git_repo, + reset_diagnostics): + """Empty repo (no files) still succeeds; warns about empty shadow.""" + ok = reset_diagnostics.init_shadow(str(tmp_git_repo)) + assert ok is True + msgs = [m for lvl, m in reset_diagnostics._diagnostics if lvl == "warning"] + assert any("No source files" in m for m in msgs) + + +def test_init_shadow_summary_printed_to_stdout(shadow_init, tmp_git_repo, + capsys): + (tmp_git_repo / "a.py").write_text("def x(): pass\n") + subprocess.run(["git", "add", "-A"], cwd=tmp_git_repo, check=True) + subprocess.run(["git", "commit", "-q", "-m", "m"], + cwd=tmp_git_repo, check=True) + shadow_init.init_shadow(str(tmp_git_repo)) + out = capsys.readouterr().out + assert "Shadow initialized." in out + assert "Files:" in out + assert "Symbols:" in out + assert "Languages:" in out + + +# --------------------------------------------------------------------------- +# main() CLI dispatcher (lines 1408-1465) +# --------------------------------------------------------------------------- + +def test_main_help_exits_zero(shadow_init, monkeypatch, capsys): + monkeypatch.setattr(sys, "argv", ["shadow-init.py", "--help"]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 0 + out = capsys.readouterr().out.lower() + assert "usage" in out or "initialize" in out + + +def test_main_unknown_flag_exits_nonzero(shadow_init, monkeypatch): + monkeypatch.setattr(sys, "argv", + ["shadow-init.py", "--definitely-not-a-flag"]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code != 0 + + +def test_main_with_explicit_root_initializes_shadow(shadow_init, tmp_git_repo, + monkeypatch): + monkeypatch.setattr(sys, "argv", + ["shadow-init.py", "--root", str(tmp_git_repo)]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 0 + assert (tmp_git_repo / ".shadow" / "_meta" / "state.json").is_file() + + +def test_main_nonexistent_root_exits_one(shadow_init, tmp_path, monkeypatch, + capsys): + bogus = tmp_path / "does" / "not" / "exist" + monkeypatch.setattr(sys, "argv", + ["shadow-init.py", "--root", str(bogus)]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 1 + err = capsys.readouterr().err + assert "does not exist" in err.lower() + + +def test_main_root_pointing_to_file_exits_one(shadow_init, tmp_path, + monkeypatch): + f = tmp_path / "regular.txt" + f.write_text("x") + monkeypatch.setattr(sys, "argv", ["shadow-init.py", "--root", str(f)]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 1 + + +def test_main_auto_detect_from_inside_repo(shadow_init, tmp_git_repo, + monkeypatch): + monkeypatch.chdir(tmp_git_repo) + monkeypatch.setattr(sys, "argv", ["shadow-init.py"]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 0 + assert (tmp_git_repo / ".shadow" / "_meta" / "state.json").is_file() + + +def test_main_auto_detect_outside_git_exits_one(shadow_init, tmp_path, + monkeypatch, capsys): + monkeypatch.chdir(tmp_path) + monkeypatch.setattr(sys, "argv", ["shadow-init.py"]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 1 + err = capsys.readouterr().err.lower() + assert "could not detect" in err or "not inside" in err + + +def test_main_dry_run_does_not_write_shadow(shadow_init, tmp_git_repo, + monkeypatch): + monkeypatch.setattr(sys, "argv", + ["shadow-init.py", "--root", str(tmp_git_repo), + "--dry-run"]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 0 + assert not (tmp_git_repo / ".shadow").exists() + + +def test_main_refuses_when_shadow_exists_without_reset(shadow_init, + tmp_git_repo, + monkeypatch): + (tmp_git_repo / ".shadow").mkdir() + monkeypatch.setattr(sys, "argv", + ["shadow-init.py", "--root", str(tmp_git_repo)]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 1 + + +def test_main_reset_replaces_existing_shadow(shadow_init, tmp_git_repo, + monkeypatch): + (tmp_git_repo / ".shadow").mkdir() + (tmp_git_repo / ".shadow" / "stale").write_text("old\n") + monkeypatch.setattr(sys, "argv", + ["shadow-init.py", "--root", str(tmp_git_repo), + "--reset"]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 0 + assert not (tmp_git_repo / ".shadow" / "stale").exists() + + +def test_main_with_coupon_demo_reset(shadow_init, coupon_demo, monkeypatch): + """Re-initializing the example coupon-demo via --reset yields a fresh + .shadow/ that round-trips through state.json correctly.""" + monkeypatch.setattr(sys, "argv", + ["shadow-init.py", "--root", str(coupon_demo), + "--reset"]) + with pytest.raises(SystemExit) as exc: + shadow_init.main() + assert exc.value.code == 0 + state = json.loads( + (coupon_demo / ".shadow" / "_meta" / "state.json").read_text() + ) + assert state["version"] == 1 + assert state["last_update_type"] == "init" + # coupon-demo's tracked files are the .py sources + assert state["total_files"] >= 1 + + +# --------------------------------------------------------------------------- +# Subprocess smoke test for __main__ block (line 1469) +# --------------------------------------------------------------------------- + +@pytest.mark.slow +@pytest.mark.integration +def test_subprocess_main_block_via_cli_help(): + """Exercises the `if __name__ == '__main__'` invocation path.""" + result = subprocess.run( + [sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True, timeout=30, + ) + assert result.returncode == 0 + assert "usage" in result.stdout.lower() diff --git a/tests/skills/shadow_frog_meditate/__init__.py b/tests/skills/shadow_frog_meditate/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/skills/shadow_frog_meditate/test_meditate_repair.py b/tests/skills/shadow_frog_meditate/test_meditate_repair.py new file mode 100644 index 0000000..de3a502 --- /dev/null +++ b/tests/skills/shadow_frog_meditate/test_meditate_repair.py @@ -0,0 +1,381 @@ +"""Tests for meditate-repair.py — dream index repair pipeline.""" +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent +SCRIPT = REPO_ROOT / "skills" / "shadow-frog-meditate" / "meditate-repair.py" + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +def make_dream(dreams_dir: Path, did: str, *, report: str = "", manifest: dict | None = None): + """Create a synthetic dream folder with optional report and manifest.""" + d = dreams_dir / did + d.mkdir(parents=True, exist_ok=True) + if report: + (d / "report.md").write_text(report) + if manifest is not None: + (d / "manifest.json").write_text(json.dumps(manifest)) + return d + + +def make_index(dreams_dir: Path, rows: list[str]) -> Path: + """Write a synthetic _index.md with header + rows.""" + header = ( + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + ) + content = header + "\n".join(rows) + "\n" + index_path = dreams_dir / "_index.md" + index_path.write_text(content) + return index_path + + +# --------------------------------------------------------------------------- +# detect_corrupted tests +# --------------------------------------------------------------------------- + +class TestDetectCorrupted: + def test_valid_not_flagged(self, meditate_repair, tmp_path): + dreams = tmp_path / "_dreams" + did = "20250101-120000Z-valid" + report = f'---\ndream_id: "{did}"\n---\n# Valid\n' + make_dream(dreams, did, report=report) + + result = meditate_repair.detect_corrupted(str(dreams)) + assert did not in result + + def test_mismatched_id_flagged(self, meditate_repair, tmp_path): + dreams = tmp_path / "_dreams" + did = "20250101-120000Z-mismatch" + # Report claims a different dream_id + report = '---\ndream_id: "20250101-120000Z-OTHER"\n---\n# Mismatch\n' + make_dream(dreams, did, report=report) + + result = meditate_repair.detect_corrupted(str(dreams)) + assert did in result + + def test_no_report_not_flagged(self, meditate_repair, tmp_path): + dreams = tmp_path / "_dreams" + did = "20250101-120000Z-noreport" + (dreams / did).mkdir(parents=True) + + result = meditate_repair.detect_corrupted(str(dreams)) + assert did not in result + + +# --------------------------------------------------------------------------- +# lookup_verdict tests +# --------------------------------------------------------------------------- + +class TestLookupVerdict: + def test_manifest_verdict(self, meditate_repair, tmp_path): + did = "20250101-120000Z-v" + d = tmp_path / did + d.mkdir() + (d / "manifest.json").write_text(json.dumps({"verdict": "useful"})) + + result = meditate_repair.lookup_verdict(str(d), "") + assert result == "useful" + + def test_dead_end_signal_in_content(self, meditate_repair, tmp_path): + did = "20250101-120000Z-dead" + d = tmp_path / did + d.mkdir() + content = "## Verdict\nThis was a dead end, no improvement observed." + result = meditate_repair.lookup_verdict(str(d), content) + assert result == "dead_end" + + def test_useful_signal_in_content(self, meditate_repair, tmp_path): + did = "20250101-120000Z-useful" + d = tmp_path / did + d.mkdir() + content = "## Verdict\nAll tests pass and the fix is confirmed." + result = meditate_repair.lookup_verdict(str(d), content) + assert result == "useful" + + def test_no_signal_returns_empty(self, meditate_repair, tmp_path): + did = "20250101-120000Z-nothing" + d = tmp_path / did + d.mkdir() + result = meditate_repair.lookup_verdict(str(d), "Nothing relevant here.") + assert result == "" + + def test_malformed_manifest_falls_through(self, meditate_repair, tmp_path): + did = "20250101-120000Z-bad" + d = tmp_path / did + d.mkdir() + (d / "manifest.json").write_text("not json") + content = "## Verdict\nThis is useful and verified." + result = meditate_repair.lookup_verdict(str(d), content) + assert result == "useful" + + +# --------------------------------------------------------------------------- +# parse_dream_id tests +# --------------------------------------------------------------------------- + +class TestParseDreamId: + def test_valid(self, meditate_repair): + result = meditate_repair.parse_dream_id("20250101-120000Z-my-slug") + assert result == ("20250101-120000Z", "my-slug") + + def test_complex_slug(self, meditate_repair): + result = meditate_repair.parse_dream_id("20250420-140000Z-cache-poison-sequence") + assert result == ("20250420-140000Z", "cache-poison-sequence") + + def test_invalid_returns_none(self, meditate_repair): + assert meditate_repair.parse_dream_id("not-a-dream-id") is None + assert meditate_repair.parse_dream_id("") is None + assert meditate_repair.parse_dream_id("2025-01-01-slug") is None + + +# --------------------------------------------------------------------------- +# repair_row tests +# --------------------------------------------------------------------------- + +class TestRepairRow: + def test_repairs_unknown_category(self, meditate_repair, tmp_path): + dreams = tmp_path / "_dreams" + did = "20250101-120000Z-repair" + report = ( + '---\ndream_id: "20250101-120000Z-repair"\n---\n' + "# Good Title\n\n**Category**: investigation\n\n## Verdict\nAll tests pass.\n" + ) + make_dream(dreams, did, report=report) + + parts = ["", did, "unknown", "unknown", did, "branch", "main", "abc"] + result = meditate_repair.repair_row(parts, str(dreams)) + assert result is not None + new_parts, changed = result + assert changed + assert new_parts[2] == "investigation" + + def test_repairs_unknown_verdict(self, meditate_repair, tmp_path): + dreams = tmp_path / "_dreams" + did = "20250101-120000Z-vrepair" + report = '---\ndream_id: "20250101-120000Z-vrepair"\n---\n# Title\n## Verdict\nDead end.\n' + make_dream(dreams, did, report=report, manifest={"verdict": "dead_end"}) + + parts = ["", did, "investigation", "unknown", did, "branch", "main", "abc"] + result = meditate_repair.repair_row(parts, str(dreams)) + assert result is not None + new_parts, changed = result + assert changed + assert new_parts[3] == "dead_end" + + def test_repairs_generic_title(self, meditate_repair, tmp_path): + dreams = tmp_path / "_dreams" + did = "20250101-120000Z-slug" + report = f'---\ndream_id: "{did}"\n---\n# A Proper Title\nBody.\n' + make_dream(dreams, did, report=report) + + # Generic title = just the slug + parts = ["", did, "investigation", "useful", "20250101-120000Z-slug", "branch", "main", "abc"] + result = meditate_repair.repair_row(parts, str(dreams)) + assert result is not None + new_parts, changed = result + assert changed + assert new_parts[4] == "A Proper Title" + + def test_repairs_unknown_category_from_frontmatter(self, meditate_repair, tmp_path): + """B8: category lives in YAML frontmatter (`category: <value>`), not a + `**Category**:` body marker. Recovery must fall back to frontmatter.""" + dreams = tmp_path / "_dreams" + did = "20250101-120000Z-fmcat" + report = ( + f'---\ndream_id: "{did}"\ncategory: "bug hunting"\nverdict: useful\n---\n' + "# Good Title\n\nBody with no Category marker.\n" + ) + make_dream(dreams, did, report=report) + + parts = ["", did, "unknown", "useful", did, "branch", "main", "abc"] + result = meditate_repair.repair_row(parts, str(dreams)) + assert result is not None + new_parts, changed = result + assert changed + assert new_parts[2] == "bug hunting" + + def test_no_report_returns_none(self, meditate_repair, tmp_path): + dreams = tmp_path / "_dreams" + did = "20250101-120000Z-gone" + (dreams / did).mkdir(parents=True) + parts = ["", did, "unknown", "unknown", did, "branch", "main", "abc"] + result = meditate_repair.repair_row(parts, str(dreams)) + assert result is None + + +# --------------------------------------------------------------------------- +# resolve_parent_in_index tests +# --------------------------------------------------------------------------- + +class TestResolveParentInIndex: + def test_exact_match(self, meditate_repair): + all_ids = ["20250101-120000Z-base", "20250102-120000Z-extend"] + result = meditate_repair.resolve_parent_in_index("20250101-120000Z-base", all_ids) + assert result == "20250101-120000Z-base" + + def test_slug_match(self, meditate_repair): + all_ids = ["20250101-120000Z-base", "20250102-120000Z-extend"] + # Different timestamp, same slug + result = meditate_repair.resolve_parent_in_index("99999999-999999Z-base", all_ids) + assert result == "20250101-120000Z-base" + + def test_no_match_returns_none(self, meditate_repair): + all_ids = ["20250101-120000Z-base"] + result = meditate_repair.resolve_parent_in_index("20250101-120000Z-nonexist", all_ids) + assert result is None + + def test_invalid_id_returns_none(self, meditate_repair): + all_ids = ["20250101-120000Z-base"] + result = meditate_repair.resolve_parent_in_index("not-valid", all_ids) + assert result is None + + +# --------------------------------------------------------------------------- +# repair_parent tests (I4 regression — step 10/11) +# --------------------------------------------------------------------------- + +class TestRepairParent: + def test_manifest_provides_parent(self, meditate_repair, tmp_path): + """Step 10: manifest parent_branch repairs the parent cell.""" + dreams = tmp_path / "_dreams" + parent_did = "20250101-120000Z-base" + child_did = "20250102-120000Z-child" + make_dream(dreams, parent_did, report=f'---\ndream_id: "{parent_did}"\n---\n# Base\n') + make_dream( + dreams, child_did, + report=f'---\ndream_id: "{child_did}"\n---\n# Child\n', + manifest={"parent_branch": f"dream/proj/{parent_did}"}, + ) + + all_ids = [parent_did, child_did] + bmap = {d: f"dream/proj/{d}" for d in all_ids} + parts = ["", child_did, "investigation", "useful", "Child", + f"dream/proj/{child_did}", "main", "abc"] + changed = meditate_repair.repair_parent(parts, str(dreams), all_ids, bmap) + assert changed + # Parent column is a BRANCH NAME (the resolved row's branch), not a dream_id. + assert parts[6] == f"dream/proj/{parent_did}" + + def test_missing_manifest_and_slug_heuristic(self, meditate_repair, tmp_path): + """Step 11: slug heuristic with compounding suffix.""" + dreams = tmp_path / "_dreams" + parent_did = "20250101-120000Z-retry-logic" + child_did = "20250102-120000Z-retry-logic-extend" + make_dream(dreams, parent_did, report=f'---\ndream_id: "{parent_did}"\n---\n# Base\n') + make_dream(dreams, child_did, report=f'---\ndream_id: "{child_did}"\n---\n# Extend\n') + + all_ids = [parent_did, child_did] + bmap = {d: f"dream/proj/{d}" for d in all_ids} + parts = ["", child_did, "investigation", "useful", "Extend", + f"dream/proj/{child_did}", "main", "abc"] + changed = meditate_repair.repair_parent(parts, str(dreams), all_ids, bmap) + assert changed + assert parts[6] == f"dream/proj/{parent_did}" + + def test_ambiguity_warning_keeps_main(self, meditate_repair, tmp_path, capsys): + """Multiple candidate parents with no exact slug match → emits warning, keeps main.""" + dreams = tmp_path / "_dreams" + # Two candidates whose slugs both start with "cache" (the base_slug after removing -fix) + parent1 = "20250101-120000Z-cache-v1" + parent2 = "20250101-130000Z-cache-v2" + # Child slug is "cache-fix"; stripping "-fix" → base_slug = "cache" + # Both parents' slugs start with "cache" so both match, neither is exact "cache" + child_did = "20250102-120000Z-cache-fix" + make_dream(dreams, parent1, report=f'---\ndream_id: "{parent1}"\n---\n# P1\n') + make_dream(dreams, parent2, report=f'---\ndream_id: "{parent2}"\n---\n# P2\n') + make_dream(dreams, child_did, report=f'---\ndream_id: "{child_did}"\n---\n# Child\n') + + all_ids = [parent1, parent2, child_did] + bmap = {d: f"dream/proj/{d}" for d in all_ids} + parts = ["", child_did, "investigation", "useful", "Child", + f"dream/proj/{child_did}", "main", "abc"] + changed = meditate_repair.repair_parent(parts, str(dreams), all_ids, bmap) + assert not changed + assert parts[6] == "main" + # Check warning was emitted + captured = capsys.readouterr() + assert "AMBIGUOUS" in captured.err + + def test_no_change_when_parent_not_main(self, meditate_repair, tmp_path): + """If parent is already not 'main', no repair attempted.""" + dreams = tmp_path / "_dreams" + did = "20250101-120000Z-test" + make_dream(dreams, did, report=f'---\ndream_id: "{did}"\n---\n# T\n') + + parts = ["", did, "investigation", "useful", "T", "branch", + "dream/proj/20250101-110000Z-other", "abc"] + changed = meditate_repair.repair_parent( + parts, str(dreams), [did], {did: f"dream/proj/{did}"} + ) + assert not changed + + +# --------------------------------------------------------------------------- +# CLI integration tests +# --------------------------------------------------------------------------- + +@pytest.mark.slow +class TestCLI: + def test_help_exits_zero(self): + result = subprocess.run( + [sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True, + ) + assert result.returncode == 0 + + def test_repair_corrupted_index(self, tmp_path): + """Runs repair against a synthetic corrupted index.""" + shadow = tmp_path / ".shadow" + dreams = shadow / "_dreams" + dreams.mkdir(parents=True) + + did_ok = "20250101-120000Z-good" + did_bad = "20250102-120000Z-needsfix" + + make_dream( + dreams, did_ok, + report=f'---\ndream_id: "{did_ok}"\ncategory: investigation\nverdict: useful\n---\n# Good Title\n', + manifest={"verdict": "useful"}, + ) + make_dream( + dreams, did_bad, + report=f'---\ndream_id: "{did_bad}"\n---\n# Real Title\n\n**Category**: bug hunting\n\n## Verdict\nAll tests pass.\n', + manifest={"verdict": "useful"}, + ) + + rows = [ + f"| {did_ok} | investigation | useful | Good Title | dream/proj/{did_ok} | main | aaa |", + f"| {did_bad} | unknown | unknown | {did_bad} | dream/proj/{did_bad} | main | bbb |", + ] + make_index(dreams, rows) + + result = subprocess.run( + [sys.executable, str(SCRIPT), "--shadow-dir", str(shadow)], + capture_output=True, text=True, + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + assert "Repaired" in result.stdout + + # Verify the index was actually fixed + repaired_content = (dreams / "_index.md").read_text() + assert "bug hunting" in repaired_content + assert "Real Title" in repaired_content + + def test_clean_index_no_repairs(self, coupon_demo): + """Running against clean coupon_demo reports 0 repairs.""" + result = subprocess.run( + [sys.executable, str(SCRIPT), "--shadow-dir", str(coupon_demo / ".shadow")], + capture_output=True, text=True, + ) + assert result.returncode == 0 + # Should say "Repaired 0 rows" + assert "Repaired 0" in result.stdout or "No repairs" in result.stdout.lower() diff --git a/tests/skills/shadow_frog_viewer/__init__.py b/tests/skills/shadow_frog_viewer/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/skills/shadow_frog_viewer/test_dream_lineage.py b/tests/skills/shadow_frog_viewer/test_dream_lineage.py new file mode 100644 index 0000000..cdd0267 --- /dev/null +++ b/tests/skills/shadow_frog_viewer/test_dream_lineage.py @@ -0,0 +1,992 @@ +"""Tests for dream-lineage.py — HTML visualization of dream experiment lineage.""" +import json +import os +import re +import subprocess +import sys +from html.parser import HTMLParser +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent +SCRIPT = REPO_ROOT / "skills" / "shadow-frog-viewer" / "dream-lineage.py" + + +# --------------------------------------------------------------------------- +# md_to_html tests +# --------------------------------------------------------------------------- + +class TestMdToHtml: + """Test the markdown-to-HTML converter.""" + + def test_simple_list(self, dream_lineage): + html = dream_lineage.md_to_html("- one\n- two") + assert "<ul>" in html + assert html.count("<li>") == 2 + assert html.count("</li>") == 2 + + def test_nested_list(self, dream_lineage): + html = dream_lineage.md_to_html("- outer\n - inner") + assert "<ul>" in html + assert "class='nested'" in html + # Both items inside one <ul> + ul_content = re.search(r"<ul>(.*?)</ul>", html, re.S) + assert ul_content + assert ul_content.group(1).count("<li") == 2 + + def test_md_to_html_no_orphan_li_outside_ul_regression(self, dream_lineage): + """Regression: all <li> tags must be inside <ul> or <ol> blocks.""" + html = dream_lineage.md_to_html("- one\n- two\n - nested\n- three") + # No orphan <li> patterns + assert "</p><li>" not in html + assert "<p><li>" not in html + # Count: all <li> must be within <ul>...</ul> + total_li = html.count("<li") + li_in_ul = sum( + block.count("<li") + for block in re.findall(r"<ul>(.*?)</ul>", html, re.S) + ) + assert total_li == li_in_ul, f"Orphan <li> found: {total_li} total vs {li_in_ul} in <ul>" + + def test_mixed_text_and_list(self, dream_lineage): + html = dream_lineage.md_to_html("text\n\n- one\n- two\n\nmore") + assert "<ul>" in html + assert "<li>" in html + # Paragraphs around the list + assert "<p>" in html + + @pytest.mark.parametrize("md,tag", [ + ("# h", "<h2>"), + ("## h", "<h3>"), + ("### h", "<h4>"), + ]) + def test_headings(self, dream_lineage, md, tag): + html = dream_lineage.md_to_html(md) + assert tag in html + + def test_inline_code(self, dream_lineage): + html = dream_lineage.md_to_html("`x`") + assert "<code>x</code>" in html + + def test_bold(self, dream_lineage): + html = dream_lineage.md_to_html("**x**") + assert "<strong>x</strong>" in html + + def test_code_blocks(self, dream_lineage): + md = "```py\ndef f():\n pass\n```" + html = dream_lineage.md_to_html(md) + assert "<pre><code>" in html + assert "def f():" in html + + def test_table(self, dream_lineage): + md = "| A | B |\n|---|---|\n| 1 | 2 |\n| 3 | 4 |" + html = dream_lineage.md_to_html(md) + assert "<table>" in html + assert "<th>" in html + assert "<td>" in html + assert html.count("<th>") == 2 + assert html.count("<td>") == 4 + + +# --------------------------------------------------------------------------- +# stable_id tests +# --------------------------------------------------------------------------- + +class TestStableId: + def test_idempotent(self, dream_lineage): + branch = "dream/coupon-demo/20260420-140000Z-cache-poison-sequence" + id1 = dream_lineage.stable_id(branch) + id2 = dream_lineage.stable_id(branch) + assert id1 == id2 + + def test_starts_with_rpt(self, dream_lineage): + assert dream_lineage.stable_id("foo/bar").startswith("rpt-") + + def test_alphanumeric_and_dash(self, dream_lineage): + result = dream_lineage.stable_id("a/b.c!d") + # Only alphanumeric and dashes + assert re.match(r"^rpt-[a-zA-Z0-9-]+$", result) + + +# --------------------------------------------------------------------------- +# tree_depth tests +# --------------------------------------------------------------------------- + +class TestTreeDepth: + def test_leaf_node(self, dream_lineage): + children = {"main": ["a"], "a": []} + assert dream_lineage.tree_depth("a", children) == 0 + + def test_one_level(self, dream_lineage): + children = {"main": ["a"], "a": ["b"]} + assert dream_lineage.tree_depth("a", children) == 1 + + def test_two_levels(self, dream_lineage): + children = {"main": ["a"], "a": ["b"], "b": ["c"]} + assert dream_lineage.tree_depth("a", children) == 2 + + +# --------------------------------------------------------------------------- +# flatten_chain tests +# --------------------------------------------------------------------------- + +class TestFlattenChain: + def test_single_node(self, dream_lineage): + meta = {"a": {}} + children = {} + result = dream_lineage.flatten_chain("a", meta, children) + assert result == [("a", 0)] + + def test_chain_dfs_order(self, dream_lineage): + meta = {"a": {}, "b": {}, "c": {}} + children = {"a": ["b"], "b": ["c"]} + result = dream_lineage.flatten_chain("a", meta, children) + assert result == [("a", 0), ("b", 1), ("c", 2)] + + def test_branching(self, dream_lineage): + meta = {"a": {}, "b": {}, "c": {}} + children = {"a": ["b", "c"]} + result = dream_lineage.flatten_chain("a", meta, children) + assert ("a", 0) in result + assert ("b", 1) in result + assert ("c", 1) in result + + +# --------------------------------------------------------------------------- +# find_shadow_dir tests +# --------------------------------------------------------------------------- + +class TestFindShadowDir: + def test_with_hint(self, dream_lineage, tmp_path): + shadow = tmp_path / ".shadow" + shadow.mkdir() + result = dream_lineage.find_shadow_dir(str(shadow)) + assert result == str(shadow) + + def test_hint_nonexistent_falls_through(self, dream_lineage, tmp_path, monkeypatch): + """When hint doesn't exist, falls through to cwd-based detection.""" + monkeypatch.chdir(tmp_path) + shadow = tmp_path / ".shadow" + shadow.mkdir() + result = dream_lineage.find_shadow_dir("/nonexistent/path") + assert result == ".shadow" or result == str(shadow) + + +# --------------------------------------------------------------------------- +# load_index tests +# --------------------------------------------------------------------------- + +class TestLoadIndex: + def test_coupon_demo(self, dream_lineage, coupon_demo): + shadow_dir = str(coupon_demo / ".shadow") + meta, children = dream_lineage.load_index(shadow_dir) + assert len(meta) == 3 + # All three are parented to main + assert len(children["main"]) == 3 + + def test_synthetic_minimal(self, dream_lineage, tmp_path): + """Synthetic index with parent references.""" + shadow = tmp_path / ".shadow" + dreams = shadow / "_dreams" + dreams.mkdir(parents=True) + + index_content = ( + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + "| 20250101-120000Z-base | investigation | useful | Base exp | dream/proj/20250101-120000Z-base | main | abc1234 |\n" + "| 20250102-120000Z-extend | investigation | useful | Extend exp | dream/proj/20250102-120000Z-extend | dream/proj/20250101-120000Z-base | def5678 |\n" + ) + (dreams / "_index.md").write_text(index_content) + # Create minimal dream dirs + (dreams / "20250101-120000Z-base").mkdir() + (dreams / "20250102-120000Z-extend").mkdir() + + meta, children = dream_lineage.load_index(str(shadow)) + assert len(meta) == 2 + # extend should be child of base, not main + base_branch = "dream/proj/20250101-120000Z-base" + extend_branch = "dream/proj/20250102-120000Z-extend" + assert extend_branch in children[base_branch] + + def test_fallback_reparenting_via_slug(self, dream_lineage, tmp_path): + """When parent branch has different timestamp, slug matching resolves it.""" + shadow = tmp_path / ".shadow" + dreams = shadow / "_dreams" + dreams.mkdir(parents=True) + + # Parent listed with wrong timestamp prefix but matching slug + index_content = ( + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + "| 20250101-120000Z-t01-base | investigation | useful | Base | dream/proj/20250101-120000Z-t01-base | main | abc1234 |\n" + "| 20250102-120000Z-t02-child | investigation | useful | Child | dream/proj/20250102-120000Z-t02-child | dream/proj/99999999-999999Z-t01-base | def5678 |\n" + ) + (dreams / "_index.md").write_text(index_content) + (dreams / "20250101-120000Z-t01-base").mkdir() + (dreams / "20250102-120000Z-t02-child").mkdir() + + meta, children = dream_lineage.load_index(str(shadow)) + base_branch = "dream/proj/20250101-120000Z-t01-base" + child_branch = "dream/proj/20250102-120000Z-t02-child" + # Child should be re-parented to base via slug match + assert child_branch in children[base_branch] + + +# --------------------------------------------------------------------------- +# load_reports tests +# --------------------------------------------------------------------------- + +class TestLoadReports: + def test_synthetic_reports(self, dream_lineage, tmp_path): + shadow = tmp_path / ".shadow" + dreams = shadow / "_dreams" + did = "20250101-120000Z-test" + dream_dir = dreams / did + dream_dir.mkdir(parents=True) + + report = ( + "---\ndream_id: \"20250101-120000Z-test\"\ncategory: investigation\n" + "verdict: useful\n---\n\n# Test Experiment\n\nBody content here.\n" + ) + (dream_dir / "report.md").write_text(report) + + manifest = {"verdict": "useful", "tests_passed": 5, "discoveries": ["a", "b"]} + (dream_dir / "manifest.json").write_text(json.dumps(manifest)) + + branch = "dream/proj/20250101-120000Z-test" + meta = {branch: {"did": did}} + dream_lineage.load_reports(str(shadow), meta) + + info = meta[branch] + assert "Test Experiment" in info["full_report"] + assert info["tests"] == "5" + assert info["discoveries_count"] == 2 + + +# --------------------------------------------------------------------------- +# CLI integration tests +# --------------------------------------------------------------------------- + +@pytest.mark.slow +class TestCLI: + def test_help_exits_zero(self): + result = subprocess.run( + [sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True, + ) + assert result.returncode == 0 + assert "Dream lineage" in result.stdout or "output" in result.stdout.lower() + + def test_output_html_in_coupon_demo(self, coupon_demo): + out_file = coupon_demo / "lineage-test.html" + result = subprocess.run( + [sys.executable, str(SCRIPT), "-o", str(out_file), + "--shadow-dir", str(coupon_demo / ".shadow")], + capture_output=True, text=True, + cwd=str(coupon_demo), + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + assert out_file.exists() + assert out_file.stat().st_size > 1024 + + def test_output_html_structure(self, coupon_demo): + out_file = coupon_demo / "lineage-struct.html" + subprocess.run( + [sys.executable, str(SCRIPT), "-o", str(out_file), + "--shadow-dir", str(coupon_demo / ".shadow")], + capture_output=True, text=True, cwd=str(coupon_demo), check=True, + ) + content = out_file.read_text() + + # Parse with HTMLParser — should not raise + errors = [] + + class Checker(HTMLParser): + def handle_starttag(self, tag, attrs): + pass + + def handle_endtag(self, tag): + pass + + def error(self, message): + errors.append(message) + + parser = Checker() + parser.feed(content) + + assert not errors + assert "<html>" in content + assert "<body>" in content + # Has panel-body or panel-overlay class + assert "panel-overlay" in content or "panel-body" in content + + +# --------------------------------------------------------------------------- +# Helpers for HTML structural verification +# --------------------------------------------------------------------------- + +def _build_dreams(shadow_dir: Path, rows, reports=None, manifests=None): + """Write a synthetic _dreams/_index.md and per-dream dirs. + + Each row is a 7-tuple matching the pipe columns: + (dream_id, category, verdict, title, branch, parent, tip_commit) + + `reports` and `manifests` are dicts keyed by dream_id mapping to + raw text / dict bodies to write under that dream's directory. + Returns the path to _dreams/. + """ + dreams = shadow_dir / "_dreams" + dreams.mkdir(parents=True, exist_ok=True) + header = ( + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + ) + body_lines = [] + for did, cat, verdict, title, branch, parent, tip in rows: + body_lines.append( + f"| {did} | {cat} | {verdict} | {title} | {branch} | {parent} | {tip} |" + ) + (dreams / did).mkdir(exist_ok=True) + (dreams / "_index.md").write_text(header + "\n".join(body_lines) + "\n") + for did, text in (reports or {}).items(): + (dreams / did / "report.md").write_text(text) + for did, data in (manifests or {}).items(): + (dreams / did / "manifest.json").write_text(json.dumps(data)) + return dreams + + +class _BalanceChecker(HTMLParser): + """Track tag open/close balance for void-aware HTML validation.""" + + VOID = { + "area", "base", "br", "col", "embed", "hr", "img", "input", + "link", "meta", "source", "track", "wbr", + } + + def __init__(self): + super().__init__() + self.stack = [] + self.errors = [] + + def handle_starttag(self, tag, attrs): + if tag not in self.VOID: + self.stack.append(tag) + + def handle_startendtag(self, tag, attrs): + pass # self-closing tags balance themselves + + def handle_endtag(self, tag): + if tag in self.VOID: + return + if not self.stack: + self.errors.append(f"close {tag} with empty stack") + return + # Allow implicit close of inline-tolerant containers above. + if self.stack[-1] == tag: + self.stack.pop() + elif tag in self.stack: + # Pop until match — record skipped tags as warnings, not errors. + while self.stack and self.stack[-1] != tag: + self.stack.pop() + if self.stack: + self.stack.pop() + else: + self.errors.append(f"close {tag} not in stack") + + +def _li_runs_inside_ul(html_text: str) -> bool: + """Verify every <li> is contained within a <ul> or <ol>. + + Same invariant as TestMdToHtml.test_md_to_html_no_orphan_li_outside_ul_regression + but applied to the full generated page. + """ + total_li = html_text.count("<li") + in_ul = sum( + block.count("<li") + for block in re.findall(r"<ul>(.*?)</ul>", html_text, re.S) + ) + in_ol = sum( + block.count("<li") + for block in re.findall(r"<ol(?:\s[^>]*)?>(.*?)</ol>", html_text, re.S) + ) + return total_li == in_ul + in_ol + + +# --------------------------------------------------------------------------- +# node_html tests (lines 311-334) +# --------------------------------------------------------------------------- + +class TestNodeHtml: + """Unit tests for the flat timeline row renderer.""" + + def test_minimal_meta_no_report(self, dream_lineage): + meta = {"b1": {"short": "exp", "cat": "investigation", "verdict": "useful"}} + children = {} + out = dream_lineage.node_html("b1", meta, children) + assert 'class="tl-row"' in out + assert "exp" in out + assert "✅" in out # verdict + # No report → no 📄 button + assert "report-btn" not in out + + def test_with_report_renders_button(self, dream_lineage): + meta = { + "b1": { + "short": "exp", "cat": "bug hunting", "verdict": "useful", + "title": "T", "tests": "5", "discoveries_count": 3, + "full_report": "body", + } + } + out = dream_lineage.node_html("b1", {**meta}, {}, with_report=True) + assert "report-btn" in out + assert "📄" in out + # badge content + assert "5 tests" in out + assert "3 disc" in out + + def test_with_report_button_suppressed(self, dream_lineage): + meta = {"b1": {"short": "exp", "full_report": "body"}} + out = dream_lineage.node_html("b1", meta, {}, with_report=False) + assert "report-btn" not in out + + def test_unknown_category_uses_default_color(self, dream_lineage): + meta = {"b1": {"short": "exp", "cat": "made-up-cat"}} + out = dream_lineage.node_html("b1", meta, {}) + # default color #607D8B from CAT_COLORS lookup fallback + assert "#607D8B" in out + + def test_depth_attribute_rendered(self, dream_lineage): + meta = {"b1": {"short": "exp", "_depth": 4}} + out = dream_lineage.node_html("b1", meta, {}) + # depth appears inside the .tl-depth pill + assert ">4<" in out + + def test_html_escapes_title(self, dream_lineage): + meta = {"b1": {"short": "exp", "title": "<script>alert(1)</script>"}} + out = dream_lineage.node_html("b1", meta, {}) + assert "<script>" not in out + assert "<script>" in out + + def test_missing_branch_falls_back_to_branch_name(self, dream_lineage): + # meta has no entry for branch → uses branch as `short` + out = dream_lineage.node_html("orphan-branch", {}, {}) + assert "orphan-branch" in out + + +# --------------------------------------------------------------------------- +# compact_node tests (lines 352-387) +# --------------------------------------------------------------------------- + +class TestCompactNode: + """Unit tests for the compact tree row renderer.""" + + def test_leaf_uses_last_connector(self, dream_lineage): + meta = {"b1": {"short": "exp", "cat": "investigation", "verdict": "useful"}} + out = dream_lineage.compact_node("b1", meta, {}) + assert "└── " in out + assert "exp" in out + + def test_non_last_uses_branch_connector(self, dream_lineage): + meta = {"b1": {"short": "exp", "cat": "investigation"}} + out = dream_lineage.compact_node("b1", meta, {}, is_last=False) + assert "├── " in out + + def test_recurses_into_children(self, dream_lineage): + meta = { + "a": {"short": "rootA", "cat": "investigation"}, + "b": {"short": "kidB", "cat": "investigation"}, + "c": {"short": "kidC", "cat": "investigation"}, + } + children = {"a": ["b", "c"]} + out = dream_lineage.compact_node("a", meta, children) + # All three nodes rendered, two as kids + assert "rootA" in out + assert "kidB" in out + assert "kidC" in out + # last kid uses └, the earlier uses ├ + assert "├── " in out + assert "└── " in out + + def test_report_btn_only_when_report_present(self, dream_lineage): + no_report = {"b1": {"short": "exp"}} + assert "ct-report-btn" not in dream_lineage.compact_node("b1", no_report, {}) + with_report = {"b1": {"short": "exp", "full_report": "x"}} + assert "ct-report-btn" in dream_lineage.compact_node("b1", with_report, {}) + + def test_tests_count_appears(self, dream_lineage): + meta = {"b1": {"short": "exp", "tests": "12"}} + out = dream_lineage.compact_node("b1", meta, {}) + assert "12t" in out + + def test_html_escapes_title(self, dream_lineage): + meta = {"b1": {"short": "exp", "title": "<x>"}} + out = dream_lineage.compact_node("b1", meta, {}) + assert "<x>" not in out.replace("<div", "").replace("<span", "") + assert "<x>" in out + + def test_deep_chain_indents(self, dream_lineage): + # a → b → c chain renders increasing prefix + meta = {k: {"short": k, "cat": "investigation"} for k in "abc"} + children = {"a": ["b"], "b": ["c"]} + out = dream_lineage.compact_node("a", meta, children) + # Three rows, and the deepest one carries the " " spacing + assert out.count("ct-line") >= 3 + + +# --------------------------------------------------------------------------- +# generate_html — happy path on coupon-demo +# --------------------------------------------------------------------------- + +class TestGenerateHtmlCouponDemo: + """Drive generate_html against the canonical fixture and assert structure.""" + + @pytest.fixture + def rendered(self, dream_lineage, coupon_demo, tmp_path, capsys): + out = tmp_path / "lineage.html" + dream_lineage.generate_html(str(coupon_demo / ".shadow"), str(out)) + # capture but don't fail on console output + capsys.readouterr() + return out.read_text() + + def test_writes_non_empty_file(self, dream_lineage, coupon_demo, tmp_path): + out = tmp_path / "lineage.html" + dream_lineage.generate_html(str(coupon_demo / ".shadow"), str(out)) + assert out.exists() + assert out.stat().st_size > 4 * 1024 + + def test_doctype_and_skeleton(self, rendered): + assert rendered.startswith("<!DOCTYPE html>") + assert "<html>" in rendered + assert "<head>" in rendered + assert "<body>" in rendered + assert "</body></html>" in rendered + + def test_all_three_dream_ids_present(self, rendered): + for did in [ + "20260420-140000Z-cache-poison-sequence", + "20260420-141000Z-bulk-min-total-interaction", + "20260420-142000Z-adversarial-inputs", + ]: + # The "short" form (after Z-) is what is rendered most places + short = did.split("Z-", 1)[1] + assert short in rendered, f"missing {short}" + + def test_title_text_present(self, rendered): + for title_frag in [ + "Cache poisoning via validate-then-calculate sequence", + "Bulk discount vs coupon min_total interaction", + "Adversarial input audit", + ]: + assert title_frag in rendered + + def test_categories_rendered_in_cat_bar(self, rendered): + # cat-bar uses Title Case + assert "Bug Hunting" in rendered + assert "Investigation" in rendered + assert "Security Audit" in rendered + + def test_tab_skeleton_classes_present(self, rendered): + for cls in ["tab-bar", "tab-content", "panel-overlay", "compact-tree"]: + assert cls in rendered + + def test_stat_dashboard_has_six_stats(self, rendered): + # Six labels in the dashboard + for label in ["Experiments", "Compounding", "Fresh", "Sessions", + "Chains", "Max Depth"]: + assert f">{label}<" in rendered + + def test_constellation_svg_present(self, rendered): + # Graph tab removed — constellation SVG should no longer be emitted + assert 'id="constellation"' not in rendered + + def test_no_orphan_li_outside_ul(self, rendered): + assert _li_runs_inside_ul(rendered), "found <li> outside any <ul>/<ol>" + + def test_html_parses_without_errors(self, rendered): + # html.parser should consume entire payload without raising. + parser = _BalanceChecker() + parser.feed(rendered) + # We only assert no fatal errors — minor imbalance is tolerated. + assert all("with empty stack" not in e for e in parser.errors), parser.errors + + def test_templates_emitted_for_each_dream(self, rendered): + # 3 reports → 3 <template> tags + assert rendered.count("<template id=") == 3 + + def test_console_summary_printed(self, dream_lineage, coupon_demo, tmp_path, capsys): + out = tmp_path / "lineage.html" + dream_lineage.generate_html(str(coupon_demo / ".shadow"), str(out)) + captured = capsys.readouterr() + assert "Wrote " in captured.out + assert "3 experiments" in captured.out + + def test_verdict_legend_appears(self, rendered): + # All three dreams are "useful" → legend shows ✅ Useful: 3 + assert "Useful" in rendered + assert "<strong>3</strong>" in rendered + + def test_fresh_count_equals_total(self, dream_lineage, coupon_demo, tmp_path, capsys): + # All three coupon-demo dreams parent to main with no children + # → 3 fresh, 0 compounding + out = tmp_path / "lineage.html" + dream_lineage.generate_html(str(coupon_demo / ".shadow"), str(out)) + msg = capsys.readouterr().out + assert "0 compounding" in msg + assert "3 fresh" in msg + + +# --------------------------------------------------------------------------- +# generate_html — synthetic edge cases +# --------------------------------------------------------------------------- + +class TestGenerateHtmlEdgeCases: + """Edge-case shadow trees: empty, missing, malformed, lineage chains.""" + + def test_empty_dreams_no_rows(self, dream_lineage, tmp_path, capsys): + """Only the header rows in _index.md — no actual experiments.""" + shadow = tmp_path / ".shadow" + _build_dreams(shadow, rows=[]) + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + capsys.readouterr() + + html = out.read_text() + assert html.startswith("<!DOCTYPE html>") + # Zero experiments dashboard + assert ">0<" in html + # No <template> tags emitted + assert "<template id=" not in html + + def test_missing_dreams_dir_exits(self, dream_lineage, tmp_path): + """generate_html → load_index sys.exit(1) when _index.md is missing.""" + shadow = tmp_path / ".shadow" + shadow.mkdir() + out = tmp_path / "out.html" + with pytest.raises(SystemExit) as exc: + dream_lineage.generate_html(str(shadow), str(out)) + assert exc.value.code == 1 + + def test_dream_in_index_but_dir_missing(self, dream_lineage, tmp_path, capsys): + """If a dream is in _index.md but its folder is missing, render anyway.""" + shadow = tmp_path / ".shadow" + # Build with one row, then delete its folder + rows = [( + "20250101-120000Z-ghost", "investigation", "useful", + "Ghost experiment", + "dream/proj/20250101-120000Z-ghost", "main", "deadbeef", + )] + _build_dreams(shadow, rows=rows) + (shadow / "_dreams" / "20250101-120000Z-ghost").rmdir() + + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + capsys.readouterr() + html = out.read_text() + # The short name should still appear in the rendered shell + assert "ghost" in html + # No template since report.md was never created + assert "<template id=" not in html + + def test_malformed_index_rows_skipped(self, dream_lineage, tmp_path, capsys): + """Lines with fewer than 8 pipe-delimited parts are silently ignored.""" + shadow = tmp_path / ".shadow" + dreams = shadow / "_dreams" + dreams.mkdir(parents=True) + # Mixture: valid row + various broken rows + content = ( + "# Dream Experiments\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|---|---|---|---|---|---|---|\n" + "| 20250101-120000Z-ok | investigation | useful | OK | " + "dream/proj/20250101-120000Z-ok | main | abc1 |\n" + "| missing pipes here\n" + "|||| not enough cols ||\n" + "\n" + "trailing prose line that is not a table row\n" + ) + (dreams / "_index.md").write_text(content) + (dreams / "20250101-120000Z-ok").mkdir() + + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + msg = capsys.readouterr().out + assert "1 experiments" in msg + assert "ok" in out.read_text() + + def test_lineage_chain_renders_both_ancestors(self, dream_lineage, tmp_path, capsys): + """A → B → C chain: all 3 short names appear and chain count = 1.""" + shadow = tmp_path / ".shadow" + rows = [ + ("20250101-120000Z-root-A", "investigation", "useful", "Root", + "dream/proj/20250101-120000Z-root-A", "main", "aaa1"), + ("20250102-120000Z-mid-B", "investigation", "useful", "Mid", + "dream/proj/20250102-120000Z-mid-B", + "dream/proj/20250101-120000Z-root-A", "bbb2"), + ("20250103-120000Z-leaf-C", "investigation", "useful", "Leaf", + "dream/proj/20250103-120000Z-leaf-C", + "dream/proj/20250102-120000Z-mid-B", "ccc3"), + ] + _build_dreams(shadow, rows=rows) + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + msg = capsys.readouterr().out + html = out.read_text() + + # Chain depth: 3-node chain = 1 root with depth 2 → max_depth = 3 + assert "1 chains" in msg + assert "max depth 3" in msg + for short in ["root-A", "mid-B", "leaf-C"]: + assert short in html + + def test_builds_on_reparents_via_report(self, dream_lineage, tmp_path, capsys): + """builds_on in report frontmatter overrides 'main' parent.""" + shadow = tmp_path / ".shadow" + rows = [ + ("20250101-120000Z-base", "investigation", "useful", "Base", + "dream/proj/20250101-120000Z-base", "main", "aaa1"), + ("20250102-120000Z-child", "investigation", "useful", "Child", + "dream/proj/20250102-120000Z-child", "main", "bbb2"), + ] + reports = { + "20250102-120000Z-child": ( + "---\n" + "dream_id: \"20250102-120000Z-child\"\n" + "builds_on: [\"dream/proj/20250101-120000Z-base\"]\n" + "---\n\n# Child\n" + ), + "20250101-120000Z-base": "---\ndream_id: base\n---\n\n# Base\n", + } + _build_dreams(shadow, rows=rows, reports=reports) + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + msg = capsys.readouterr().out + # 1 root → 1 chain, child has compounded onto base + assert "1 chains" in msg + assert "1 compounding" in msg + + def test_manifest_parent_branch_reparents(self, dream_lineage, tmp_path, capsys): + """manifest.json parent_branch overrides _index.md 'main' parent.""" + shadow = tmp_path / ".shadow" + rows = [ + ("20250101-120000Z-a", "investigation", "useful", "A", + "dream/proj/20250101-120000Z-a", "main", "aaa1"), + ("20250102-120000Z-b", "investigation", "useful", "B", + "dream/proj/20250102-120000Z-b", "main", "bbb2"), + ] + manifests = { + "20250102-120000Z-b": { + "parent_branch": "dream/proj/20250101-120000Z-a", + "discoveries": [], + }, + } + _build_dreams(shadow, rows=rows, manifests=manifests) + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + msg = capsys.readouterr().out + assert "1 compounding" in msg + + def test_fresh_grouped_by_category(self, dream_lineage, tmp_path, capsys): + """Multiple categories → multiple fresh-group blocks.""" + shadow = tmp_path / ".shadow" + rows = [ + ("20250101-120000Z-i", "investigation", "useful", "I", + "dream/proj/20250101-120000Z-i", "main", "111"), + ("20250102-120000Z-b", "bug hunting", "dead_end", "B", + "dream/proj/20250102-120000Z-b", "main", "222"), + ("20250103-120000Z-s", "security audit", "useful", "S", + "dream/proj/20250103-120000Z-s", "main", "333"), + ] + _build_dreams(shadow, rows=rows) + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + capsys.readouterr() + html = out.read_text() + assert html.count('class="fresh-group"') == 3 + # Dead-end ❌ rendered + assert "❌" in html + # Dead End legend appears + assert "Dead End" in html + + def test_unknown_category_falls_back(self, dream_lineage, tmp_path, capsys): + """A category not in the well-known list still renders as 'unknown'.""" + shadow = tmp_path / ".shadow" + rows = [( + "20250101-120000Z-weird", "unknown", "unknown", "Weird", + "dream/proj/20250101-120000Z-weird", "main", "abc", + )] + _build_dreams(shadow, rows=rows) + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + capsys.readouterr() + html = out.read_text() + # The "Unknown" cat-stat block + assert "Unknown" in html + + def test_report_renders_into_template(self, dream_lineage, tmp_path, capsys): + shadow = tmp_path / ".shadow" + rows = [( + "20250101-120000Z-rep", "investigation", "useful", "R", + "dream/proj/20250101-120000Z-rep", "main", "abc", + )] + reports = { + "20250101-120000Z-rep": ( + "---\ndream_id: r\n---\n\n# Big Heading\n\n" + "- item one\n- item two\n\n" + "Some prose.\n" + ), + } + _build_dreams(shadow, rows=rows, reports=reports) + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + capsys.readouterr() + html = out.read_text() + # template body contains markdown→html output + assert "<template id=" in html + assert "<h2>Big Heading</h2>" in html + assert "<li>item one</li>" in html + + def test_session_count_groups_by_date_hour(self, dream_lineage, tmp_path, capsys): + """Sessions = unique YYYYMMDD-HHMM prefixes.""" + shadow = tmp_path / ".shadow" + rows = [ + ("20250101-1200-a", "investigation", "useful", "A", + "dream/proj/20250101-1200-a", "main", "1"), + ("20250101-1259-b", "investigation", "useful", "B", + "dream/proj/20250101-1259-b", "main", "2"), + ("20250102-0800-c", "investigation", "useful", "C", + "dream/proj/20250102-0800-c", "main", "3"), + ] + _build_dreams(shadow, rows=rows) + out = tmp_path / "out.html" + dream_lineage.generate_html(str(shadow), str(out)) + capsys.readouterr() + html = out.read_text() + # 3 sessions: 20250101-1200, 20250101-1259, 20250102-0800 + m = re.search(r'<div class="num">(\d+)</div><div class="label">Sessions</div>', html) + assert m and m.group(1) == "3" + + +# --------------------------------------------------------------------------- +# CLI integration: argparse + __main__ dispatch +# --------------------------------------------------------------------------- + +@pytest.mark.slow +class TestCLIMain: + """Cover lines 31-36 (parse_args) and 1062-1064 (__main__ block).""" + + def test_default_output_path(self, coupon_demo): + # No -o flag → writes dream-lineage.html in cwd + result = subprocess.run( + [sys.executable, str(SCRIPT), + "--shadow-dir", str(coupon_demo / ".shadow")], + capture_output=True, text=True, cwd=str(coupon_demo), + ) + assert result.returncode == 0, result.stderr + default = coupon_demo / "dream-lineage.html" + assert default.exists() + assert default.stat().st_size > 1024 + + def test_custom_output_path(self, coupon_demo, tmp_path): + target = tmp_path / "nested" / "out.html" + target.parent.mkdir() + result = subprocess.run( + [sys.executable, str(SCRIPT), + "-o", str(target), + "--shadow-dir", str(coupon_demo / ".shadow")], + capture_output=True, text=True, cwd=str(coupon_demo), + ) + assert result.returncode == 0, result.stderr + assert target.exists() + assert target.read_text().startswith("<!DOCTYPE html>") + + def test_shadow_dir_auto_detect_from_cwd(self, coupon_demo): + # When --shadow-dir is omitted, find_shadow_dir scans cwd for .shadow + out_file = coupon_demo / "auto.html" + result = subprocess.run( + [sys.executable, str(SCRIPT), "-o", str(out_file)], + capture_output=True, text=True, cwd=str(coupon_demo), + ) + assert result.returncode == 0, result.stderr + assert out_file.exists() + + def test_missing_shadow_dir_exits_nonzero(self, tmp_path): + # No .shadow/ anywhere → find_shadow_dir prints ERROR and exits 1 + empty = tmp_path / "empty" + empty.mkdir() + result = subprocess.run( + [sys.executable, str(SCRIPT)], + capture_output=True, text=True, cwd=str(empty), + ) + assert result.returncode == 1 + assert "ERROR" in result.stderr + + def test_summary_line_count_in_stdout(self, coupon_demo, tmp_path): + out_file = tmp_path / "out.html" + result = subprocess.run( + [sys.executable, str(SCRIPT), + "-o", str(out_file), + "--shadow-dir", str(coupon_demo / ".shadow")], + capture_output=True, text=True, cwd=str(coupon_demo), + ) + assert result.returncode == 0 + assert "3 experiments" in result.stdout + assert "Wrote" in result.stdout + + +# --- Regression: tree_depth cycle guard --- +# Bug surfaced by Phase-3 audit: prior tree_depth(node, children) had no +# cycle guard, so a malformed _index.md with a self-parent or A<->B loop +# would crash with RecursionError. Now guarded via _seen set. + +class TestTreeDepthCycleGuard: + def test_self_parent_does_not_recurse(self, dream_lineage): + children = {"A": ["A"]} + depth = dream_lineage.tree_depth("A", children) + assert isinstance(depth, int) + assert depth >= 0 + + def test_two_node_cycle_does_not_recurse(self, dream_lineage): + children = {"A": ["B"], "B": ["A"]} + depth = dream_lineage.tree_depth("A", children) + assert isinstance(depth, int) + assert depth >= 0 + + def test_three_node_cycle_does_not_recurse(self, dream_lineage): + children = {"A": ["B"], "B": ["C"], "C": ["A"]} + depth = dream_lineage.tree_depth("A", children) + assert isinstance(depth, int) + assert depth >= 0 + + def test_cycle_with_branch_does_not_recurse(self, dream_lineage): + children = {"A": ["B", "D"], "B": ["A"], "D": []} + depth = dream_lineage.tree_depth("A", children) + assert isinstance(depth, int) + assert depth >= 1 + + def test_no_cycle_still_gives_correct_depth(self, dream_lineage): + children = {"A": ["B"], "B": ["C"], "C": []} + assert dream_lineage.tree_depth("A", children) == 2 + assert dream_lineage.tree_depth("B", children) == 1 + assert dream_lineage.tree_depth("C", children) == 0 + + def test_leaf_node_returns_zero(self, dream_lineage): + assert dream_lineage.tree_depth("leaf", {}) == 0 + + def test_generate_html_survives_cyclic_index(self, dream_lineage, tmp_path): + # End-to-end: a malformed _dreams/_index.md with a cycle does + # not crash generate_html (would have raised RecursionError + # before the fix). + shadow = tmp_path / ".shadow" + (shadow / "_dreams").mkdir(parents=True) + (shadow / "_dreams" / "_index.md").write_text( + "# Dream Index\n\n" + "| dream_id | category | verdict | title | branch | parent | tip_commit |\n" + "|----------|----------|---------|-------|--------|--------|------------|\n" + "| ida | exploration | useful | T-A | dream/proj/A | dream/proj/B | abc |\n" + "| idb | exploration | useful | T-B | dream/proj/B | dream/proj/A | def |\n" + ) + out = tmp_path / "lineage.html" + dream_lineage.generate_html(shadow, out) + assert out.exists() diff --git a/tests/skills/shadow_frog_viewer/test_shadow_viewer.py b/tests/skills/shadow_frog_viewer/test_shadow_viewer.py new file mode 100644 index 0000000..1ce2bd8 --- /dev/null +++ b/tests/skills/shadow_frog_viewer/test_shadow_viewer.py @@ -0,0 +1,2184 @@ +r"""Tests for `skills/shadow-frog-viewer/shadow-viewer.py`. + +Philosophy: USE REAL FILES (per `minimal-mocking-tests`). The viewer is a +pure-read tool — every test either constructs a small shadow tree on +disk and calls a viewer function, or exercises the CLI via subprocess +against the `coupon_demo` fixture. + +Test categories: + * In-process function tests (no `@pytest.mark.slow`): exercise + parsing helpers directly via the `shadow_viewer` fixture. + * CLI integration tests (`@pytest.mark.slow @pytest.mark.integration`): + invoke shadow-viewer.py as a subprocess against `coupon_demo`. + +B3 regression: a discovery whose continuation lines include +``Dream report: `_dreams/<slug>/` `` must extract the slug path into +`meta["dream_report"]` and must NOT include "Dream report" or the slug +in the discovery body text. +""" +import json +import os +import re +import subprocess +import sys +import textwrap +from datetime import datetime + +import pytest + + +# --- Helpers --------------------------------------------------------------- + + +def _write_shadow(shadow_dir, rel_path, content): + """Write `content` to <shadow_dir>/<rel_path>, creating parents.""" + p = shadow_dir / rel_path + p.parent.mkdir(parents=True, exist_ok=True) + p.write_text(textwrap.dedent(content), encoding="utf-8") + return p + + +def _make_shadow_root(tmp_path): + """Create an empty .shadow/ dir under tmp_path and return it.""" + sd = tmp_path / ".shadow" + sd.mkdir() + return sd + + +def _run_viewer(repo_root, cwd, *args): + """Run shadow-viewer.py as a subprocess from `cwd`.""" + script = repo_root / "skills/shadow-frog-viewer/shadow-viewer.py" + return subprocess.run( + [sys.executable, str(script), *args], + cwd=str(cwd), + capture_output=True, + text=True, + check=False, + ) + + +# --- parse_discovery ------------------------------------------------------- + + +def test_parse_discovery_standard(shadow_viewer): + """Basic verified/exploration discovery, no labels.""" + line = "- Caches None for invalid codes" + cont = [" _(verified, source: exploration)_"] + d = shadow_viewer.parse_discovery(line, cont) + assert d["text"] == "Caches None for invalid codes" + assert d["status"] == "verified" + assert d["source"] == "exploration" + assert "labels" not in d + assert "dream_report" not in d + + +def test_parse_discovery_with_labels(shadow_viewer): + """Labels are parsed into a list, trimmed, lowercase comma-split.""" + line = "- Foo" + cont = [" _(verified, source: user, labels: [bug, security])_"] + d = shadow_viewer.parse_discovery(line, cont) + assert d["text"] == "Foo" + assert d["status"] == "verified" + assert d["source"] == "user" + assert d["labels"] == ["bug", "security"] + + +@pytest.mark.parametrize("status", ["verified", "uncertain", "refuted"]) +def test_parse_discovery_status_variants(shadow_viewer, status): + line = "- Some discovery" + cont = [f" _({status}, source: exploration)_"] + d = shadow_viewer.parse_discovery(line, cont) + assert d["status"] == status + assert d["source"] == "exploration" + + +@pytest.mark.parametrize("source", ["exploration", "user", "interaction"]) +def test_parse_discovery_source_variants(shadow_viewer, source): + line = "- Some discovery" + cont = [f" _(verified, source: {source})_"] + d = shadow_viewer.parse_discovery(line, cont) + assert d["source"] == source + + +def test_parse_discovery_also_involves(shadow_viewer): + """`Also involves:` populates a list of file::symbol anchors.""" + line = "- A multi-symbol discovery" + cont = [ + " _(verified, source: exploration)_", + " Also involves: `inventory.py::validate_coupon`, `cart.py::COUPON_CACHE`", + ] + d = shadow_viewer.parse_discovery(line, cont) + assert d["text"] == "A multi-symbol discovery" + assert d["also_involves"] == [ + "inventory.py::validate_coupon", + "cart.py::COUPON_CACHE", + ] + # also_involves line must not leak into the body text + assert "Also involves" not in d["text"] + + +def test_parse_discovery_b3_dream_report_regression(shadow_viewer): + """B3 regression: Dream report goes into meta['dream_report'] and is + excluded from the body text.""" + line = "- Case-variant lookups create duplicate cache entries" + cont = [ + " _(verified, source: exploration, labels: [bug, performance])_", + " Dream report: `_dreams/20260420-140000Z-cache-poison-sequence/`", + ] + d = shadow_viewer.parse_discovery(line, cont) + # Body text is preserved, with no Dream report leakage + assert d["text"] == "Case-variant lookups create duplicate cache entries" + assert "Dream report" not in d["text"] + assert "_dreams/" not in d["text"] + # meta["dream_report"] captures the backtick payload (slug folder path) + assert d["dream_report"] == ( + "_dreams/20260420-140000Z-cache-poison-sequence/" + ) + + +def test_parse_discovery_b3_dream_report_with_also_involves(shadow_viewer): + """Dream report + Also involves on the same discovery — both extracted, + neither leaks into body text.""" + line = "- Discovery with both extras" + cont = [ + " _(verified, source: exploration, labels: [security])_", + " Dream report: `_dreams/20260420-142000Z-adversarial-inputs/`", + " Also involves: `cart.py::get_coupon`, `cart.py::COUPON_CACHE`", + ] + d = shadow_viewer.parse_discovery(line, cont) + assert d["text"] == "Discovery with both extras" + assert "Dream report" not in d["text"] + assert "Also involves" not in d["text"] + assert d["dream_report"] == ( + "_dreams/20260420-142000Z-adversarial-inputs/" + ) + assert d["also_involves"] == [ + "cart.py::get_coupon", + "cart.py::COUPON_CACHE", + ] + + +def test_parse_discovery_multiline_body(shadow_viewer): + """Lines that are neither metadata nor structured extras are appended to + the body text.""" + line = "- Lead sentence." + cont = [ + " continuation prose", + " _(verified, source: exploration)_", + ] + d = shadow_viewer.parse_discovery(line, cont) + assert "Lead sentence." in d["text"] + assert "continuation prose" in d["text"] + assert d["status"] == "verified" + + +def test_parse_discovery_no_metadata(shadow_viewer): + """Bullet with no metadata blob still returns a dict with text but no + status/source keys.""" + d = shadow_viewer.parse_discovery("- bare bullet", []) + assert d["text"] == "bare bullet" + assert "status" not in d + assert "source" not in d + + +def test_parse_discovery_preferences_source_only(shadow_viewer): + """Preferences use `_(source: user)_` (no status). Extracts source.""" + d = shadow_viewer.parse_discovery( + "- Prefer X over Y", [" _(source: user)_"] + ) + assert d["text"] == "Prefer X over Y" + assert d["source"] == "user" + assert "status" not in d + + +def test_parse_discovery_none_input_does_not_crash(shadow_viewer): + """Passing a non-string line shouldn't raise — should return a dict.""" + d = shadow_viewer.parse_discovery(None, None) + assert isinstance(d, dict) + assert "text" in d + + +# --- parse_shadow_file ----------------------------------------------------- + + +def test_parse_shadow_file_placeholder(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + f = _write_shadow(sd, "foo.py.md", """\ + # Shadow: foo.py + + **Language**: Python | **Lines**: 10 + + _No discoveries yet._ + """) + res = shadow_viewer.parse_shadow_file(f) + assert res["source_file"] == "foo.py" + assert res["language"] == "Python" + assert res["lines"] == 10 + assert res["symbols"] == [] + assert res["discoveries"] == [] + assert res["cross_references"] == [] + assert res["parse_errors"] == [] + + +def test_parse_shadow_file_one_symbol_one_discovery(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + f = _write_shadow(sd, "auth.py.md", """\ + # Shadow: auth.py + + **Language**: Python | **Lines**: 42 + + ## `authenticate` + + - Returns None on expired tokens, silently. + _(verified, source: exploration, labels: [security])_ + """) + res = shadow_viewer.parse_shadow_file(f) + assert res["source_file"] == "auth.py" + assert res["symbols"] == ["authenticate"] + assert len(res["discoveries"]) == 1 + d = res["discoveries"][0] + assert d["symbol"] == "authenticate" + assert d["file"] == "auth.py" + assert d["status"] == "verified" + assert d["source"] == "exploration" + assert d["labels"] == ["security"] + assert "Returns None" in d["text"] + + +def test_parse_shadow_file_cross_references_backpointers( + shadow_viewer, tmp_path +): + sd = _make_shadow_root(tmp_path) + f = _write_shadow(sd, "bar.py.md", """\ + # Shadow: bar.py + + ## `func` + + - A discovery. + _(verified, source: exploration)_ + + ## Cross-References + + - [my-cross-cutting](_cross/my-cross-cutting.md) + (involves `bar.py::func`) + - [another-one](_cross/another-one.md) + """) + res = shadow_viewer.parse_shadow_file(f) + assert res["symbols"] == ["func"] + # Cross-references back-pointer link labels are collected + assert "my-cross-cutting" in res["cross_references"] + assert "another-one" in res["cross_references"] + # Cross-ref bullets are NOT mistaken for discoveries + assert len(res["discoveries"]) == 1 + + +def test_parse_shadow_file_file_level_and_cross_refs(shadow_viewer, tmp_path): + """`## File-Level` discoveries are tagged with symbol='file-level' and + `## Cross-References` bullets are not treated as discoveries.""" + sd = _make_shadow_root(tmp_path) + f = _write_shadow(sd, "mix.py.md", """\ + # Shadow: mix.py + + ## File-Level + + - A module-wide observation. + _(verified, source: exploration)_ + + ## `helper` + + - A symbol discovery. + _(verified, source: user)_ + + ## Cross-References + + - [shared](_cross/shared.md) + """) + res = shadow_viewer.parse_shadow_file(f) + discs = res["discoveries"] + assert len(discs) == 2 + by_sym = {d["symbol"]: d for d in discs} + assert "file-level" in by_sym + assert "helper" in by_sym + assert by_sym["file-level"]["source"] == "exploration" + assert by_sym["helper"]["source"] == "user" + assert res["cross_references"] == ["shared"] + + +def test_parse_shadow_file_malformed_does_not_crash(shadow_viewer, tmp_path): + """Bizarre / structurally broken content shouldn't raise.""" + sd = _make_shadow_root(tmp_path) + f = _write_shadow(sd, "junk.py.md", """\ + # Shadow: junk.py + **Language**: notnumeric | **Lines**: notanumber + + ## not a backtick heading + - orphan bullet with no metadata + ## `realsym` + - real disc + _(verified, source: exploration)_ + """) + res = shadow_viewer.parse_shadow_file(f) + # Parse succeeds despite weirdness + assert res["source_file"] == "junk.py" + # The bad "Lines" cell stays None (int parse skipped) + assert res["lines"] is None + # `realsym` is captured; the non-backtick heading is not a symbol + assert "realsym" in res["symbols"] + assert "not a backtick heading" not in res["symbols"] + + +def test_parse_shadow_file_missing_file_returns_error( + shadow_viewer, tmp_path +): + """Reading a non-existent path records a parse_error, doesn't raise.""" + res = shadow_viewer.parse_shadow_file(tmp_path / "ghost.md") + assert res["parse_errors"] + assert res["symbols"] == [] + assert res["discoveries"] == [] + + +# --- parse_cross_cutting --------------------------------------------------- + + +def test_parse_cross_cutting_empty_dir(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + (sd / "_cross").mkdir() + assert shadow_viewer.parse_cross_cutting(sd) == [] + + +def test_parse_cross_cutting_no_dir(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + # _cross/ not created + assert shadow_viewer.parse_cross_cutting(sd) == [] + + +def test_parse_cross_cutting_one_entry(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_cross/example-pattern.md", """\ + # Example pattern + + **Category**: pattern + **Refs**: + - `cart.py::calculate_total` + - `inventory.py::validate_coupon` + + **Discovery**: A multi-file pattern observed across the codebase. + + _(verified, source: exploration, labels: [bug])_ + """) + entries = shadow_viewer.parse_cross_cutting(sd) + assert len(entries) == 1 + e = entries[0] + assert e["slug"] == "example-pattern" + assert e["title"] == "Example pattern" + assert e["category"] == "pattern" + assert "cart.py::calculate_total" in e["refs"] + assert "inventory.py::validate_coupon" in e["refs"] + assert "multi-file pattern" in e["discovery"] + assert e["status"] == "verified" + assert e["source"] == "exploration" + assert e["labels"] == ["bug"] + + +def test_parse_cross_cutting_no_labels(shadow_viewer, tmp_path): + """A cross-cutting entry without labels is still parsed, with no + `labels` key in the entry.""" + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_cross/no-label.md", """\ + # Plain entry + + **Category**: behavior + **Refs**: + - `foo.py::bar` + + **Discovery**: Something happens. + + _(uncertain, source: exploration)_ + """) + entries = shadow_viewer.parse_cross_cutting(sd) + assert len(entries) == 1 + e = entries[0] + assert e["status"] == "uncertain" + assert e["source"] == "exploration" + assert "labels" not in e + + +def test_parse_cross_cutting_minor_format_variation(shadow_viewer, tmp_path): + """File missing a `**Category**:` field doesn't crash; entry is still + emitted with whatever fields could be parsed.""" + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_cross/sparse.md", """\ + # Sparse entry + + Some prose with no structured fields. + + _(refuted, source: user)_ + """) + entries = shadow_viewer.parse_cross_cutting(sd) + assert len(entries) == 1 + e = entries[0] + assert e["slug"] == "sparse" + assert e["title"] == "Sparse entry" + assert e["status"] == "refuted" + assert e["source"] == "user" + + +# --- parse_prefs ----------------------------------------------------------- + + +def test_parse_prefs_missing_file(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + assert shadow_viewer.parse_prefs(sd) == [] + + +def test_parse_prefs_zero(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_prefs.md", """\ + # Preferences + + _No preferences recorded yet._ + """) + assert shadow_viewer.parse_prefs(sd) == [] + + +def test_parse_prefs_one(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_prefs.md", """\ + # Preferences + + - Always use type hints on public APIs. + _(source: user)_ + """) + prefs = shadow_viewer.parse_prefs(sd) + assert len(prefs) == 1 + assert prefs[0]["text"] == "Always use type hints on public APIs." + assert prefs[0]["source"] == "user" + assert prefs[0]["type"] == "preference" + + +def test_parse_prefs_three(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_prefs.md", """\ + # Preferences + + - Use kebab-case for slugs. + _(source: user)_ + - Never commit secrets to source control. + _(source: interaction)_ + - Prefer fail-fast for required dependencies. + _(source: user)_ + """) + prefs = shadow_viewer.parse_prefs(sd) + assert len(prefs) == 3 + texts = [p["text"] for p in prefs] + assert any("kebab-case" in t for t in texts) + assert any("secrets" in t for t in texts) + assert any("fail-fast" in t for t in texts) + sources = [p["source"] for p in prefs] + assert "user" in sources + assert "interaction" in sources + + +# --- load_state ------------------------------------------------------------ + + +def test_load_state_missing(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + assert shadow_viewer.load_state(sd) == {} + + +def test_load_state_valid_json(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + (sd / "_meta").mkdir() + payload = { + "version": 1, + "total_files": 5, + "total_discoveries": 42, + "last_update_type": "auto", + } + (sd / "_meta" / "state.json").write_text( + json.dumps(payload), encoding="utf-8" + ) + state = shadow_viewer.load_state(sd) + assert state == payload + + +def test_load_state_malformed_json(shadow_viewer, tmp_path): + sd = _make_shadow_root(tmp_path) + (sd / "_meta").mkdir() + (sd / "_meta" / "state.json").write_text( + "{not valid json", encoding="utf-8" + ) + state = shadow_viewer.load_state(sd) + assert state == {} + + +def test_load_state_non_object_json(shadow_viewer, tmp_path): + """JSON parses but is not a dict — returns empty dict sentinel.""" + sd = _make_shadow_root(tmp_path) + (sd / "_meta").mkdir() + (sd / "_meta" / "state.json").write_text("[1, 2, 3]", encoding="utf-8") + assert shadow_viewer.load_state(sd) == {} + + +# --- get_all_shadow_files -------------------------------------------------- + + +def test_get_all_shadow_files_excludes_special(shadow_viewer, tmp_path): + """Per-file shadows are returned; _cross/, _dreams/, _meta/, _index.md, + _prefs.md, state.json are all excluded.""" + sd = _make_shadow_root(tmp_path) + # Files that SHOULD be returned + _write_shadow(sd, "a.py.md", "# Shadow: a.py\n") + _write_shadow(sd, "src/b.py.md", "# Shadow: src/b.py\n") + _write_shadow(sd, "deep/nested/c.py.md", "# Shadow: deep/nested/c.py\n") + # Files that should be EXCLUDED + _write_shadow(sd, "_index.md", "# Shadow Index\n") + _write_shadow(sd, "_prefs.md", "# Preferences\n") + _write_shadow(sd, "_cross/some.md", "# Some cross\n") + _write_shadow(sd, "_dreams/dream-1/report.md", "# A dream\n") + _write_shadow(sd, "_meta/state.json", "{}") + # Junk files that are not .md and shouldn't show up anyway + (sd / "notes.json").write_text("{}", encoding="utf-8") + + files = shadow_viewer.get_all_shadow_files(sd) + rels = sorted(f.relative_to(sd).as_posix() for f in files) + assert rels == ["a.py.md", "deep/nested/c.py.md", "src/b.py.md"] + + +def test_get_all_shadow_files_against_coupon_demo(shadow_viewer, coupon_demo): + sd = coupon_demo / ".shadow" + files = shadow_viewer.get_all_shadow_files(sd) + rels = sorted(f.relative_to(sd).as_posix() for f in files) + assert rels == ["cart.py.md", "inventory.py.md", "test_cart.py.md"] + # Make sure none of the special files leaked through + for r in rels: + assert not r.startswith(("_cross/", "_dreams/", "_meta/")) + assert r not in ("_index.md", "_prefs.md") + + +# --- collect_all_discoveries (against fixture) ----------------------------- + + +def test_collect_all_discoveries_counts(shadow_viewer, coupon_demo): + """Coupon demo has 33 per-file discoveries (matches state.json).""" + sd = coupon_demo / ".shadow" + all_disc = shadow_viewer.collect_all_discoveries(sd) + # state.json claims 33 total discoveries + assert len(all_disc) == 33 + + # File breakdown: cart=14, inventory=10, test_cart=9 + by_file = {} + for d in all_disc: + by_file.setdefault(d.get("file"), 0) + by_file[d.get("file")] += 1 + assert by_file == {"cart.py": 14, "inventory.py": 10, "test_cart.py": 9} + + +def test_collect_all_discoveries_shadow_path_and_mtime( + shadow_viewer, coupon_demo +): + sd = coupon_demo / ".shadow" + all_disc = shadow_viewer.collect_all_discoveries(sd) + for d in all_disc: + assert "shadow_path" in d + assert d["shadow_path"].endswith(".md") + assert "shadow_mtime" in d + assert isinstance(d["shadow_mtime"], float) + + +def test_collect_all_discoveries_b3_no_dream_report_in_text( + shadow_viewer, coupon_demo +): + """B3 regression on real fixture: no discovery body should contain + 'Dream report' or the literal `_dreams/` slug path.""" + sd = coupon_demo / ".shadow" + all_disc = shadow_viewer.collect_all_discoveries(sd) + leaks = [ + d for d in all_disc + if "Dream report" in d.get("text", "") + or "_dreams/" in d.get("text", "") + ] + assert leaks == [], ( + f"Dream report leaked into {len(leaks)} discovery body/bodies: " + f"{[d['text'][:80] for d in leaks]}" + ) + + # And at least some discoveries actually have dream_report metadata + # (the fixture has several Dream report continuation lines) + with_dream = [d for d in all_disc if d.get("dream_report")] + assert len(with_dream) >= 3, ( + "Expected coupon-demo fixture to have multiple Dream-report-tagged " + f"discoveries; found {len(with_dream)}" + ) + for d in with_dream: + assert d["dream_report"].startswith("_dreams/") + + +# --- CLI: --summary -------------------------------------------------------- + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_summary(repo_root, coupon_demo): + r = _run_viewer(repo_root, coupon_demo, "--summary") + assert r.returncode == 0, r.stderr + out = r.stdout + # Header is "Files shadowed:", "Symbols tracked:", "Discoveries:" per + # current --summary output. Task wording used "Total files:"/"Symbols:" + # /"Discoveries:" loosely — match the actual labels. + assert "Files shadowed:" in out + assert "Symbols tracked:" in out + assert "Discoveries:" in out + # Cross-cutting block exists + assert "Cross-cutting" in out + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_default_view_is_summary(repo_root, coupon_demo): + """Running with no flags should produce the summary view.""" + r = _run_viewer(repo_root, coupon_demo) + assert r.returncode == 0, r.stderr + assert "Shadow Knowledge Base Summary" in r.stdout + + +# --- CLI: --search --------------------------------------------------------- + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_search_matches_text(repo_root, coupon_demo): + r = _run_viewer(repo_root, coupon_demo, "--search", "coupon") + assert r.returncode == 0, r.stderr + # Header echoes the query and results were found + assert "'coupon'" in r.stdout + # Some matched line should contain the query (case-insensitive) + assert "coupon" in r.stdout.lower() + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_search_no_results(repo_root, coupon_demo): + r = _run_viewer(repo_root, coupon_demo, "--search", "zzznotpresentzzz") + assert r.returncode == 0, r.stderr + assert "No results for 'zzznotpresentzzz'" in r.stdout + + +# --- CLI: --top ------------------------------------------------------------ + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_top_cart_py_header_and_no_dream_report_leak( + repo_root, coupon_demo +): + """B3 regression at the CLI layer: --top output for a file whose shadow + contains `Dream report:` continuation lines must NOT inline that text + in any discovery body.""" + r = _run_viewer(repo_root, coupon_demo, "--top", "cart.py") + assert r.returncode == 0, r.stderr + out = r.stdout + # Header form: "Top N of M actionable discoveries for cart.py:" + assert "actionable discoveries for cart.py:" in out + # Match the documented prefix exactly + assert out.splitlines()[0].startswith("Top ") + assert "for cart.py:" in out.splitlines()[0] + + # B3: no literal "Dream report:" or raw `_dreams/...` slug paths in any + # of the discovery bullets + assert "Dream report:" not in out + assert "_dreams/" not in out + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_top_label_filter_bug(repo_root, coupon_demo): + r = _run_viewer( + repo_root, coupon_demo, "--top", "cart.py", "--top-labels", "bug" + ) + assert r.returncode == 0, r.stderr + out = r.stdout + assert "for cart.py:" in out + # Every bulleted result line should mention 'bug' in its label bracket + bullet_lines = [ + l for l in out.splitlines() if l.startswith("- [") + ] + assert bullet_lines, f"No bullets in --top output:\n{out}" + for line in bullet_lines: + bracket = line.split("]", 1)[0] + assert "bug" in bracket, ( + f"Expected 'bug' label in bracket of: {line}" + ) + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_top_unknown_file(repo_root, coupon_demo): + """A file with no shadow + no cross refs reports nothing actionable.""" + r = _run_viewer(repo_root, coupon_demo, "--top", "does/not/exist.py") + assert r.returncode == 0, r.stderr + assert "No actionable discoveries" in r.stdout + + +# --- CLI: --labels (repo-wide) --------------------------------------------- + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_labels_bug(repo_root, coupon_demo): + r = _run_viewer(repo_root, coupon_demo, "--labels", "bug") + assert r.returncode == 0, r.stderr + out = r.stdout + assert "label(s): bug" in out + # There are multiple bug-labeled discoveries in the fixture + assert "[bug]" in out + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_labels_unknown(repo_root, coupon_demo): + r = _run_viewer(repo_root, coupon_demo, "--labels", "nonexistent") + assert r.returncode == 0, r.stderr + assert "No discoveries with label(s): nonexistent" in r.stdout + + +# --- CLI: --recent --------------------------------------------------------- + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_recent_caps_at_n(repo_root, coupon_demo): + r = _run_viewer(repo_root, coupon_demo, "--recent", "5") + assert r.returncode == 0, r.stderr + out = r.stdout + assert "Most Recent Discoveries (top 5)" in out + # Each recent entry has a timestamp prefix " [YYYY-MM-DD HH:MM]" + entries = [l for l in out.splitlines() if l.strip().startswith("[20")] + assert len(entries) <= 5 + # Coupon-demo has > 5 total items so we expect exactly 5 + assert len(entries) == 5 + + +# --- CLI: --prefs ---------------------------------------------------------- + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_prefs_empty(repo_root, coupon_demo): + """Coupon-demo ships with no preferences recorded.""" + r = _run_viewer(repo_root, coupon_demo, "--prefs") + assert r.returncode == 0, r.stderr + assert "No preferences recorded yet." in r.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_prefs_populated(repo_root, coupon_demo): + """After populating _prefs.md, --prefs lists them.""" + prefs_path = coupon_demo / ".shadow" / "_prefs.md" + prefs_path.write_text( + textwrap.dedent("""\ + # Preferences + + - Use snake_case for Python identifiers. + _(source: user)_ + - Avoid mutable default arguments. + _(source: interaction)_ + """), + encoding="utf-8", + ) + r = _run_viewer(repo_root, coupon_demo, "--prefs") + assert r.returncode == 0, r.stderr + assert "Project Preferences (2 total)" in r.stdout + assert "snake_case" in r.stdout + assert "mutable default" in r.stdout + + +# --- CLI: --check-invariants ----------------------------------------------- + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_check_invariants_clean(repo_root, coupon_demo): + r = _run_viewer(repo_root, coupon_demo, "--check-invariants") + assert r.returncode == 0, ( + f"stdout:\n{r.stdout}\nstderr:\n{r.stderr}" + ) + assert "✓ Invariants OK" in r.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_check_invariants_detects_missing_cross_file( + repo_root, coupon_demo +): + """Deleting a _cross/ file leaves dangling back-pointers in per-file + shadows. Invariant #5 must flag this as a violation.""" + cross_path = ( + coupon_demo / ".shadow" / "_cross" + / "coupon-case-normalization-mismatch.md" + ) + assert cross_path.is_file() + cross_path.unlink() + + r = _run_viewer(repo_root, coupon_demo, "--check-invariants") + assert r.returncode != 0, ( + f"Expected nonzero exit when _cross/ file missing; got 0.\n" + f"stdout:\n{r.stdout}\nstderr:\n{r.stderr}" + ) + # Violations report the dangling slug + assert "coupon-case-normalization-mismatch" in r.stdout + assert "cross-ref" in r.stdout + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_check_invariants_detects_bad_heading(repo_root, coupon_demo): + """Renaming a symbol heading to a non-backtick form is a heading-format + violation.""" + cart_path = coupon_demo / ".shadow" / "cart.py.md" + text = cart_path.read_text(encoding="utf-8") + # `## `COUPON_CACHE`` -> `## COUPON_CACHE` (drop backticks) + mutated = text.replace("## `COUPON_CACHE`", "## COUPON_CACHE", 1) + assert mutated != text, "Substitution did not match" + cart_path.write_text(mutated, encoding="utf-8") + + r = _run_viewer(repo_root, coupon_demo, "--check-invariants") + assert r.returncode != 0, ( + f"Expected nonzero exit for bad heading; got 0.\n" + f"stdout:\n{r.stdout}\nstderr:\n{r.stderr}" + ) + assert "heading" in r.stdout + + +# --- CLI: missing shadow dir ---------------------------------------------- + + +@pytest.mark.slow +@pytest.mark.integration +def test_cli_missing_shadow_dir_fails(repo_root, tmp_path): + """Running with no shadow and a bogus --shadow-dir exits 1.""" + r = _run_viewer( + repo_root, tmp_path, + "--shadow-dir", str(tmp_path / "nope"), "--summary", + ) + assert r.returncode == 1 + assert "No .shadow/ directory found" in r.stderr + + +# =========================================================================== +# RENDER FUNCTIONS: in-process tests +# +# The CLI integration tests above invoke shadow-viewer.py as a subprocess — +# that exercises the dispatcher but doesn't contribute to coverage of the +# loaded module. The tests below call view_* and main() directly via the +# `shadow_viewer` fixture so coverage actually accumulates. +# =========================================================================== + + +# --- shared helpers -------------------------------------------------------- + + +def _call_main(shadow_viewer, argv): + """Invoke `shadow_viewer.main()` in-process with the given argv. + + Returns the integer exit code. We mutate `sys.argv` directly (no + monkeypatch / no mocking) and always restore it in a finally. + """ + saved_argv = sys.argv + sys.argv = ["shadow-viewer.py", *argv] + try: + shadow_viewer.main() + return 0 + except SystemExit as e: + code = e.code + if code is None: + return 0 + if isinstance(code, int): + return code + return 1 + finally: + sys.argv = saved_argv + + +def _make_minimal_shadow(tmp_path, extras=None): + """Build a minimal but valid .shadow/ tree in tmp_path. + + Returns the .shadow/ Path. `extras` is a dict of {relpath: content} + appended on top of the baseline. + """ + sd = tmp_path / ".shadow" + sd.mkdir() + (sd / "_meta").mkdir() + (sd / "_meta" / "state.json").write_text( + json.dumps({ + "version": 1, + "last_update_at": "2026-04-20T16:30:00Z", + "last_update_type": "manual", + "last_commit": "deadbeef" * 5, + }), + encoding="utf-8", + ) + _write_shadow(sd, "foo.py.md", """\ + # Shadow: foo.py + + **Language**: Python | **Lines**: 10 + + ## `bar` + + - A neat bug. + _(verified, source: exploration, labels: [bug])_ + """) + for rel, content in (extras or {}).items(): + _write_shadow(sd, rel, content) + return sd + + +# =========================================================================== +# view_summary +# =========================================================================== + + +class TestViewSummary: + """In-process tests for `view_summary(shadow_dir)`.""" + + def test_basic_header_on_coupon_demo( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "Shadow Knowledge Base Summary" in out + assert "=" * 50 in out + assert "Files shadowed:" in out + assert "Symbols tracked:" in out + assert "Discoveries:" in out + assert "Preferences:" in out + assert "Cross-cutting:" in out + + def test_counts_reflect_fixture( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + # 3 source files, 33 discoveries, 3 cross-cutting (per fixture) + assert "Files shadowed: 3" in out + assert "Discoveries: 33" in out + assert "Cross-cutting: 3" in out + + def test_by_source_section_renders( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "By source:" in out + # All discoveries in the fixture are source: exploration + assert "exploration" in out + + def test_by_status_section_renders( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "By status:" in out + assert "verified" in out + + def test_by_label_section_renders( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "By label:" in out + assert "bug" in out + assert "security" in out + + def test_per_file_table_lists_files( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + # Per-file table header + assert "File" in out + assert "Symbols" in out + assert "Disc." in out + # All three fixture files appear in the table + assert "cart.py" in out + assert "inventory.py" in out + assert "test_cart.py" in out + + def test_cross_cutting_titles_section( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "Cross-cutting discoveries:" in out + assert "Coupon case normalization mismatch" in out + assert "[edge-case]" in out + + def test_state_info_section( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "Last update:" in out + assert "Last commit:" in out + # The fixture state.json says last_update_type: dream + assert "(dream)" in out + + def test_empty_shadow_dir_renders_zeros( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "Files shadowed: 0" in out + assert "Discoveries: 0" in out + # With zero discoveries there's no source/status breakdown + assert "By source:" not in out + assert "By status:" not in out + assert "By label:" not in out + # And no state info (no state.json) + assert "Last update:" not in out + + def test_corrupted_state_json_does_not_break_summary( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_minimal_shadow(tmp_path) + # Clobber state.json with junk + (sd / "_meta" / "state.json").write_text( + "{ not json", encoding="utf-8" + ) + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + # Header and counts still rendered + assert "Shadow Knowledge Base Summary" in out + assert "Files shadowed: 1" in out + # State section silently dropped + assert "Last update:" not in out + + def test_unreadable_shadow_file_does_not_break_summary( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_minimal_shadow(tmp_path) + # Create a file with invalid UTF-8 — parse_shadow_file records a + # parse_error but returns a result. view_summary should still + # render the rest of the report. + bad = sd / "bad.py.md" + bad.write_bytes(b"\xff\xfe\x00garbage\x00") + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "Shadow Knowledge Base Summary" in out + assert "Files shadowed: 2" in out + + def test_more_than_20_files_truncated( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + for i in range(25): + _write_shadow( + sd, f"file{i:02d}.py.md", + f"# Shadow: file{i:02d}.py\n\n## `sym{i}`\n\n" + f"- D\n _(verified, source: exploration)_\n", + ) + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "Files shadowed: 25" in out + assert "... and 5 more files" in out + + def test_no_cross_cutting_dir_omits_section( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_minimal_shadow(tmp_path) + # no _cross/ dir + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "Cross-cutting: 0" in out + # The titled list is suppressed when empty + assert "Cross-cutting discoveries:" not in out + + def test_summary_no_prefs(self, shadow_viewer, tmp_path, capsys): + sd = _make_minimal_shadow(tmp_path) + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "Preferences: 0" in out + + def test_summary_with_prefs(self, shadow_viewer, tmp_path, capsys): + sd = _make_minimal_shadow(tmp_path) + _write_shadow(sd, "_prefs.md", """\ + # Preferences + + - Pref one. + _(source: user)_ + - Pref two. + _(source: interaction)_ + """) + shadow_viewer.view_summary(sd) + out = capsys.readouterr().out + assert "Preferences: 2" in out + + +# =========================================================================== +# view_search +# =========================================================================== + + +class TestViewSearch: + """In-process tests for `view_search(shadow_dir, query)`.""" + + def test_finds_text_match(self, shadow_viewer, coupon_demo, capsys): + sd = coupon_demo / ".shadow" + shadow_viewer.view_search(sd, "tax") + out = capsys.readouterr().out + assert "Search: 'tax'" in out + assert "results" in out + # The "8% tax" / "Tax rate" discoveries on cart.py match + assert "cart.py" in out + + def test_finds_symbol_match(self, shadow_viewer, coupon_demo, capsys): + sd = coupon_demo / ".shadow" + shadow_viewer.view_search(sd, "COUPON_CACHE") + out = capsys.readouterr().out + assert "COUPON_CACHE" in out + assert "cart.py" in out + + def test_finds_file_name_match( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + # Searching for the literal file name pulls every discovery in + # that file (file_name_hit branch). + shadow_viewer.view_search(sd, "inventory.py") + out = capsys.readouterr().out + assert "inventory.py" in out + assert "matches" in out + + def test_case_insensitive(self, shadow_viewer, coupon_demo, capsys): + sd = coupon_demo / ".shadow" + shadow_viewer.view_search(sd, "COUPON") + upper = capsys.readouterr().out + shadow_viewer.view_search(sd, "coupon") + lower = capsys.readouterr().out + # Same number of result lines either way + assert ("results" in upper) and ("results" in lower) + # And both contain at least one match + assert "::" in upper + assert "::" in lower + + def test_no_matches_prints_no_results( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_search(sd, "zzzdoesnotexistzzz") + out = capsys.readouterr().out + assert "No results for 'zzzdoesnotexistzzz'." in out + + def test_finds_cross_cutting_title( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + # Title of _cross/coupon-case-normalization-mismatch.md is + # "Coupon case normalization mismatch" + shadow_viewer.view_search(sd, "normalization mismatch") + out = capsys.readouterr().out + assert "Cross-cutting" in out + assert "Coupon case normalization mismatch" in out + assert "Category:" in out + assert "edge-case" in out + + def test_finds_cross_cutting_by_ref( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + # Search for a ref that appears in a _cross file + shadow_viewer.view_search(sd, "apply_bulk_discount") + out = capsys.readouterr().out + assert "Cross-cutting" in out + assert "Mutation through discount pipeline" in out + + def test_finds_also_involves_match( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## `foo` + + - A discovery. + _(verified, source: exploration)_ + Also involves: `b.py::weird_symbol_zzz` + """) + shadow_viewer.view_search(sd, "weird_symbol_zzz") + out = capsys.readouterr().out + # The match flag is "also_involves" and a line shows that ref + assert "weird_symbol_zzz" in out + assert "Also involves:" in out + + def test_finds_preference_match( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_minimal_shadow(tmp_path) + _write_shadow(sd, "_prefs.md", """\ + # Preferences + + - Prefer rusty pelicans for everything. + _(source: user)_ + """) + shadow_viewer.view_search(sd, "pelican") + out = capsys.readouterr().out + assert "Preferences" in out + assert "pelican" in out.lower() + assert "[user]" in out + + def test_groups_per_file_results_by_file( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_search(sd, "coupon") + out = capsys.readouterr().out + # Per-file groups present a header like "cart.py (N matches)" + assert re.search(r"cart\.py \(\d+ matches\)", out) + + def test_results_show_status_and_source( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_search(sd, "tax") + out = capsys.readouterr().out + assert "(verified, source: exploration)" in out + + def test_total_count_in_header( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_search(sd, "coupon") + out = capsys.readouterr().out + m = re.search(r"\((\d+) results\)", out) + assert m is not None + assert int(m.group(1)) > 0 + + def test_empty_shadow_returns_no_results( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + shadow_viewer.view_search(sd, "anything") + out = capsys.readouterr().out + assert "No results for 'anything'." in out + + +# =========================================================================== +# view_prefs +# =========================================================================== + + +class TestViewPrefs: + """In-process tests for `view_prefs(shadow_dir)`.""" + + def test_missing_prefs_file(self, shadow_viewer, tmp_path, capsys): + sd = _make_shadow_root(tmp_path) + shadow_viewer.view_prefs(sd) + out = capsys.readouterr().out + assert "No preferences recorded yet." in out + + def test_empty_prefs_placeholder( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_prefs.md", """\ + # Preferences + + _No preferences recorded yet._ + """) + shadow_viewer.view_prefs(sd) + out = capsys.readouterr().out + assert "No preferences recorded yet." in out + + def test_populated_prefs_lists_count_and_sources( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_prefs.md", """\ + # Preferences + + - Use snake_case. + _(source: user)_ + - Avoid mutable default args. + _(source: interaction)_ + - Prefer fail-fast for required deps. + _(source: user)_ + """) + shadow_viewer.view_prefs(sd) + out = capsys.readouterr().out + assert "Project Preferences (3 total)" in out + assert "[user]" in out + assert "[interaction]" in out + assert "snake_case" in out + assert "fail-fast" in out + assert "mutable default" in out + + def test_against_coupon_demo_is_empty( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_prefs(sd) + out = capsys.readouterr().out + assert "No preferences recorded yet." in out + + +# =========================================================================== +# view_labels +# =========================================================================== + + +class TestViewLabels: + """In-process tests for `view_labels(shadow_dir, label_filter)`.""" + + @pytest.mark.parametrize("label", [ + "bug", "security", "performance", "feature-gap", "tech-debt", + ]) + def test_each_label_returns_results_on_fixture( + self, shadow_viewer, coupon_demo, capsys, label + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_labels(sd, label) + out = capsys.readouterr().out + assert f"label(s): {label}" in out + assert f"[{label}]" in out + # Result header always has "(N results)" + m = re.search(r"\((\d+) results\)", out) + assert m and int(m.group(1)) >= 1 + + def test_unknown_label_no_results( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_labels(sd, "zznotalabelzz") + out = capsys.readouterr().out + assert "No discoveries with label(s): zznotalabelzz" in out + + def test_multiple_labels_comma_split( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_labels(sd, "bug,security") + out = capsys.readouterr().out + assert "label(s): bug, security" in out + # Both grouped section headers present + assert "[bug]" in out + assert "[security]" in out + + def test_label_filter_is_lowercased( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_labels(sd, "BUG") + out = capsys.readouterr().out + # Filter is lowercased before matching + assert "label(s): bug" in out + assert "[bug]" in out + + def test_cross_cutting_labels_included( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_labels(sd, "bug") + out = capsys.readouterr().out + # Cross-cutting entries are prefixed with `_cross/<file>` in the + # file column. + assert "_cross/" in out + + def test_also_labeled_displayed( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + # One cart.py discovery has labels [bug, performance] + shadow_viewer.view_labels(sd, "bug") + out = capsys.readouterr().out + assert "Also labeled:" in out + + def test_empty_shadow_no_results( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + shadow_viewer.view_labels(sd, "bug") + out = capsys.readouterr().out + assert "No discoveries with label(s): bug" in out + + def test_each_result_row_has_file_and_symbol( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_labels(sd, "feature-gap") + out = capsys.readouterr().out + # `::` separator for file::symbol form + assert "::" in out + assert "(verified, source: exploration)" in out + + +# =========================================================================== +# view_recent +# =========================================================================== + + +class TestViewRecent: + """In-process tests for `view_recent(shadow_dir, count)`.""" + + def test_default_count_caps_at_10( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_recent(sd, 10) + out = capsys.readouterr().out + assert "Most Recent Discoveries (top 10)" in out + entries = [ + l for l in out.splitlines() if l.strip().startswith("[20") + ] + # Fixture has 33 per-file + 3 cross + 0 prefs > 10 + assert len(entries) == 10 + + def test_custom_count(self, shadow_viewer, coupon_demo, capsys): + sd = coupon_demo / ".shadow" + shadow_viewer.view_recent(sd, 3) + out = capsys.readouterr().out + assert "Most Recent Discoveries (top 3)" in out + entries = [ + l for l in out.splitlines() if l.strip().startswith("[20") + ] + assert len(entries) == 3 + + def test_count_larger_than_available( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_minimal_shadow(tmp_path) + shadow_viewer.view_recent(sd, 50) + out = capsys.readouterr().out + # Only 1 discovery exists in the minimal shadow + entries = [ + l for l in out.splitlines() if l.strip().startswith("[20") + ] + assert len(entries) == 1 + + def test_empty_shadow_prints_nothing( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + shadow_viewer.view_recent(sd, 10) + out = capsys.readouterr().out + assert "No discoveries found." in out + + def test_recently_modified_file_appears_first( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + # Bump test_cart.py's mtime to "now" so its discoveries should + # rank first. + target = sd / "test_cart.py.md" + now = datetime.now().timestamp() + os.utime(target, (now + 10, now + 10)) + shadow_viewer.view_recent(sd, 3) + out = capsys.readouterr().out + # First entry block should reference test_cart.py + first_entry_idx = out.find("[20") + first_block = out[first_entry_idx:first_entry_idx + 400] + assert "test_cart.py" in first_block + + def test_includes_cross_cutting_type( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + # Bump a cross file mtime so it shows up in the top + cf = sd / "_cross" / "coupon-case-normalization-mismatch.md" + now = datetime.now().timestamp() + os.utime(cf, (now + 100, now + 100)) + shadow_viewer.view_recent(sd, 5) + out = capsys.readouterr().out + assert "(cross-cutting)" in out + + def test_includes_preferences_when_present( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_minimal_shadow(tmp_path) + _write_shadow(sd, "_prefs.md", """\ + # Preferences + + - A pref we care about. + _(source: user)_ + """) + shadow_viewer.view_recent(sd, 10) + out = capsys.readouterr().out + assert "(preference)" in out + assert "A pref we care about" in out + # Preference-typed rows show `source:` not `(verified, ...)` + assert "source: user" in out + + def test_no_cross_dir_does_not_crash( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_minimal_shadow(tmp_path) + # No _cross/ created + shadow_viewer.view_recent(sd, 10) + out = capsys.readouterr().out + assert "Most Recent Discoveries" in out + + def test_entries_include_timestamp_and_kind( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + shadow_viewer.view_recent(sd, 2) + out = capsys.readouterr().out + # Entry header line format: " [YYYY-MM-DD HH:MM] (kind)" + assert re.search( + r"\[\d{4}-\d{2}-\d{2} \d{2}:\d{2}\] \((discovery|cross-cutting|preference)\)", + out, + ) + + +# =========================================================================== +# view_check_invariants +# =========================================================================== + + +class TestViewCheckInvariants: + """In-process tests for `view_check_invariants(shadow_dir)`. + + Each test builds a deliberately broken `.shadow/` tree in tmp_path + and asserts that the right violation kind is reported. + """ + + def test_clean_coupon_demo_returns_zero( + self, shadow_viewer, coupon_demo, capsys + ): + rc = shadow_viewer.view_check_invariants(coupon_demo / ".shadow") + out = capsys.readouterr().out + assert rc == 0 + assert "✓ Invariants OK" in out + + def test_empty_shadow_returns_zero( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 0 + assert "Invariants OK" in out + + def test_missing_cross_file_is_violation( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + (sd / "_cross" / "coupon-case-normalization-mismatch.md").unlink() + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "cross-ref" in out + assert "coupon-case-normalization-mismatch" in out + + def test_invalid_status_enum(self, shadow_viewer, tmp_path, capsys): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## `foo` + + - Bad status. + _(maybeverified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "enum" in out + assert "maybeverified" in out + + def test_invalid_source_enum(self, shadow_viewer, tmp_path, capsys): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## `foo` + + - Bad source. + _(verified, source: psychic)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "enum" in out + assert "psychic" in out + + def test_invalid_label(self, shadow_viewer, tmp_path, capsys): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## `foo` + + - Bad label. + _(verified, source: exploration, labels: [unicorn])_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "enum" in out + assert "unicorn" in out + + def test_heading_without_backticks( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## not_in_backticks + + - Hi. + _(verified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "heading" in out + + def test_file_level_heading_does_not_violate( + self, shadow_viewer, tmp_path, capsys + ): + """`## File-Level` is allowed without backticks.""" + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## File-Level + + - A file-level discovery. + _(verified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 0, out + assert "Invariants OK" in out + + def test_also_involves_without_backticks( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## `foo` + + - With bad anchors. + _(verified, source: exploration)_ + Also involves: b.py::bar, c.py::baz + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "anchor" in out + + def test_also_involves_missing_symbol_after_colons( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## `foo` + + - Empty sym. + _(verified, source: exploration)_ + Also involves: `b.py::` + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + # The empty-symbol form fails the strict regex (which requires + # non-empty after ::), so the file_sym_re finds zero anchors and + # we hit the "needs backtick anchors" branch instead. + assert "anchor" in out + assert "needs `file::symbol`" in out + + def test_cross_missing_category( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## `foo` + + - A discovery. + _(verified, source: exploration)_ + """) + _write_shadow(sd, "_cross/no-cat.md", """\ + # No category here + + **Refs**: + - `a.py::foo` + + **Discovery**: Something. + + _(verified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "schema" in out + assert "Category" in out + + def test_cross_invalid_category( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_cross/bad-cat.md", """\ + # Bad category here + + **Category**: bogus + **Refs**: + - `a.py::foo` + + **Discovery**: Something. + + _(verified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "enum" in out + assert "bogus" in out + + def test_cross_missing_metadata_line( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_cross/no-meta.md", """\ + # No meta here + + **Category**: pattern + **Refs**: + - `a.py::foo` + + **Discovery**: Something. + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "schema" in out + assert "missing trailing" in out + + def test_cross_missing_refs_block( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_cross/no-refs.md", """\ + # No refs here + + **Category**: pattern + + **Discovery**: Something. + + _(verified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "schema" in out + assert "Refs" in out + + def test_cross_ref_missing_symbol( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + # Note: file_sym_re demands "::" in backticks. We use a backtick + # ref with no symbol after `::`. + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## `foo` + + - hi. + _(verified, source: exploration)_ + + ## Cross-References + + - [bad-anchor](_cross/bad-anchor.md) + """) + _write_shadow(sd, "_cross/bad-anchor.md", """\ + # Bad anchor + + **Category**: pattern + **Refs**: + - `a.py::` + + **Discovery**: stuff. + + _(verified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "anchor" in out + + def test_back_pointer_missing( + self, shadow_viewer, tmp_path, capsys + ): + """_cross/x.md references a.py::foo but a.py.md has no + Cross-References section pointing back to x.md.""" + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## `foo` + + - A discovery. + _(verified, source: exploration)_ + """) + _write_shadow(sd, "_cross/orphan.md", """\ + # Orphan + + **Category**: pattern + **Refs**: + - `a.py::foo` + + **Discovery**: stuff. + + _(verified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "cross-ref" in out + assert "does not link back to" in out + + def test_back_pointer_references_nonexistent_file( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "_cross/ghost.md", """\ + # Ghost + + **Category**: pattern + **Refs**: + - `does/not/exist.py::foo` + + **Discovery**: stuff. + + _(verified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + assert "no such shadow file exists" in out + + def test_violation_line_format( + self, shadow_viewer, tmp_path, capsys + ): + """Output is grep-friendly: `path:line: kind: message`.""" + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## not_backticked + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 1 + # At least one line of form path:line: kind: msg + assert re.search(r"a\.py\.md:\d+: heading: ", out) + + def test_multiple_violations_reported( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## not_backticked + + ## `foo` + + - bad. + _(maybeverified, source: psychic, labels: [unicorn])_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + err = capsys.readouterr().err + assert rc == 1 + # heading + 3 enum violations = at least 4 lines + violation_lines = [ + l for l in out.splitlines() + if re.match(r"^[^:]+:\d+: \w+: ", l) + ] + assert len(violation_lines) >= 3 + + def test_count_summary_emitted_on_stderr( + self, shadow_viewer, tmp_path, capsys + ): + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", "## not_backticked\n") + rc = shadow_viewer.view_check_invariants(sd) + captured = capsys.readouterr() + assert rc == 1 + # Final count line goes to stderr + assert "invariant violation(s) found" in captured.err + + def test_special_headings_allowed( + self, shadow_viewer, tmp_path, capsys + ): + """`## Notes`, `## Metadata`, `## File-Level Notes` are allowed + without backticks.""" + sd = _make_shadow_root(tmp_path) + _write_shadow(sd, "a.py.md", """\ + # Shadow: a.py + + ## Notes + + Some prose. + + ## Metadata + + Some metadata. + + ## File-Level Notes + + More prose. + + ## `real_sym` + + - A discovery. + _(verified, source: exploration)_ + """) + rc = shadow_viewer.view_check_invariants(sd) + out = capsys.readouterr().out + assert rc == 0, out + + +# =========================================================================== +# main() — in-process via sys.argv (covers the dispatcher) +# =========================================================================== + + +class TestMainInProcess: + """Drive `main()` directly so coverage of the dispatch arms is captured. + + All paths use `--shadow-dir <coupon_demo/.shadow>` to avoid relying + on cwd. Where we need a true CLI smoke (e.g., to verify `--help` + output), use subprocess. + """ + + def _shadow(self, coupon_demo): + return str(coupon_demo / ".shadow") + + def test_summary_dispatch( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, ["--shadow-dir", self._shadow(coupon_demo), + "--summary"] + ) + out = capsys.readouterr().out + assert rc == 0 + assert "Shadow Knowledge Base Summary" in out + + def test_no_args_default_is_summary( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, ["--shadow-dir", self._shadow(coupon_demo)] + ) + out = capsys.readouterr().out + assert rc == 0 + assert "Shadow Knowledge Base Summary" in out + + def test_search_dispatch( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, + ["--shadow-dir", self._shadow(coupon_demo), + "--search", "coupon"], + ) + out = capsys.readouterr().out + assert rc == 0 + assert "Search: 'coupon'" in out + + def test_prefs_dispatch( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, + ["--shadow-dir", self._shadow(coupon_demo), "--prefs"], + ) + out = capsys.readouterr().out + assert rc == 0 + assert "No preferences recorded yet." in out + + def test_labels_dispatch_bug( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, + ["--shadow-dir", self._shadow(coupon_demo), + "--labels", "bug"], + ) + out = capsys.readouterr().out + assert rc == 0 + assert "label(s): bug" in out + + def test_recent_dispatch_with_n( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, + ["--shadow-dir", self._shadow(coupon_demo), + "--recent", "4"], + ) + out = capsys.readouterr().out + assert rc == 0 + assert "Most Recent Discoveries (top 4)" in out + + def test_recent_dispatch_no_n( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, + ["--shadow-dir", self._shadow(coupon_demo), "--recent"], + ) + out = capsys.readouterr().out + assert rc == 0 + # Default count is 10 + assert "Most Recent Discoveries (top 10)" in out + + def test_top_dispatch( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, + ["--shadow-dir", self._shadow(coupon_demo), + "--top", "cart.py"], + ) + out = capsys.readouterr().out + assert rc == 0 + assert "for cart.py:" in out + + def test_top_dispatch_with_labels_and_limits( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, + ["--shadow-dir", self._shadow(coupon_demo), + "--top", "cart.py", + "--top-labels", "bug", + "--top-limit", "2", + "--top-max-chars", "0"], + ) + out = capsys.readouterr().out + assert rc == 0 + bullets = [l for l in out.splitlines() if l.startswith("- [")] + assert len(bullets) <= 2 + + def test_check_invariants_clean_exits_zero( + self, shadow_viewer, coupon_demo, capsys + ): + rc = _call_main( + shadow_viewer, + ["--shadow-dir", self._shadow(coupon_demo), + "--check-invariants"], + ) + out = capsys.readouterr().out + assert rc == 0 + assert "Invariants OK" in out + + def test_check_invariants_dirty_exits_one( + self, shadow_viewer, coupon_demo, capsys + ): + sd = coupon_demo / ".shadow" + # Inject a heading violation + cart = sd / "cart.py.md" + cart.write_text( + cart.read_text(encoding="utf-8").replace( + "## `COUPON_CACHE`", "## COUPON_CACHE", 1 + ), + encoding="utf-8", + ) + rc = _call_main( + shadow_viewer, + ["--shadow-dir", str(sd), "--check-invariants"], + ) + out = capsys.readouterr().out + assert rc == 1 + assert "heading" in out + + def test_explicit_shadow_dir_missing( + self, shadow_viewer, tmp_path, capsys + ): + bogus = tmp_path / "does-not-exist" + rc = _call_main( + shadow_viewer, ["--shadow-dir", str(bogus), "--summary"] + ) + err = capsys.readouterr().err + assert rc == 1 + assert "No .shadow/ directory found" in err + + def test_auto_detect_via_chdir( + self, shadow_viewer, coupon_demo, monkeypatch, capsys + ): + """No --shadow-dir: cwd is walked up to find .shadow/.""" + monkeypatch.chdir(coupon_demo) + rc = _call_main(shadow_viewer, ["--summary"]) + out = capsys.readouterr().out + assert rc == 0 + assert "Shadow Knowledge Base Summary" in out + + def test_auto_detect_no_shadow_in_cwd( + self, shadow_viewer, tmp_path, monkeypatch, capsys + ): + monkeypatch.chdir(tmp_path) + rc = _call_main(shadow_viewer, ["--summary"]) + err = capsys.readouterr().err + assert rc == 1 + assert "No .shadow/ directory found" in err + + +# --- main(): subprocess smoke (covers true argv parsing, help, errors) ---- + + +class TestMainSubprocess: + """End-to-end CLI smoke. Subprocess output is the contract here — + these don't add coverage but they catch dispatcher / argparse regressions + the in-process tests can't (e.g. --help, mutually exclusive errors).""" + + @pytest.mark.slow + @pytest.mark.integration + def test_help_exits_zero(self, repo_root, coupon_demo): + r = _run_viewer(repo_root, coupon_demo, "--help") + assert r.returncode == 0 + # argparse prints usage and the description + assert "usage:" in r.stdout.lower() + assert "--summary" in r.stdout + assert "--search" in r.stdout + assert "--check-invariants" in r.stdout + + @pytest.mark.slow + @pytest.mark.integration + def test_unknown_flag_exits_nonzero(self, repo_root, coupon_demo): + r = _run_viewer(repo_root, coupon_demo, "--no-such-flag") + assert r.returncode != 0 + assert "unrecognized" in r.stderr or "unrecognized" in r.stdout + + @pytest.mark.slow + @pytest.mark.integration + def test_mutually_exclusive_flags(self, repo_root, coupon_demo): + """--summary and --prefs are in the same exclusive group.""" + r = _run_viewer( + repo_root, coupon_demo, "--summary", "--prefs" + ) + assert r.returncode != 0 + # argparse error mentions "not allowed with" + assert "not allowed with" in r.stderr diff --git a/tests/test_install_sh.py b/tests/test_install_sh.py new file mode 100644 index 0000000..4654b36 --- /dev/null +++ b/tests/test_install_sh.py @@ -0,0 +1,251 @@ +"""Tests for install.sh — Skills+hooks installer. + +Exercises: --help, --project requirement, --project with target, idempotency, +expected layout. +""" +import os +import subprocess +from pathlib import Path + +import json +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent +INSTALL_SCRIPT = REPO_ROOT / "install.sh" + +EXPECTED_SKILLS = [ + "shadow-frog", + "shadow-frog-init", + "shadow-frog-update", + "shadow-frog-dream", + "shadow-frog-meditate", + "shadow-frog-viewer", +] + + +def _base_env(extras: dict | None = None) -> dict: + env = { + "PATH": os.environ.get("PATH", "/usr/bin:/bin:/usr/local/bin"), + "HOME": "/nonexistent", + "GIT_CONFIG_GLOBAL": "/dev/null", + "GIT_CONFIG_SYSTEM": "/dev/null", + "LANG": "en_US.UTF-8", + } + if extras: + env.update(extras) + return env + + +def run_install(*args: str, env_extra: dict | None = None) -> subprocess.CompletedProcess: + """Run install.sh with given arguments.""" + env = _base_env(env_extra) + return subprocess.run( + ["bash", str(INSTALL_SCRIPT), *args], + capture_output=True, + text=True, + env=env, + ) + + +@pytest.mark.slow +@pytest.mark.integration +class TestInstallHelp: + def test_help_exits_zero(self): + result = run_install("--help") + assert result.returncode == 0 + assert "Usage" in result.stdout + + +@pytest.mark.slow +@pytest.mark.integration +class TestInstallProject: + """--project installs skills + hooks into .github/ of target.""" + + def test_project_install_creates_skills(self, tmp_path): + target = tmp_path / "myproject" + target.mkdir() + result = run_install("--project", str(target)) + assert result.returncode == 0 + + # Check skills copied + for skill in EXPECTED_SKILLS: + skill_dir = target / ".github" / "skills" / skill + assert skill_dir.is_dir(), f"Missing skill dir: {skill}" + assert (skill_dir / "SKILL.md").is_file(), f"Missing SKILL.md in {skill}" + + def test_project_install_creates_hooks(self, tmp_path): + target = tmp_path / "myproject" + target.mkdir() + result = run_install("--project", str(target)) + assert result.returncode == 0 + + hooks_dir = target / ".github" / "hooks" + assert (hooks_dir / "hooks.json").is_file() + assert (hooks_dir / "scripts" / "shadow-frog-pre-tool.sh").is_file() + assert (hooks_dir / "scripts" / "shadow-frog-check-init.sh").is_file() + + def test_project_install_creates_copilot_instructions(self, tmp_path): + target = tmp_path / "myproject" + target.mkdir() + result = run_install("--project", str(target)) + assert result.returncode == 0 + + instructions = target / ".github" / "copilot-instructions.md" + assert instructions.is_file() + content = instructions.read_text() + assert "shadowfrog:agent-context" in content + + def test_project_install_idempotent(self, tmp_path): + """Running twice doesn't fail and re-applies cleanly.""" + target = tmp_path / "myproject" + target.mkdir() + r1 = run_install("--project", str(target)) + r2 = run_install("--project", str(target)) + assert r1.returncode == 0 + assert r2.returncode == 0 + + # Instructions file should have only ONE shadowfrog block + instructions = target / ".github" / "copilot-instructions.md" + content = instructions.read_text() + assert content.count("<!-- shadowfrog:agent-context -->") == 1 + + def test_project_install_no_hooks_flag(self, tmp_path): + """--no-hooks skips hook installation.""" + target = tmp_path / "myproject" + target.mkdir() + result = run_install("--project", str(target), "--no-hooks") + assert result.returncode == 0 + + hooks_dir = target / ".github" / "hooks" + assert not hooks_dir.exists() + + def test_project_install_no_context_flag(self, tmp_path): + """--no-context skips copilot-instructions injection.""" + target = tmp_path / "myproject" + target.mkdir() + result = run_install("--project", str(target), "--no-context") + assert result.returncode == 0 + + instructions = target / ".github" / "copilot-instructions.md" + assert not instructions.exists() + + +@pytest.mark.slow +@pytest.mark.integration +class TestProjectRequired: + """--project is mandatory; ShadowFrog is never installed globally.""" + + def test_no_project_fails(self, tmp_path): + """Running with no --project must fail loudly (no global install).""" + fake_home = tmp_path / "home" + (fake_home / ".copilot").mkdir(parents=True) + result = run_install(env_extra={"HOME": str(fake_home)}) + assert result.returncode != 0 + assert "--project" in (result.stdout + result.stderr) + + def test_no_project_claude_fails(self, tmp_path): + """--agent claude with no --project must also fail.""" + fake_home = tmp_path / "home" + (fake_home / ".claude").mkdir(parents=True) + result = run_install("--agent", "claude", env_extra={"HOME": str(fake_home)}) + assert result.returncode != 0 + assert "--project" in (result.stdout + result.stderr) + + +@pytest.mark.slow +@pytest.mark.integration +class TestInvalidAgent: + def test_invalid_agent_value_fails(self, tmp_path): + target = tmp_path / "p" + target.mkdir() + result = run_install("--agent", "vscode", "--project", str(target)) + assert result.returncode != 0 + assert "must be 'copilot' or 'claude'" in (result.stdout + result.stderr) + + +@pytest.mark.slow +@pytest.mark.integration +class TestClaudeProjectInstall: + """--agent claude --project installs into .claude/ + CLAUDE.md.""" + + def test_skills_go_to_claude_skills(self, tmp_path): + target = tmp_path / "proj" + target.mkdir() + result = run_install("--agent", "claude", "--project", str(target)) + assert result.returncode == 0, result.stderr + for skill in EXPECTED_SKILLS: + skill_dir = target / ".claude" / "skills" / skill + assert skill_dir.is_dir(), f"Missing claude skill dir: {skill}" + assert (skill_dir / "SKILL.md").is_file() + # Copilot tree must NOT be created for a claude install + assert not (target / ".github").exists() + + def test_hooks_create_settings_and_scripts(self, tmp_path): + target = tmp_path / "proj" + target.mkdir() + result = run_install("--agent", "claude", "--project", str(target)) + assert result.returncode == 0, result.stderr + + settings = target / ".claude" / "settings.json" + assert settings.is_file() + data = json.loads(settings.read_text()) + assert "SessionStart" in data["hooks"] + assert "PreToolUse" in data["hooks"] + + scripts = target / ".claude" / "hooks" / "scripts" + assert (scripts / "shadow-frog-check-init.sh").is_file() + assert (scripts / "shadow-frog-pre-tool.sh").is_file() + + def test_context_injected_into_claude_md(self, tmp_path): + target = tmp_path / "proj" + target.mkdir() + result = run_install("--agent", "claude", "--project", str(target)) + assert result.returncode == 0, result.stderr + claude_md = target / "CLAUDE.md" + assert claude_md.is_file() + assert "shadowfrog:agent-context" in claude_md.read_text() + + def test_settings_merge_preserves_existing_hooks(self, tmp_path): + target = tmp_path / "proj" + target.mkdir() + claude_dir = target / ".claude" + claude_dir.mkdir() + existing = { + "hooks": { + "PreToolUse": [ + {"matcher": "Bash", "hooks": [ + {"type": "command", "command": "/usr/local/bin/my-guard.sh"} + ]} + ] + }, + "model": "claude-opus", + } + (claude_dir / "settings.json").write_text(json.dumps(existing)) + + result = run_install("--agent", "claude", "--project", str(target)) + assert result.returncode == 0, result.stderr + + data = json.loads((claude_dir / "settings.json").read_text()) + # Pre-existing unrelated key preserved + assert data["model"] == "claude-opus" + # User's own hook preserved + cmds = [h["command"] for g in data["hooks"]["PreToolUse"] for h in g["hooks"]] + assert "/usr/local/bin/my-guard.sh" in cmds + # Ours added + assert any("shadow-frog-pre-tool.sh" in c for c in cmds) + + def test_idempotent(self, tmp_path): + target = tmp_path / "proj" + target.mkdir() + run_install("--agent", "claude", "--project", str(target)) + r2 = run_install("--agent", "claude", "--project", str(target)) + assert r2.returncode == 0, r2.stderr + + # No duplicate context block + claude_md = target / "CLAUDE.md" + assert claude_md.read_text().count("<!-- shadowfrog:agent-context -->") == 1 + + # No duplicate hook handler + data = json.loads((target / ".claude" / "settings.json").read_text()) + cmds = [h["command"] for g in data["hooks"]["PreToolUse"] for h in g["hooks"]] + assert sum("shadow-frog-pre-tool.sh" in c for c in cmds) == 1 diff --git a/tests/test_smoke.py b/tests/test_smoke.py new file mode 100644 index 0000000..d35b3ee --- /dev/null +++ b/tests/test_smoke.py @@ -0,0 +1,60 @@ +"""Smoke test verifying the shared test infrastructure works. + +If this test fails, no other test in the suite will be reliable — +fix the conftest fixtures first. +""" +import subprocess + + +def test_repo_root_resolves(repo_root): + assert (repo_root / "claude.md").is_file() + assert (repo_root / "skills").is_dir() + + +def test_all_script_fixtures_load( + shadow_init, shadow_viewer, dream_reconcile, dream_validate, + dream_coverage, dream_lineage, meditate_repair, +): + for mod, expected_attr in [ + (shadow_init, "main"), + (shadow_viewer, "main"), + (dream_reconcile, "main"), + (dream_validate, "main"), + (dream_coverage, "main"), + (dream_lineage, "parse_args"), + (meditate_repair, "main"), + ]: + assert hasattr(mod, expected_attr), \ + f"{mod.__name__} missing expected attribute {expected_attr!r}" + + +def test_coupon_demo_src_intact(coupon_demo_src): + assert (coupon_demo_src / ".shadow" / "_index.md").is_file() + assert (coupon_demo_src / "cart.py").is_file() + + +def test_coupon_demo_copy_is_independent(coupon_demo, coupon_demo_src): + assert coupon_demo != coupon_demo_src + assert (coupon_demo / ".shadow" / "_index.md").is_file() + # Mutate the copy and confirm the source is untouched. + (coupon_demo / "cart.py").write_text("# mutated by test\n") + assert (coupon_demo_src / "cart.py").read_text() != "# mutated by test\n" + + +def test_coupon_demo_is_git_repo(coupon_demo): + out = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=coupon_demo, + capture_output=True, text=True, check=True, + ) + assert len(out.stdout.strip()) == 40 + + +def test_tmp_git_repo_is_empty(tmp_git_repo): + assert tmp_git_repo.is_dir() + assert (tmp_git_repo / ".git").is_dir() + # No commits yet + result = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=tmp_git_repo, + capture_output=True, text=True, + ) + assert result.returncode != 0 # HEAD doesn't exist