From 3108ec9ea15fec71f7e371eaa2e58b9e59133fd7 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Sat, 3 Oct 2026 15:56:41 -0400 Subject: [PATCH 1/4] perf(ci): run each test lane only on the suites the change selects `scope` now plans the test lanes once with scripts/plan-test-lanes.sh: it runs the affected-suite selector over the diff and gives each suite to the lane of its ecosystem. test-bash gets one to six legs packed from measured suite-seconds (scripts/suite-seconds.txt), new test-python and test-node jobs run the selected Python modules and the Node packages the change reaches, and each leg installs only the optional toolchains its suites need. A lane with no work skips as a job; ci-status counts that skip as success only where scope succeeded and its run_ row says false, which scripts/check-docs-only-gate.sh now pins (property 12). test-windows.yml gets a Linux scope-windows job that runs the same planner, and each Windows step runs only when the change selects its suite (scripts/test-windows-plan.txt); a schedule run at 09:17 and 16:17 UTC runs every step. selection-audit.sh reads the new unmapped fallback (the file's language corpus) from the selector itself. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/requirements-ci-animation.txt | 2 +- .github/workflows/ci.yml | 706 +++++++++++-------------- .github/workflows/selection-audit.yml | 4 +- .github/workflows/test-windows.yml | 172 ++++-- docs/ci-runner-routing.md | 44 +- scripts/affected-tests-no-suite.txt | 5 +- scripts/affected-tests.test.sh | 21 +- scripts/check-docs-only-gate.sh | 95 +++- scripts/check-docs-only-gate.test.sh | 68 ++- scripts/ci-ordered-skips.test.sh | 86 +++ scripts/plan-test-lanes.sh | 391 ++++++++++++++ scripts/plan-test-lanes.test.sh | 236 +++++++++ scripts/selection-audit.sh | 34 +- scripts/selection-audit.test.sh | 40 +- scripts/suite-seconds.txt | 734 ++++++++++++++++++++++++++ scripts/test-windows-plan.txt | 33 ++ 16 files changed, 2163 insertions(+), 508 deletions(-) create mode 100755 scripts/ci-ordered-skips.test.sh create mode 100755 scripts/plan-test-lanes.sh create mode 100755 scripts/plan-test-lanes.test.sh create mode 100644 scripts/suite-seconds.txt create mode 100644 scripts/test-windows-plan.txt diff --git a/.github/requirements-ci-animation.txt b/.github/requirements-ci-animation.txt index 762aace79b..61831a0d85 100644 --- a/.github/requirements-ci-animation.txt +++ b/.github/requirements-ci-animation.txt @@ -1,6 +1,6 @@ # CI-only Linux x64 cp314 wheel pins for the animation regression suites # (plugins/animation/scripts/test_inkstats.py, test_woodcut_marks.py), so they -# RUN in test-bash instead of skipping green. The versions must equal +# RUN in test-python instead of skipping green. The versions must equal # plugins/animation/requirements.in, so bump the two together. # Kept apart from requirements-ci.txt so the jobs that read only that file do # not install these wheels. Regenerate with: diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 07827b29ee..94d6b8505d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -114,20 +114,22 @@ concurrency: # says what a job does: `lint` runs linters over source, `check` holds a # repository contract across files, `test` executes code. The domain says over # what: `repo`, `shell`, `plugins`, `skills`, or the ecosystem a test job runs -# (`bash`). One resolver, `scope`, five working jobs, `lint-repo`, -# `lint-shell`, `check-plugins`, `check-skills` and `test-bash`, and one -# aggregate, `ci-status`, whose name, like this workflow's and this file's, is -# what the org ci-gate ruleset and its verifiers read. No ordinals, no tool -# name for a multi-tool job, no runner OS for a portable lane, and no job name -# another workflow here also uses. A gate step keeps its `id`, its aggregator -# key, in whichever job it runs. +# (`bash`, `python`, `node`). One resolver, `scope`, seven working jobs, +# `lint-repo`, `lint-shell`, `check-plugins`, `check-skills`, `test-bash`, +# `test-python` and `test-node`, and one aggregate, `ci-status`, whose name, +# like this workflow's and this file's, is what the org ci-gate ruleset and its +# verifiers read. No ordinals, no tool name for a multi-tool job, no runner OS +# for a portable lane, and no job name another workflow here also uses. A gate +# step keeps its `id`, its aggregator key, in whichever job it runs. # # A DOMAIN IS A JOB ONLY WHEN THAT SHORTENS THE CRITICAL PATH BY MORE THAN A -# JOB COSTS: about 15 s of fixed setup bare, about 40 s with a toolchain. A -# smaller domain rides a neighbour as steps, which is why the Node sub-projects -# and the disk-hygiene module are steps of `test-bash`. The informational -# Windows lane runs in .github/workflows/test-windows.yml: nothing gates on -# it, and in this file it held the run open past its own required lanes. +# JOB COSTS: about 15 s of fixed setup bare, about 40 s with a toolchain. The +# three test lanes are jobs for a different reason: each runs only when the +# change selects suites of its ecosystem, and skips otherwise, so a change to +# shell code starts no Python or Node runner and a C# or docs change starts +# none. The informational Windows lane runs in +# .github/workflows/test-windows.yml: nothing gates on it, and in this file it +# held the run open past its own required lanes. # # EVERY JOB'S TIMEOUT IS ceil(max(1.5 x p95, 1.1 x max) / 60) MINUTES, at least # 3 and never above 16, from the job's wall over 117 pull-request and 100 push @@ -135,13 +137,17 @@ concurrency: # (the case a checker change and the scheduled run pay). Until a job has runs # of its own, its wall is the sum of the measured seconds of the steps it # holds, plus the setup of the job whose toolchain it copies. Job, p95 / max -# seconds, minutes: scope 24 / 35 plus the selector (up to 21 s measured), 3; +# seconds, minutes: scope 24 / 35 plus the planner (up to 21 s measured), 3; # lint-repo 244 / 265, 7, sized for its whole-tree load (the push runs); # lint-shell 345 / 385 with a whole-repository ShellCheck, 9; check-plugins # 110 / 120, 3; check-skills at most 585 / 645, the wall of a measured job -# that ran all of its steps and more, 15; test-bash per leg at most 357 / 593, -# the same bound, 11; ci-status 9 / 42, 3. A job that outgrows its timeout is -# made faster or split, not given a longer one. +# that ran all of its steps and more, 15; test-bash per leg about 230 on the +# whole tree (171 s of suites estimated from scripts/suite-seconds.txt, a 158 s +# suite among them, plus about 55 s of setup), 9; test-python per leg about +# 260, the 199 s planning server suite plus setup, 7; test-node about 50 (31 s +# of package chains measured on a warm npm cache), sized for a cold install, 5; +# ci-status 9 / 42, 3. A job that outgrows its timeout is made faster or split, +# not given a longer one. jobs: # Single resolution of the diff scope. Every gated step reads a # `needs.scope.outputs.*` row instead of re-running the detector, so the @@ -208,18 +214,27 @@ jobs: outputs: run_full: ${{ steps.detect.outputs.docs_only != 'true' }} run_tests: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true }} - run_node: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['node'] != 'false' }} - run_python: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['python'] != 'false' }} + # Whether each test lane has work, from the plan below. A lane whose row + # is 'false' skips as a job, and `ci-status` passes that skip only + # against this row. + run_bash: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.bash != 'false' }} + run_python: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.python != 'false' }} + run_node: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.node != 'false' }} run_workflows: ${{ steps.detect.outputs.docs_only != 'true' && fromJSON(steps.match.outputs.results || '{}')['workflows'] != 'false' }} run_skill_checker: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['skill_checker'] != 'false' }} run_manifests: ${{ steps.detect.outputs.docs_only != 'true' && fromJSON(steps.match.outputs.results || '{}')['manifests'] != 'false' }} # VALUES, not gates: the ref every diff-scoped step diffs against (empty - # for the whole tree), read only as an env entry; the test-bash matrix - # as a JSON list of leg numbers; and what each leg installs, a JSON map - # from leg number to the optional toolchains its slice needs. + # for the whole tree), read only as an env entry; and the plan of each + # test lane: its matrix as a JSON list of leg numbers, the suites of + # each leg, and the optional toolchains each leg installs. lane_base: ${{ steps.base.outputs.ref }} - test_legs: ${{ steps.legs.outputs.legs }} - test_needs: ${{ steps.legs.outputs.needs }} + bash_legs: ${{ steps.plan.outputs.bash_legs }} + bash_plan: ${{ steps.plan.outputs.bash_plan }} + bash_needs: ${{ steps.plan.outputs.bash_needs }} + python_legs: ${{ steps.plan.outputs.python_legs }} + python_plan: ${{ steps.plan.outputs.python_plan }} + python_needs: ${{ steps.plan.outputs.python_needs }} + node_packages: ${{ steps.plan.outputs.node_packages }} steps: # A FULL RUN MARKS ITS SHA IN FLIGHT BEFORE ANYTHING ELSE. A contract-only # run on this SHA reads the newest bot-written `ci-lanes` status once and @@ -259,8 +274,8 @@ jobs: # away while pending, or a red one, leaves its commits in the next range, # so a break stays in every later range until a run passes. A push with no # such run in the last 50, or whose range touches the shared test - # machinery (this workflow, the actions it calls, the suite runner and - # selector, scripts/lib/, the toolchain pins), and every schedule or + # machinery (this workflow, the actions it calls, the suite runner, the + # selector and planner, scripts/lib/, the toolchain pins), and every schedule or # dispatch run, get no base: the whole tree. - name: Resolve the diff base id: base @@ -295,7 +310,7 @@ jobs: [ -n "$base" ] || whole "No green push run among the last 50 is an ancestor of $GITHUB_SHA" infra=$(git diff --name-only "$base" HEAD -- .github/workflows/ci.yml .github/actions \ '.github/requirements-ci*.txt' 'scripts/run-plugin-tests*' 'scripts/affected-tests*' \ - 'scripts/run-outside-node-suites*' 'scripts/outside-node-*.txt' scripts/lib \ + 'scripts/plan-test-lanes*' 'scripts/run-outside-node-suites*' 'scripts/outside-node-*.txt' scripts/lib \ .shellcheckrc .node-version .python-version package.json package-lock.json) || whole "git diff $base failed" [ -z "$infra" ] || whole "The range since $base touches the shared test machinery: ${infra//$'\n'/ }" @@ -313,11 +328,11 @@ jobs: # Every group repeats the toolchain and configuration paths on purpose: a # pull request touching only a lockfile, only a workflow file or only the # ShellCheck configuration changes what every lane would find, so it must - # run every lane rather than none. `shell`, `docs` and `powershell` feed - # no output today: `shell` and `powershell` are, with `python`, the + # run every lane rather than none. `shell`, `python`, `docs` and + # `powershell` feed no output: `shell`, `python` and `powershell` are the # groups the path filters of .github/workflows/test-windows.yml repeat, - # and the three unread groups exist so the group set matches the program - # plan and a consumer can be added without touching the resolver. + # and the unread groups exist so a consumer can be added without touching + # the resolver. The test lanes read the plan below, not a group. # Patterns are ROOT-ANCHORED gitignore rules, and the action rejects `!`, # `?` and `+` outright. # @@ -350,27 +365,6 @@ jobs: **/*.bash plugins/docs-naming/skills/generate-file-name-gate/** .claude/docs-naming.json - node: - .github/** - scripts/** - package.json - package-lock.json - pyproject.toml - uv.lock - requirements*.txt - .node-version - .python-version - .shellcheckrc - **/package.json - **/package-lock.json - **/*.js - **/*.cjs - **/*.mjs - **/*.ts - plugins/miro/** - plugins/ai-briefing/skills/generate/** - plugins/knowledge/skills/video-digest/** - plugins/knowledge/skills/course-digest/** python: .github/** scripts/** @@ -434,67 +428,32 @@ jobs: plugins/*/.claude-plugin/** scripts/check-manifest-duplicate-keys* .python-version - # THE test-bash FAN-OUT IS SIZED FROM THE SELECTION it will run: one leg - # per 25 suites, one to four. Every leg derives the same selection again - # and keeps its shard, so this count only sizes the matrix. Four whenever - # there is nothing to count: the whole tree, an UNMAPPED file (the legs - # fall back to the full corpus), or a selector that failed here. One on a - # docs-only diff, where every leg reports not applicable. Skipped on a - # draft, where test-bash does not run. Never fails this job: a wrong - # size costs runners, not coverage. - - name: Plan the test-bash fan-out - id: legs - if: github.event.pull_request.draft != true + # THE TEST LANES ARE PLANNED HERE, ONCE. scripts/plan-test-lanes.sh runs + # the affected-suite selector over the diff and gives each test lane its + # suites, packed into legs by suite-seconds, and each leg the optional + # toolchains its suites need; its header states every rule. A lane the + # plan gives nothing skips as a job. No diff base plans the whole tree. + # An UNMAPPED file adds the whole corpus of its language and is recorded + # as a warning, so its rate is countable per run. A planner that fails + # fails this job and so every lane: there is no plan to run, and a + # guessed one could leave a selected suite unrun. Skipped on a draft and + # on a docs-only diff, where no test lane runs. + - name: Plan the test lanes + id: plan + if: github.event.pull_request.draft != true && steps.detect.outputs.docs_only != 'true' env: DIFF_BASE: ${{ steps.base.outputs.ref }} - DOCS_ONLY: ${{ steps.detect.outputs.docs_only }} run: | - legs=4 - if [ -n "$DIFF_BASE" ] && [ "$DOCS_ONLY" = true ]; then - legs=1 - elif [ -n "$DIFF_BASE" ]; then - if scripts/affected-tests.sh --with-always --base "$DIFF_BASE" >"$RUNNER_TEMP/selection.txt" 2>/dev/null; then - suites=$(grep -c . "$RUNNER_TEMP/selection.txt" || true) - legs=$(((suites + 24) / 25)) - legs=$((legs < 1 ? 1 : legs > 4 ? 4 : legs)) - echo "The selection holds $suites suite(s): $legs leg(s)." - else - echo "No selection to count (an UNMAPPED file, or the selector failed): four legs." - fi + if ! scripts/plan-test-lanes.sh ${DIFF_BASE:+--base "$DIFF_BASE"} >"$RUNNER_TEMP/plan.txt"; then + echo "::error::scripts/plan-test-lanes.sh failed; no test lane can run without its plan." + exit 1 fi - list=0 - for ((i = 1; i < legs; i++)); do list+=",$i"; done - echo "legs=[$list]" >>"$GITHUB_OUTPUT" - # WHAT EACH LEG INSTALLS. A leg keeps every legs-th suite of the sorted - # selection, the same partition affected-tests.sh --shard takes, so - # its slice is known here: the animation wheels, the inventory's - # parser packages and the DuckDB CLI are installed only on a leg whose - # slice has a suite that needs them. With no counted selection every - # leg installs everything ("all"), and so does a leg the map does not - # name. An under-install is loud, not quiet: those suites fail rather - # than skip without their dependency (ANIMATION_REQUIRE_DEPS, - # INVENTORY_REQUIRE_ACORN). - needs='{}' - if [ -n "${suites:-}" ]; then - needs=$(awk -v legs="$legs" ' - { leg = (NR - 1) % legs } - /^plugins\/animation\// { has[leg, "animation"] = 1 } - /^plugins\/harness-ops\/skills\/inventory\// { has[leg, "inventory"] = 1 } - /^plugins\/harness-ops\// { has[leg, "duckdb"] = 1 } - END { - printf "{" - for (l = 0; l < legs; l++) { - s = "" - if ((l, "animation") in has) s = s "animation " - if ((l, "inventory") in has) s = s "inventory " - if ((l, "duckdb") in has) s = s "duckdb " - printf "%s\"%d\":\"%s\"", (l ? "," : ""), l, s - } - printf "}" - }' "$RUNNER_TEMP/selection.txt") + cat "$RUNNER_TEMP/plan.txt" >>"$GITHUB_OUTPUT" + unmapped=$(sed -n 's/^unmapped=//p' "$RUNNER_TEMP/plan.txt") + if [ "${unmapped:-0}" != 0 ]; then + echo "::warning::$unmapped changed file(s) map to no suite; the plan runs the whole corpus of each one's language." + printf 'affected-tests fallback: %s unmapped file(s)\n' "$unmapped" >>"$GITHUB_STEP_SUMMARY" fi - echo "needs=$needs" >>"$GITHUB_OUTPUT" - echo "Per-leg installs: $needs" # Public repository: every lane runs on GitHub-hosted ubuntu-24.04 (free for # public repositories); the local-runner selector is not permitted here. @@ -1544,8 +1503,8 @@ jobs: run: python3 scripts/check-manifest-duplicate-keys.py # The root npm install, for the steps below that read it: htmlhint and the - # claude CLI from node_modules/.bin, and node for the launcher resolver, - # the prerequisite gate and the counter ceilings. Gated like those steps, + # claude CLI from node_modules/.bin, and node for the prerequisite gate + # and the counter ceilings. Gated like those steps, # so a docs-only diff does not pay for it. - name: Set up Node if: needs.scope.outputs.run_tests == 'true' @@ -1632,10 +1591,6 @@ jobs: fi scripts/check-standards-contract-bump.sh "origin/$BASE_REF" - - name: Test the exec-form bash launcher resolver - if: needs.scope.outputs.run_tests == 'true' - run: node --test lib/exec-bash.resolver.test.mjs - - name: Check for unregistered or drifted cross-plugin source clusters if: needs.scope.outputs.run_tests == 'true' run: scripts/check-cross-plugin-source-drift.sh --check @@ -2012,68 +1967,52 @@ jobs: portability-lint=${{ steps.portability.outcome }} run: scripts/aggregate-hygiene-results.sh - # THE CONTRACT SUITES: the affected selection of the plugin contract corpus, - # and two small domains that ride here as steps, the four Node sub-projects - # and the Python disk-hygiene module. Each of those two is under 30 s of work - # per run (measured: 15 to 32 s and 8 to 14 s), and a job of its own would - # cost more in setup (about 40 s with a toolchain) than it takes off any leg. - # One checkout, one Node toolchain, one Python toolchain, one ShellCheck - # install per leg. + # THE TEST LANES RUN WHAT THE SELECTION SAYS. `scope` plans them once + # (scripts/plan-test-lanes.sh): the shell suites the change selects go to + # `test-bash`, its Python suites to `test-python`, and the Node packages they + # reach to `test-node`; a Node suite with a sibling .test.sh runs through + # that wrapper in `test-bash`, and a Pester suite in test-windows.yml. The + # whole tree (a schedule, a dispatch, a push with no usable base, or a change + # to this workflow or its actions) runs every suite of every lane. + # + # ORDERED SKIPS. Each test lane carries the contract-only and draft gates and + # one more term, its own `run_` row of `scope`: a lane the plan gives no + # work skips as a job instead of paying a runner and a queue slot to report + # nothing. `ci-status` passes such a skip only where `scope` succeeded and + # its row for that lane says 'false'; any other skipped lane is red + # (scripts/check-docs-only-gate.sh pins both halves). + # + # ONE TO SIX LEGS, SIZED FROM THE SELECTION. The plan packs the selected + # suites into legs of about 120 suite-seconds each (scripts/suite-seconds.txt), + # longest first onto the least-loaded leg, at most six, and six on the whole + # tree. A leg costs about 45 s of setup before it runs a suite, so the count + # follows the work rather than a fixed fan-out. The job key stays one entry in + # `ci-status.needs` at any size. `fail-fast: false` because a leg is a slice + # of one selection: cancelling the others on the first failure would hide + # every other failure behind whichever leg lost the race. # - # The job skips only on a contract-only event or a draft (see `lint-repo`); - # otherwise only the inner steps are gated, and a docs-only diff reports an - # honest evaluated-and-not-applicable success through the reporter at the - # foot of the job. + # Three suites at a time per leg, not four (scripts/run-plugin-tests.sh + # --jobs 3). The runner has 4 vCPUs and the suites are spawn-bound, so four + # looked like the shape that pays, but at four three separate suites failed + # across two runs by producing EMPTY output from an external command on a + # path with no clock in it. See #3694. The suites that assert wall-clock + # ceilings run one at a time regardless, from + # scripts/run-plugin-tests-serial.txt, and the plan weighs them three times. test-bash: needs: [scope] - # The contract-only gate, verbatim, and the draft gate (see `ci-status`). - if: ${{ !(github.event.pull_request.head.repo.full_name == github.repository && (contains(fromJSON('["labeled","unlabeled"]'), github.event.action) || (github.event.action == 'edited' && !github.event.changes.base))) && github.event.pull_request.draft != true }} + # The contract-only gate, verbatim, the draft gate (see `ci-status`), and + # this lane's ordered skip. + if: ${{ !(github.event.pull_request.head.repo.full_name == github.repository && (contains(fromJSON('["labeled","unlabeled"]'), github.event.action) || (github.event.action == 'edited' && !github.event.changes.base))) && github.event.pull_request.draft != true && needs.scope.outputs.run_bash == 'true' }} runs-on: ubuntu-24.04 - # A whole-tree leg runs its share of the shell corpus and then of the - # Python corpus, whose planning surface server suite alone takes about - # 200 s; the leg that also runs the Node sub-projects ran past 11 minutes. - timeout-minutes: 16 - # ONE TO FOUR LEGS, SIZED FROM THE SELECTION. The job key stays one entry in - # `ci-status.needs` at any size; the contract suites are partitioned across - # up to four runners instead of executed by one: the affected selection on a - # pull request or a push, the full corpus on the whole tree (#3705). The - # `scope` job sizes the fan-out at one leg per 25 selected suites, up to - # four, and four whenever it has no selection to count (the whole tree, an - # UNMAPPED file): a leg costs about 45 s of setup before it runs a suite, so - # a 10-suite selection on four runners paid for three idle ones. The - # dominant step was measured at 435 s p50 running 146 to 165 suites strictly - # sequentially, so more runners was the first lever. The second is - # `--jobs 3` on the selector path below: the selector hands its selection - # to run-plugin-tests.sh, which is what the whole-tree path and the - # UNMAPPED fallback already did, so both paths run three suites at a time on - # the same runner through the same worker and the same - # scripts/run-plugin-tests-serial.txt. Three, never four (#3694). - # - # `fail-fast: false` because a leg is a slice of one test corpus: cancelling - # the other three on the first failure would hide every other failure in the - # same change set behind whichever leg lost the race. - # - # The shard spec is read from `strategy.job-index` and `strategy.job-total`, - # not from `matrix.leg`. The matrix key exists to SIZE the fan-out; the - # runtime pair is what the shard arithmetic uses, so the numerator and the - # denominator come from the same place and cannot drift from each other or - # from the matrix that produced them. - # - # EVERY STEP AFTER THE CONTRACT SUITES IS ASSIGNED TO ONE LEG, through a - # `[ "$LEG" = $(( % LEGS)) ] || exit 0` first line rather than a step - # condition: scripts/check-docs-only-gate.sh accepts exactly one condition - # shape on a gated step, and a step that exits 0 reports success, never - # `skipped`. The two groups (2: the disk-hygiene module, 3: the four Node - # sub-projects) are whole on one leg because their steps depend on each - # other in order, and the modulo folds them onto the legs there are, so - # every step runs EXACTLY ONCE per run at any size. + timeout-minutes: 9 strategy: fail-fast: false matrix: - leg: ${{ fromJSON(needs.scope.outputs.test_legs || '[0,1,2,3]') }} + leg: ${{ fromJSON(needs.scope.outputs.bash_legs || '[0]') }} env: LEG: ${{ strategy.job-index }} - LEGS: ${{ strategy.job-total }} + PLAN: ${{ needs.scope.outputs.bash_plan }} + LEG_PLAN: ${{ needs.scope.outputs.bash_needs }} steps: - name: Check out uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -2081,24 +2020,18 @@ jobs: persist-credentials: false - name: Fetch base uses: ./.github/actions/checkout-with-base # zizmor: ignore[self-repository] - # One setup-node for the root package and all four sub-projects, keyed on - # every lockfile any of them installs from, so a change to any one of - # them invalidates the cache the others share. + # Node, Python, ShellCheck, shfmt, Biome and Ruff are declared here so + # hosted and local runs exercise the same contract suites instead of + # inheriting different image tool inventories. - name: Set up Node - if: needs.scope.outputs.run_tests == 'true' uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 with: node-version-file: .node-version cache: npm cache-dependency-path: | package-lock.json - plugins/miro/server/package-lock.json - plugins/ai-briefing/skills/generate/output/build/package-lock.json - plugins/knowledge/skills/video-digest/extraction/package-lock.json - plugins/knowledge/skills/course-digest/extraction/package-lock.json plugins/harness-ops/skills/inventory/scripts/js/package-lock.json - name: Set up Python - if: needs.scope.outputs.run_tests == 'true' uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version-file: .python-version @@ -2107,36 +2040,27 @@ jobs: .github/requirements-ci.txt .github/requirements-ci-animation.txt - name: Install and verify ShellCheck toolchain - if: needs.scope.outputs.run_tests == 'true' # This canonical action also makes the exact ShellCheck version - # available to the later bash-format contract tests in this same job. - # It verifies the install on one small file that sources nothing; the + # available to the bash-format contract suites in this same job. It + # verifies the install on one small file that sources nothing; the # ShellCheck gate in `lint-shell` is what lints the repository. uses: melodic-software/ci-workflows/.github/actions/shellcheck@a267a27f7a321452267e20c82d57699b6c057cb0 # v0.30.2 with: files: scripts/aggregate-hygiene-results.sh rcfile: .shellcheckrc - # Contract tests gate independently of the CLI validate step in - # `check-plugins`: a validate-infra problem (e.g. the CLI needing auth) - # can never mask a real contract-test failure. The eval-coverage gate's - # suite reads `shfmt --to-json` and exits 2 when shfmt is missing, so this - # job installs the same v3.14.1 pin as check-skills. Node, Python, - # ShellCheck, Biome, and Ruff are declared here so hosted and local runs - # exercise the same contract suites instead of inheriting different image - # tool inventories. - # - # The shfmt and DuckDB downloads are cached by version (four red main runs - # in two weeks were these downloads failing), and a miss retries transient - # failures. `--no-progress-meter` instead of `-s` keeps curl's "Will - # retry" warning in the log. The SHA-256 check runs on a hit too. + # The eval-coverage gate's suite reads `shfmt --to-json` and exits 2 when + # shfmt is missing, so this job installs the same v3.14.1 pin as + # check-skills. The shfmt and DuckDB downloads are cached by version (four + # red main runs in two weeks were these downloads failing), and a miss + # retries transient failures. `--no-progress-meter` instead of `-s` keeps + # curl's "Will retry" warning in the log. The SHA-256 check runs on a hit + # too. - name: Restore the pinned tool downloads - if: needs.scope.outputs.run_tests == 'true' uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ~/.cache/ci-tools key: ci-tools-${{ runner.os }}-shfmt-v3.14.1-duckdb-v1.5.5 - name: Install pinned shfmt - if: needs.scope.outputs.run_tests == 'true' env: SHFMT_VERSION: v3.14.1 SHFMT_SHA256: 76e77641faa025814b77f153b29796b8e6fa2fca03e0c76a691608b86c7ea7bf @@ -2153,12 +2077,10 @@ jobs: shfmt --version test "$(shfmt --version)" = "$SHFMT_VERSION" # The animation wheels and the inventory's parser packages go only on a - # leg whose slice needs them, per the `scope` job's plan (`test_needs`): - # a leg it does not name, or no plan at all, installs everything. + # leg whose suites need them, per the plan (`LEG_PLAN`). An under-install + # is loud, not quiet: those suites fail rather than skip without their + # dependency (ANIMATION_REQUIRE_DEPS, INVENTORY_REQUIRE_ACORN). - name: Install locked plugin test toolchains - if: needs.scope.outputs.run_tests == 'true' - env: - LEG_PLAN: ${{ needs.scope.outputs.test_needs }} run: | # Dependabot bumps the animation wheels here (its /.github pip entry) and the plugin's # lock (its /plugins/animation entry) in separate pull requests, so a one-sided bump @@ -2166,11 +2088,12 @@ jobs: diff <(grep -oE '^[A-Za-z0-9._-]+==[^ ]+' plugins/animation/requirements.in | sort) \ <(grep -oE '^[A-Za-z0-9._-]+==[^ ]+' .github/requirements-ci-animation.txt | sort) \ || { echo "::error::.github/requirements-ci-animation.txt and plugins/animation/requirements.in pin different versions; bump the two together"; exit 1; } - leg_needs=$(jq -r --arg leg "$LEG" '.[$leg] // "all"' <<<"${LEG_PLAN:-"{}"}" 2>/dev/null) || leg_needs=all + leg_needs=$(jq -er --arg leg "$LEG" '.[$leg]' <<<"$LEG_PLAN") || + { echo "::error::the plan names no toolchains for leg $LEG"; exit 1; } echo "This leg installs: ${leg_needs:-the base toolchains only}" npm ci requirements=(--requirement .github/requirements-ci.txt) - if [[ "$leg_needs" == all || "$leg_needs" == *animation* ]]; then + if [[ "$leg_needs" == *animation* ]]; then requirements+=(--requirement .github/requirements-ci-animation.txt) fi python -m pip install --user --only-binary=:all: --require-hashes "${requirements[@]}" @@ -2178,23 +2101,19 @@ jobs: echo "$HOME/.local/bin" >> "$GITHUB_PATH" # The inventory's parser packages, installed the way a first # --reader=parser run installs them (npm ci from the committed - # lockfile, here into the checkout's .work/), so test_fixture_parse.py - # and test_parser_reader.py run instead of skipping. - if [[ "$leg_needs" == all || "$leg_needs" == *inventory* ]]; then + # lockfile, here into the checkout's .work/). + if [[ "$leg_needs" == *inventory* ]]; then python plugins/harness-ops/skills/inventory/scripts/parser_reader.py --install fi # The harness-ops observability suites (prune-otel-store, hook-latency, # session-compare) query fixture stores with the DuckDB CLI and skip those # cases without it, so the plan names every harness-ops suite for it. - name: Install DuckDB CLI - if: needs.scope.outputs.run_tests == 'true' env: DUCKDB_VERSION: v1.5.5 DUCKDB_SHA256: c61f21485e6e41d3a0c28ce9904ea18346309cf427b4cf9479bc3564348dc885 - LEG_PLAN: ${{ needs.scope.outputs.test_needs }} run: | - leg_needs=$(jq -r --arg leg "$LEG" '.[$leg] // "all"' <<<"${LEG_PLAN:-"{}"}" 2>/dev/null) || leg_needs=all - if [[ "$leg_needs" != all && "$leg_needs" != *duckdb* ]]; then + if [[ "$(jq -r --arg leg "$LEG" '.[$leg] // ""' <<<"$LEG_PLAN")" != *duckdb* ]]; then echo "No suite on this leg reads DuckDB." exit 0 fi @@ -2207,63 +2126,10 @@ jobs: gunzip "$RUNNER_TEMP/duckdb/duckdb.gz" chmod +x "$RUNNER_TEMP/duckdb/duckdb" echo "$RUNNER_TEMP/duckdb" >> "$GITHUB_PATH" - # ONLY THE SUITES THE DIFF AFFECTS, on a pull request and on a push. - # affected-tests.sh derives the covering suites from the files changed - # since the `scope` job's diff base and runs them; with no base (a - # schedule or dispatch run, or a push that could not use one) the full - # corpus runs. Its exit contract decides what happens next, and each - # branch is deliberate: - # - # 0 selected and green, or nothing to do. - # 3 every shell suite it selected ran, but it also selected suites in - # other ecosystems whose runner it will not guess. Those suites are - # RUN HERE rather than waved through: the steps further down this - # job cover four Node sub-projects and one Python module, so - # treating exit 3 as success would report a pass over suites that - # never executed. That is the precise thing the selector's own - # header refuses to do. - # 1 an unmapped changed file OR a failing suite. Only the first is - # recoverable, and it announces itself with UNMAPPED: on stderr: - # nothing here knows what covers that file, so fall back to the full - # shell corpus and record the fallback so its rate is countable per - # run. The shell corpus holds no Python or Node suite, so those run - # as on exit 3, from this leg's slice of the `--unmapped-corpus` - # listing: the selection's own, plus every Python suite when the - # unmapped file is a .py and every Node suite when it is Node. - # * anything else fails. - # - # Three suites at a time, not four. The runner has 4 vCPUs and the suites - # are spawn-bound, so four looked like the shape that pays, but at four - # three separate suites failed across two runs by producing EMPTY output - # from an external command on a path with no clock in it. Serialising - # each one in turn only moves the symptom to the next suite, so the job - # count is the lever rather than the allowlist. See #3694. The default - # stays 1 for Windows dev boxes; the suites that assert wall-clock - # ceilings run one at a time regardless, from - # scripts/run-plugin-tests-serial.txt. - # - # SHARDED ACROSS THE MATRIX. `--shard /` narrows the SELECTION - # after the selector has derived it in full, so every leg answers the same - # unmapped question and the union of the four legs is exactly the suite - # set one job used to run. A leg that draws nothing exits 0 in about a - # second, which is what the short-circuit path needs: if a leg SKIPPED - # instead, the matrix result would be `skipped` and `ci-status`, which is - # fail-closed on `skipped`, would red the pull request. - # - # The whole tree, and the UNMAPPED fallback, run the full shell corpus - # through run-plugin-tests.sh with the same `--shard`, which partitions - # its discovered suites the same way, so the four legs run it once - # between them. The whole tree also runs every Python suite, each leg - # every legs-th one. - # - # ONE PYTEST PROCESS PER PYTHON SUITE, under the pinned pytest, which - # collects unittest.TestCase suites as well as pytest-style ones. Suites - # in different directories import same-named helpers (`from conftest - # import ...`), and a shared process hands each the first one loaded. - - name: Run plugin contract tests - if: needs.scope.outputs.run_tests == 'true' + # A leg with no suites is a broken plan, never "nothing to do": the plan + # makes no more legs than it has suites. + - name: Run this leg's shell suites env: - DIFF_BASE: ${{ needs.scope.outputs.lane_base }} # Empty on every event but a dispatch that asked for the lanes, which # is what the gate in the bench suite reads as "stay deferred". BENCH_LANES: ${{ inputs.bench_lanes && '1' || '' }} @@ -2274,92 +2140,131 @@ jobs: # The code-tidying eval fixtures fail on a missing tree-sitter here instead of skipping green. CODE_TIDYING_REQUIRE_TREE_SITTER: '1' run: | - # Eval fixtures are data the skill evals read, not suites. - pytest_each() { - awk '!/\/evals\/fixtures\//' "$1" | tr '\n' '\0' | - xargs -0 -r -t -n1 python -m pytest -q -o tmp_path_retention_policy=none -- - } - # A listing's suites the selector does not run itself: Python here, - # Node through each package's own npm test. Pester is the - # test-windows lane. #3703. - run_delegated() { - local rc=0 - grep -E '(^|/)test_[^/]*\.py$' "$1" >"$RUNNER_TEMP/python-suites.txt" || true - grep -Ev '(^$|\.test\.sh$|\.py$)' "$1" >"$RUNNER_TEMP/outside-node-paths.txt" || true - pytest_each "$RUNNER_TEMP/python-suites.txt" || rc=1 - if [ -s "$RUNNER_TEMP/outside-node-paths.txt" ]; then - scripts/run-outside-node-suites.sh --paths "$RUNNER_TEMP/outside-node-paths.txt" || rc=1 - fi - return "$rc" - } - if [ -z "$DIFF_BASE" ]; then - status=0 - scripts/run-plugin-tests.sh --jobs 3 --shard "$LEG/$LEGS" || status=1 - # The shell corpus is sharded above. The outside-Node packages are - # a fixed pair, so one leg runs them once on the whole-tree path (#3703). - if [ "$LEG" = $((3 % LEGS)) ]; then - scripts/run-outside-node-suites.sh || status=1 - fi - git ls-files | awk -v leg="$LEG" -v legs="$LEGS" '{ b = $0; sub(/.*\//, "", b) } - b ~ /^test_.*\.py$/ && n++ % legs == leg' >"$RUNNER_TEMP/python-suites.txt" - pytest_each "$RUNNER_TEMP/python-suites.txt" || status=1 - exit "$status" + jq -r --arg leg "$LEG" '.[$leg] // [] | .[]' <<<"$PLAN" >"$RUNNER_TEMP/leg-suites.txt" + if [ ! -s "$RUNNER_TEMP/leg-suites.txt" ]; then + echo "::error::the plan gives leg $LEG no suites; a leg exists only to run some." + exit 1 fi - err="$RUNNER_TEMP/affected-tests.err" - set +e - scripts/affected-tests.sh --run --jobs 3 --shard "$LEG/$LEGS" --with-always --base "$DIFF_BASE" 2>"$err" - status=$? - set -e - cat "$err" >&2 - case "$status" in - 0) - ;; - 3) - # The selector prints the suites it declined to run under a - # "NOT RUN:" heading, one indented path each. - awk '/^NOT RUN:/{f=1} f && /^ - /{sub(/^ - /,""); print}' "$err" >"$RUNNER_TEMP/delegated.txt" - run_delegated "$RUNNER_TEMP/delegated.txt" - ;; - 1) - if grep -q '^UNMAPPED:' "$err"; then - summary=$(grep -m1 '^UNMAPPED:' "$err") - echo "::warning::$summary Falling back to the full shell corpus." - printf 'affected-tests fallback: %s\n' "$summary" >> "$GITHUB_STEP_SUMMARY" - status=0 - scripts/run-plugin-tests.sh --jobs 3 --shard "$LEG/$LEGS" || status=1 - # Same fixed pair the whole-tree path runs once on leg 3 (#3703). - if [ "$LEG" = $((3 % LEGS)) ]; then - scripts/run-outside-node-suites.sh || status=1 - fi - listed=0 - scripts/affected-tests.sh --unmapped-corpus --shard "$LEG/$LEGS" --base "$DIFF_BASE" \ - >"$RUNNER_TEMP/listing.txt" 2>"$RUNNER_TEMP/listing.err" || listed=$? - if [ "$listed" != 0 ] && [ "$listed" != 4 ]; then - cat "$RUNNER_TEMP/listing.err" >&2 - echo "::error::affected-tests.sh --unmapped-corpus exited $listed, so the Python and Node suites to run are unknown." - exit 1 - fi - grep -v '\.test\.sh$' "$RUNNER_TEMP/listing.txt" >"$RUNNER_TEMP/delegated.txt" || true - run_delegated "$RUNNER_TEMP/delegated.txt" || status=1 - exit "$status" - else - exit 1 - fi - ;; - *) - exit "$status" - ;; - esac + scripts/run-plugin-tests.sh --jobs 3 --suites-from "$RUNNER_TEMP/leg-suites.txt" + + # THE PYTHON SUITES THE CHANGE SELECTS, one pytest process per module, on one + # to four legs of about 180 suite-seconds (one module at a time). Every module + # on the whole tree or after a change to a Python pin. The pinned pytest + # collects unittest.TestCase suites as well as pytest-style ones, the + # disk-hygiene module among them. Suites in different directories import + # same-named helpers (`from conftest import ...`), and a shared process hands + # each the first one loaded, hence a process per module. Node and the root + # npm install come too: Python suites run the claude CLI from + # node_modules/.bin and the inventory's parser packages through node. + test-python: + needs: [scope] + # The contract-only gate, verbatim, the draft gate (see `ci-status`), and + # this lane's ordered skip (see `test-bash`). + if: ${{ !(github.event.pull_request.head.repo.full_name == github.repository && (contains(fromJSON('["labeled","unlabeled"]'), github.event.action) || (github.event.action == 'edited' && !github.event.changes.base))) && github.event.pull_request.draft != true && needs.scope.outputs.run_python == 'true' }} + runs-on: ubuntu-24.04 + timeout-minutes: 7 + strategy: + fail-fast: false + matrix: + leg: ${{ fromJSON(needs.scope.outputs.python_legs || '[0]') }} + env: + LEG: ${{ strategy.job-index }} + PLAN: ${{ needs.scope.outputs.python_plan }} + LEG_PLAN: ${{ needs.scope.outputs.python_needs }} + steps: + - name: Check out + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - name: Fetch base + uses: ./.github/actions/checkout-with-base # zizmor: ignore[self-repository] + - name: Set up Node + uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version-file: .node-version + cache: npm + cache-dependency-path: | + package-lock.json + plugins/harness-ops/skills/inventory/scripts/js/package-lock.json + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version-file: .python-version + cache: pip + cache-dependency-path: | + .github/requirements-ci.txt + .github/requirements-ci-animation.txt + - name: Install locked plugin test toolchains + run: | + # The same pin-equality check test-bash makes: a Python pin change runs this lane. + diff <(grep -oE '^[A-Za-z0-9._-]+==[^ ]+' plugins/animation/requirements.in | sort) \ + <(grep -oE '^[A-Za-z0-9._-]+==[^ ]+' .github/requirements-ci-animation.txt | sort) \ + || { echo "::error::.github/requirements-ci-animation.txt and plugins/animation/requirements.in pin different versions; bump the two together"; exit 1; } + leg_needs=$(jq -er --arg leg "$LEG" '.[$leg]' <<<"$LEG_PLAN") || + { echo "::error::the plan names no toolchains for leg $LEG"; exit 1; } + echo "This leg installs: ${leg_needs:-the base toolchains only}" + npm ci + requirements=(--requirement .github/requirements-ci.txt) + if [[ "$leg_needs" == *animation* ]]; then + requirements+=(--requirement .github/requirements-ci-animation.txt) + fi + python -m pip install --user --only-binary=:all: --require-hashes "${requirements[@]}" + echo "$GITHUB_WORKSPACE/node_modules/.bin" >> "$GITHUB_PATH" + echo "$HOME/.local/bin" >> "$GITHUB_PATH" + # test_fixture_parse.py and test_parser_reader.py run instead of skipping. + if [[ "$leg_needs" == *inventory* ]]; then + python plugins/harness-ops/skills/inventory/scripts/parser_reader.py --install + fi + - name: Run this leg's Python suites + env: + ANIMATION_REQUIRE_DEPS: '1' + INVENTORY_REQUIRE_ACORN: '1' + CODE_TIDYING_REQUIRE_TREE_SITTER: '1' + run: | + jq -r --arg leg "$LEG" '.[$leg] // [] | .[]' <<<"$PLAN" >"$RUNNER_TEMP/leg-suites.txt" + if [ ! -s "$RUNNER_TEMP/leg-suites.txt" ]; then + echo "::error::the plan gives leg $LEG no suites; a leg exists only to run some." + exit 1 + fi + tr '\n' '\0' <"$RUNNER_TEMP/leg-suites.txt" | + xargs -0 -r -t -n1 python -m pytest -q -o tmp_path_retention_policy=none -- + + # THE NODE PACKAGES THE CHANGE REACHES: the four sub-projects with their own + # install, typecheck, lint and test chains (the miro MCP server, the + # ai-briefing build, the video-digest and course-digest extraction pipelines) + # and the packages of scripts/outside-node-packages.txt. A package runs when + # a file under it, its test facade or a `file:` dependency changed, when a + # selected Node suite sits under it, on a Node pin, and on the whole tree + # (`PACKAGES`). A step for a package the plan does not name exits 0 on + # its first line. One job: the chains measure 31 s together. + test-node: + needs: [scope] + # The contract-only gate, verbatim, the draft gate (see `ci-status`), and + # this lane's ordered skip (see `test-bash`). + if: ${{ !(github.event.pull_request.head.repo.full_name == github.repository && (contains(fromJSON('["labeled","unlabeled"]'), github.event.action) || (github.event.action == 'edited' && !github.event.changes.base))) && github.event.pull_request.draft != true && needs.scope.outputs.run_node == 'true' }} + runs-on: ubuntu-24.04 + timeout-minutes: 5 + env: + PACKAGES: ${{ needs.scope.outputs.node_packages }} + steps: + - name: Check out + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + # One setup-node for every package, keyed on every lockfile they install + # from, so a change to any one of them invalidates the cache the others + # share. + - name: Set up Node + uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version-file: .node-version + cache: npm + cache-dependency-path: | + plugins/miro/server/package-lock.json + plugins/ai-briefing/skills/generate/output/build/package-lock.json + plugins/knowledge/skills/video-digest/extraction/package-lock.json + plugins/knowledge/skills/course-digest/extraction/package-lock.json - # SUITES ARE NOT STEPS. The suites of scripts/ and lib/ (gate self-tests, - # the shared libraries, the selector's own) are corpus suites: the - # contract-test step above runs them when the selector reaches them, and - # every one of them on the whole tree. The few whose cases assert against - # the live repository rather than fixtures ride every selection through - # `--with-always` (scripts/affected-tests-always.txt). What follows are - # the Node sub-projects and the disk-hygiene module, each assigned to one - # leg. - # # The miro plugin ships a Node MCP server as TypeScript source. Claude Code # starts src/launch.ts, which installs the runtime dependencies from the # committed lockfile into the plugin data directory on first launch. The @@ -2368,34 +2273,29 @@ jobs: # but cannot install or serve MCP is caught here, not on a consumer's # machine. - name: Install dependencies - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/miro/server "* ]] || exit 0 npm ci working-directory: plugins/miro/server - name: Typecheck - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/miro/server "* ]] || exit 0 npm run typecheck working-directory: plugins/miro/server - name: Lint - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/miro/server "* ]] || exit 0 npm run lint working-directory: plugins/miro/server - name: Test - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/miro/server "* ]] || exit 0 npm test working-directory: plugins/miro/server - name: Smoke-test the MCP server over stdio, installed on demand - if: needs.scope.outputs.run_node == 'true' working-directory: plugins/miro/server run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/miro/server "* ]] || exit 0 export CLAUDE_PLUGIN_DATA="$RUNNER_TEMP/miro-plugin-data" rm -rf "${CLAUDE_PLUGIN_DATA:?}" for run in first second; do @@ -2418,37 +2318,30 @@ jobs: # installs (install-links via the package's .npmrc), so manifest/lockfile # drift breaks `npm ci` only on a clean install. - name: Install dependencies - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/knowledge/skills/video-digest/extraction "* ]] || exit 0 bash plugins/knowledge/skills/video-digest/scripts/run-tests.sh install - name: Typecheck - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/knowledge/skills/video-digest/extraction "* ]] || exit 0 bash plugins/knowledge/skills/video-digest/scripts/run-tests.sh build - name: Test - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/knowledge/skills/video-digest/extraction "* ]] || exit 0 bash plugins/knowledge/skills/video-digest/scripts/run-tests.sh test # ai-briefing's build/render pipeline is a Node package (native # `node --test` suite) no other lane installs or runs — including # test/url-policy.test.js, which asserts the SSRF gate (lib/url-policy.js) # that decides which URLs the briefing validator is allowed to - # dereference. Without these steps that suite runs locally only, so a - # future refactor could silently break the predicate with nothing red in - # CI (#1488). + # dereference (#1488). - name: Install dependencies - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/ai-briefing/skills/generate/output/build "* ]] || exit 0 bash plugins/ai-briefing/skills/generate/scripts/run-tests.sh install - name: Test - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/ai-briefing/skills/generate/output/build "* ]] || exit 0 bash plugins/ai-briefing/skills/generate/scripts/run-tests.sh test # The knowledge plugin's course-digest extraction pipeline is a Node @@ -2457,36 +2350,30 @@ jobs: # pulled in as file: dependencies via the package's .npmrc # install-links=true (#1507). - name: Install dependencies - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/knowledge/skills/course-digest/extraction "* ]] || exit 0 bash plugins/knowledge/skills/course-digest/scripts/run-tests.sh install - name: Typecheck - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/knowledge/skills/course-digest/extraction "* ]] || exit 0 bash plugins/knowledge/skills/course-digest/scripts/run-tests.sh build - name: Test - if: needs.scope.outputs.run_node == 'true' run: | - [ "$LEG" = $((3 % LEGS)) ] || exit 0 + [[ " $PACKAGES " == *" plugins/knowledge/skills/course-digest/extraction "* ]] || exit 0 bash plugins/knowledge/skills/course-digest/scripts/run-tests.sh test - # #2871: the full test_hygiene module. Its ten TestCase classes need a - # check that cannot take hygiene.test.sh's SKIP-when-Python-absent path. - # The Windows lane stays GuardTests-only (NT allowlist, plus two failures - # named there). Suite is not host-destructive (tempdir fixtures; no live - # drive-root scans). Measured 317 OK. - - name: Run disk-hygiene tests - if: needs.scope.outputs.run_python == 'true' - working-directory: plugins/disk-hygiene/skills/clean/scripts + # The packages of scripts/outside-node-packages.txt, each through its own + # `npm test` (#3703); the runner takes a package directory as a path + # under that package. + - name: Test the other Node packages run: | - [ "$LEG" = $((2 % LEGS)) ] || exit 0 - python -m unittest -v test_hygiene - - - name: Report not applicable to a docs-only diff - if: needs.scope.outputs.run_tests == 'false' - run: echo "The diff is within the docs-only allowlist (scripts/docs-only-paths.txt); the plugin contract suite and the sub-project builds cannot be affected — reporting success." + tr ' ' '\n' <<<"$PACKAGES" | grep -Fxf <(grep -v '^#' scripts/outside-node-packages.txt | grep .) \ + >"$RUNNER_TEMP/outside-node-paths.txt" || true + if [ ! -s "$RUNNER_TEMP/outside-node-paths.txt" ]; then + echo "The plan names none of these packages." + exit 0 + fi + scripts/run-outside-node-suites.sh --paths "$RUNNER_TEMP/outside-node-paths.txt" ci-status: needs: @@ -2496,6 +2383,8 @@ jobs: - check-plugins - check-skills - test-bash + - test-python + - test-node # Fail-closed through execution: !cancelled() (never a success-guard) so a # lane failure still runs this required aggregate and the aggregation below # turns it red. A success-guard would skip the job, and a skipped required @@ -2597,6 +2486,35 @@ jobs: fi echo "::error::draft: lanes not run. Mark the pull request ready to lint and test this SHA." exit 1 + # ORDERED SKIPS ARE CHECKED HERE. A test lane `test-` skips as a job + # when `scope`'s `run_` row says the change gives it no work. Its + # `skipped` becomes `success` for the aggregate only when `scope` + # succeeded and that row is exactly 'false'; every other result passes + # through as it is, so a skip that hid work stays `skipped` and the + # aggregate fails it. Pinned by scripts/ci-ordered-skips.test.sh, which + # runs this body. + - name: Check each skipped lane against scope + id: lanes + if: ${{ !cancelled() }} + env: + LANES: ${{ toJSON(needs) }} + run: | + results=$(jq -r ' + .scope as $s + | to_entries + | map(.key as $lane | .value.result as $r + | if $r == "skipped" and $s.result == "success" + and ($lane | test("^test-[a-z]+$")) + and ($s.outputs["run_" + ($lane | ltrimstr("test-"))] == "false") + then "success" else $r end) + | join(" ")' <<<"$LANES") + if [ -z "$results" ]; then + echo "::error::no lane results could be read." + exit 1 + fi + jq -r 'to_entries[] | "\(.key): \(.value.result)"' <<<"$LANES" + echo "Results after the ordered-skip check: $results" + echo "results=$results" >>"$GITHUB_OUTPUT" - name: Aggregate lane results # `!cancelled()` so a failing pr-contract step does not skip the # aggregation: the job still fails on the contract step's exit code, and @@ -2604,10 +2522,10 @@ jobs: if: ${{ !cancelled() }} uses: melodic-software/ci-workflows/.github/actions/ci-status@a267a27f7a321452267e20c82d57699b6c057cb0 # v0.30.2 with: - # Derived from the needs graph above — the single source of truth for - # the lane list. Adding a lane to needs automatically extends this - # check. - results: ${{ join(needs.*.result, ' ') }} + # Every lane of the needs graph above — the single source of truth + # for the lane list — through the ordered-skip check. Adding a lane + # to needs automatically extends this check. + results: ${{ steps.lanes.outputs.results }} # A `skipped` lane is red, a draft's included (see `Fail a draft`). treat-skipped-as: fail # A CONTRACT-ONLY RUN NEVER WAITS. It reads the newest bot-written diff --git a/.github/workflows/selection-audit.yml b/.github/workflows/selection-audit.yml index 09863e6b84..bf4634fef2 100644 --- a/.github/workflows/selection-audit.yml +++ b/.github/workflows/selection-audit.yml @@ -57,8 +57,8 @@ jobs: with: persist-credentials: false # The suites read what they read only when their tools are present, so - # the toolchains match ci.yml's test-bash lane (Node, Python, ShellCheck, - # shfmt, the inventory parser packages, DuckDB); a suite that skips for a + # the toolchains match ci.yml's test-bash and test-python lanes (Node, + # Python, ShellCheck, shfmt, the inventory parser packages, DuckDB); a suite that skips for a # missing tool reads less and can hide a gap, never invent one. - name: Set up Node uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 diff --git a/.github/workflows/test-windows.yml b/.github/workflows/test-windows.yml index fc9ddc8859..c17e4efc7c 100644 --- a/.github/workflows/test-windows.yml +++ b/.github/workflows/test-windows.yml @@ -16,26 +16,39 @@ name: test-windows # false green. Inside the required closure the same condition would let a lane # report success having run nothing, which both of ci.yml's gates reject. # -# SCOPE IS DECIDED AT TRIGGER LEVEL. Being outside the required closure is -# also what makes `on.paths` legal here: a workflow a path filter never starts -# leaves no pending required check behind. The `paths` lists below are the -# union of the `shell`, `python` and `powershell` groups of ci.yml's -# change-detection filter table, so a diff that touches none of them (docs -# only, say) starts no run and no Linux resolver job. Keep them in step with -# those three groups. GitHub reads at most the first 300 changed files for a -# `paths` filter, so a larger diff whose matching files fall past that point -# starts no run; accepted for an informational lane. +# THE SELECTION DECIDES WHAT RUNS. `scope-windows` runs +# scripts/plan-test-lanes.sh, the planner ci.yml's `scope` runs, over the same +# diff, and a Windows step runs only when the change selects its suite: each +# step's key and the suites that start it are lines of +# scripts/test-windows-plan.txt, and a job runs when one of its steps does. A +# change to this file or to .github/actions/, a schedule run and a dispatch run +# every step. A push diffs against the push before it. # -# Draft pull requests run nothing: every job gates on the draft flag, and -# `ready_for_review` starts the run when the draft flips. Contract-only events -# (`edited`, `labeled`, `unlabeled`) are not triggers, since this lane reads -# nothing from the pull request body or labels. Nor is `merge_group`: a merge -# queue waits only on required checks, and this lane is not one. +# `on.paths` still decides whether a run starts at all, because being outside +# the required closure is also what makes it legal here: a workflow a path +# filter never starts leaves no pending required check behind. The `paths` +# lists below are the union of the `shell`, `python` and `powershell` groups of +# ci.yml's change-detection filter table, so a diff that touches none of them +# (docs only, say) starts no run. Keep them in step with those three groups. +# GitHub reads at most the first 300 changed files for a `paths` filter, so a +# larger diff whose matching files fall past that point starts no run; accepted +# for an informational lane, and the scheduled run covers it. +# +# Draft pull requests run nothing: `scope-windows` gates on the draft flag and +# every other job needs it, and `ready_for_review` starts the run when the +# draft flips. Contract-only events (`edited`, `labeled`, `unlabeled`) are not +# triggers, since this lane reads nothing from the pull request body or labels. +# Nor is `merge_group`: a merge queue waits only on required checks, and this +# lane is not one. on: # Present for parity with ci.yml, which takes a dispatch to run its bench - # lanes. This lane has none; a dispatch here just runs the Windows suites. + # lanes. This lane has none; a dispatch here runs every Windows step. workflow_dispatch: + # Every Windows step, at ci.yml's whole-tree hours: the net for a Windows + # suite a change reaches through something the selector does not see. + schedule: + - cron: '17 9,16 * * *' push: branches: [main] paths: @@ -93,26 +106,54 @@ concurrency: cancel-in-progress: true jobs: - # The only jobs on a Windows runner in this repository, and deliberately so. - # Every other suite is platform-agnostic bash string logic that a Linux - # runner exercises identically — `lib/powershell/ps-command.sh`, for - # instance, carries zero `OSTYPE`/`cygpath`/`uname` branches despite - # classifying PowerShell commands, so Linux tests it faithfully. + # Which Windows jobs and steps this change runs, on a Linux runner: the + # selector takes about 20 s there and several times that under Git Bash. + scope-windows: + # Not on a draft; always on push, schedule and dispatch, where the flag is absent. + if: ${{ github.event.pull_request.draft != true }} + runs-on: ubuntu-24.04 + timeout-minutes: 3 + outputs: + jobs: ${{ steps.plan.outputs.windows_jobs }} + steps: ${{ steps.plan.outputs.windows_steps }} + steps: + - name: Check out + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - name: Fetch base + uses: ./.github/actions/checkout-with-base # zizmor: ignore[self-repository] + - name: Plan the Windows steps the change selects + id: plan + env: + BASE_REF: ${{ github.base_ref }} + BEFORE: ${{ github.event.before }} + run: | + base='' + case "$GITHUB_EVENT_NAME" in + pull_request) base="origin/$BASE_REF" ;; + push) git merge-base --is-ancestor "$BEFORE" HEAD 2>/dev/null && base="$BEFORE" ;; + *) ;; + esac + [ -n "$base" ] || echo "::notice::A $GITHUB_EVENT_NAME run with no usable base runs every Windows step." + scripts/plan-test-lanes.sh ${base:+--base "$base"} >>"$GITHUB_OUTPUT" + + # Every suite runs on Linux too. A suite is here because a Linux runner cannot + # exercise part of what it tests: # - # Five surfaces are the exception. `lib/hook-utils.sh` carries OSTYPE-gated - # branches a Linux runner NEVER EXECUTES — case-insensitive path folding for - # the Windows filesystem, and `cygpath` short-name (8.3) resolution. #2774's - # disk-hygiene read-only Bash allowlist is NT-specific (MSYS path mapping, - # .exe basename stripping, ProgramFiles trust roots), which is how an inert - # Windows allowlist shipped green. And the MSYS-to-Windows emit helper exists - # for its NT branch: cygpath's answer cannot be faked on Linux, so on a Linux - # runner those cases report NOT EXERCISED and the suite is green having - # proved nothing about them. And the gaming plugin's DLSS 5 script is - # Windows-only by its own header (backslash paths, Authenticode), so its - # selftest SKIPs on every other OS and ran in no CI before this lane. And - # the exec-form bash launcher (#3686) resolves a real Git Bash, refuses the - # System32 WSL relay and spawns bash from node.exe with no shell, none of - # which a Linux runner does. + # `lib/hook-utils.sh` carries OSTYPE-gated branches a Linux runner NEVER + # EXECUTES — case-insensitive path folding for the Windows filesystem, and + # `cygpath` short-name (8.3) resolution. #2774's disk-hygiene read-only Bash + # allowlist is NT-specific (MSYS path mapping, .exe basename stripping, + # ProgramFiles trust roots), which is how an inert Windows allowlist shipped + # green. The MSYS-to-Windows emit helper exists for its NT branch: cygpath's + # answer cannot be faked on Linux, so on a Linux runner those cases report NOT + # EXERCISED and the suite is green having proved nothing about them. The + # gaming plugin's DLSS 5 script is Windows-only by its own header (backslash + # paths, Authenticode), so its selftest SKIPs on every other OS. And the + # exec-form bash launcher (#3686) resolves a real Git Bash, refuses the + # System32 WSL relay and spawns bash from node.exe with no shell, none of which + # a Linux runner does. # # GUARDRAILS (#4527). The guardrails suites whose guards branch on OSTYPE, # cygpath or MSYS run here too: block-windows-drive-tmp, hardcoded-path-check, @@ -124,7 +165,8 @@ jobs: # # Keep these jobs SMALL. Adding a platform-agnostic suite here buys no # coverage and pays Windows' process-creation cost (~140ms/spawn vs ~3ms on - # Linux). + # Linux). A new step needs a line in scripts/test-windows-plan.txt under its + # job, keyed as its `if:` is. # # SIX PARALLEL JOBS (#5830, #5988). Every slow step is a bash suite bound by # that spawn cost, so the lane's wall time is its longest job. The rename @@ -145,10 +187,10 @@ jobs: # Pester + kindle-dedrm 6 = 197 # Each suite is a step of its own so its log shows its own duration. Put a # new suite in the job with the smallest total, and keep each install step - # in the job whose suites need it: Pester before each Pester suite - # (rename-2, lib), PyYAML before the exec-form probe (hooks). Do not run - # suites concurrently inside a job: hook-utils and markdown-format assert - # timing that CPU contention breaks. + # in the job whose suites need it, gated like the step it serves: Pester + # before each Pester suite (rename-2, lib), PyYAML before the exec-form probe + # (hooks). Do not run suites concurrently inside a job: hook-utils and + # markdown-format assert timing that CPU contention breaks. # # HOST-SKIP SUITES (#3683). Three suites probe the host at run time and two # print a SKIP line when they cannot pin their cases, so their Windows @@ -171,8 +213,8 @@ jobs: # Tuesday. Free: GitHub Actions is free for public repositories on standard # GitHub-hosted runners, Windows included. test-windows-rename-1: - # Not on a draft; always on push and dispatch, where the flag is absent. - if: ${{ github.event.pull_request.draft != true }} + needs: [scope-windows] + if: ${{ contains(fromJSON(needs.scope-windows.outputs.jobs || '[]'), 'test-windows-rename-1') }} runs-on: windows-2025 timeout-minutes: 20 defaults: @@ -191,18 +233,19 @@ jobs: # here: the rename executor's case-only `git mv` path and its # index-not-filesystem existence test cannot be exercised on a # case-sensitive runner, so a regression in either would ship green from - # Linux alone. Named explicitly, like the fixed suites above, rather than - # discovered. Shard 2 runs in `test-windows-rename-2`. + # Linux alone. Shard 2 runs in `test-windows-rename-2`. - name: Test the file-name rename executor on Windows (shard 1 of 2) + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/docs-naming/skills/realign-file-names/scripts/apply-rename.test.sh') env: APPLY_RENAME_TEST_SHARD: "1" run: bash plugins/docs-naming/skills/realign-file-names/scripts/apply-rename.test.sh - name: Test run-guards on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/guardrails/hooks/run-guards.test.sh') run: bash plugins/guardrails/hooks/run-guards.test.sh test-windows-rename-2: - # `test-windows-rename-1`'s gate, verbatim. - if: ${{ github.event.pull_request.draft != true }} + needs: [scope-windows] + if: ${{ contains(fromJSON(needs.scope-windows.outputs.jobs || '[]'), 'test-windows-rename-2') }} runs-on: windows-2025 timeout-minutes: 20 defaults: @@ -218,19 +261,24 @@ jobs: with: python-version-file: .python-version - name: Test the file-name rename executor on Windows (shard 2 of 2) + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/docs-naming/skills/realign-file-names/scripts/apply-rename.test.sh') env: APPLY_RENAME_TEST_SHARD: "2" run: bash plugins/docs-naming/skills/realign-file-names/scripts/apply-rename.test.sh - name: Test the piped early-exit grep gate on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'scripts/check-pipefail-grep-q.test.sh') run: bash scripts/check-pipefail-grep-q.test.sh - name: Test the drive-root litter detector on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'scripts/check-drive-root-litter.test.sh') run: bash scripts/check-drive-root-litter.test.sh - name: Test the MSYS-to-Windows emit helper on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'scripts/emit-windows-path.test.sh') run: bash scripts/emit-windows-path.test.sh # Pester has no Linux lane. These are the suites affected-tests.sh names # and declines: machine-health's own runner here, and the kindle-dedrm # tests in `test-windows-lib`, which installs Pester itself. #3703. - name: Install Pester 5.7 + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/machine-health/skills/audit/scripts/run-tests.ps1') shell: pwsh run: | $found = Get-Module -ListAvailable Pester | Where-Object { $_.Version -ge [version]'5.7.0' } | Select-Object -First 1 @@ -238,12 +286,13 @@ jobs: Install-Module Pester -MinimumVersion 5.7.0 -Force -Scope CurrentUser -SkipPublisherCheck } - name: Run machine-health Pester suites + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/machine-health/skills/audit/scripts/run-tests.ps1') shell: pwsh run: pwsh -NoProfile -NonInteractive -File plugins/machine-health/skills/audit/scripts/run-tests.ps1 test-windows-secrets: - # `test-windows-rename-1`'s gate, verbatim. - if: ${{ github.event.pull_request.draft != true }} + needs: [scope-windows] + if: ${{ contains(fromJSON(needs.scope-windows.outputs.jobs || '[]'), 'test-windows-secrets') }} runs-on: windows-2025 timeout-minutes: 20 defaults: @@ -259,11 +308,14 @@ jobs: with: python-version-file: .python-version - name: Test secret-pattern-detection on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/guardrails/hooks/secret-pattern-detection.test.sh') run: bash plugins/guardrails/hooks/secret-pattern-detection.test.sh - name: Test coverage-manifest on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/guardrails/hooks/coverage-manifest.test.sh') run: bash plugins/guardrails/hooks/coverage-manifest.test.sh # A host-skip suite of #3683 (see the job comment above). - name: Test the skill-visibility audit on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/harness-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.test.sh') run: bash plugins/harness-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.test.sh # GuardTests only, not the full hygiene module: two of its tests fail on # windows-2025 and are already green on Linux — the preview case expects @@ -271,12 +323,13 @@ jobs: # handoff case compares path strings where 8.3 `RUNNER~1` differs from # the long `runneradmin`. The other nine classes gate on the Linux run. - name: Run disk-hygiene GuardTests on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/disk-hygiene/skills/clean/scripts/test_hygiene.py') working-directory: plugins/disk-hygiene/skills/clean/scripts run: python -m unittest -v test_hygiene.GuardTests test-windows-guards: - # `test-windows-rename-1`'s gate, verbatim. - if: ${{ github.event.pull_request.draft != true }} + needs: [scope-windows] + if: ${{ contains(fromJSON(needs.scope-windows.outputs.jobs || '[]'), 'test-windows-guards') }} runs-on: windows-2025 timeout-minutes: 20 defaults: @@ -292,13 +345,15 @@ jobs: with: python-version-file: .python-version - name: Test block-hook-bypass on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/guardrails/hooks/block-hook-bypass.test.sh') run: bash plugins/guardrails/hooks/block-hook-bypass.test.sh - name: Test hardcoded-path-check on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/guardrails/hooks/hardcoded-path-check.test.sh') run: bash plugins/guardrails/hooks/hardcoded-path-check.test.sh test-windows-hooks: - # `test-windows-rename-1`'s gate, verbatim. - if: ${{ github.event.pull_request.draft != true }} + needs: [scope-windows] + if: ${{ contains(fromJSON(needs.scope-windows.outputs.jobs || '[]'), 'test-windows-hooks') }} runs-on: windows-2025 timeout-minutes: 20 defaults: @@ -317,23 +372,28 @@ jobs: # run time, so a green step is only evidence once its log shows whether # the probe fired or the cases ran. - name: Test the markdown-format hook on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/markdown-format/hooks/markdown-format.test.sh') run: bash plugins/markdown-format/hooks/markdown-format.test.sh - name: Test cloud-bootstrap plugin accounting on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), '.claude/hooks/cloud-bootstrap-plugins.test.sh') run: bash .claude/hooks/cloud-bootstrap-plugins.test.sh # The launcher test spawns a real Git Bash through node.exe, with stdin, # an argument and exit 2 passed through, and reads every hooks.json row # for the node exec form. - name: Test the guardrails exec-bash launcher on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/guardrails/hooks/exec-bash.test.sh') run: bash plugins/guardrails/hooks/exec-bash.test.sh # The probe reads hook frontmatter with PyYAML and exits 2 when the # module is missing. requirements-ci.txt also pins a Linux-only zizmor # wheel, so this step installs only the PyYAML pin from that file. - name: Install the pinned YAML reader + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'scripts/check-exec-form-windows-probe.sh') run: python -m pip install --disable-pip-version-check "pyyaml==$(awk '/^pyyaml==/ { sub(/^pyyaml==/, ""); sub(/[[:space:]\\].*$/, ""); print; exit }' .github/requirements-ci.txt)" # Exec form on Windows is a real .exe plus an args array, with no shell. # REQUIRE_SPAWN makes a missing node.exe a failure here. A non-Windows # run of the same script skips that half and does not clear #90495. - name: Probe Windows exec-form args delivery + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'scripts/check-exec-form-windows-probe.sh') env: EXEC_FORM_WINDOWS_PROBE_REQUIRE_SPAWN: "1" run: bash scripts/check-exec-form-windows-probe.sh @@ -342,11 +402,12 @@ jobs: # the real node.exe. The launcher test above is the one that spawns a # real Git Bash through node.exe. - name: Test the exec-form bash launcher resolver on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'lib/exec-bash.resolver.test.mjs') run: node --test lib/exec-bash.resolver.test.mjs test-windows-lib: - # `test-windows-rename-1`'s gate, verbatim. - if: ${{ github.event.pull_request.draft != true }} + needs: [scope-windows] + if: ${{ contains(fromJSON(needs.scope-windows.outputs.jobs || '[]'), 'test-windows-lib') }} runs-on: windows-2025 timeout-minutes: 20 defaults: @@ -362,17 +423,21 @@ jobs: with: python-version-file: .python-version - name: Test block-windows-drive-tmp on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/guardrails/hooks/block-windows-drive-tmp.test.sh') run: bash plugins/guardrails/hooks/block-windows-drive-tmp.test.sh - name: Run shared lib tests on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'lib/hook-utils.test.sh') run: bash lib/hook-utils.test.sh # pwsh directly, not the .test.sh wrapper: the wrapper SKIPs with exit 0 # when pwsh is missing, which would read as green here. The selftest's # exit code is its failure count. - name: Run the gaming DLSS 5 selftest on Windows + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/gaming/skills/dlss5/scripts/Invoke-Dlss5Mod.ps1') run: pwsh -NoProfile -NonInteractive -File plugins/gaming/skills/dlss5/scripts/Invoke-Dlss5Mod.ps1 -Verb selftest # `test-windows-rename-2` installs Pester for its own suite; this job runs # on a separate runner and installs it again. - name: Install Pester 5.7 + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/kindle-dedrm/skills/manage/tests') shell: pwsh run: | $found = Get-Module -ListAvailable Pester | Where-Object { $_.Version -ge [version]'5.7.0' } | Select-Object -First 1 @@ -380,6 +445,7 @@ jobs: Install-Module Pester -MinimumVersion 5.7.0 -Force -Scope CurrentUser -SkipPublisherCheck } - name: Run kindle-dedrm Pester suites + if: contains(fromJSON(needs.scope-windows.outputs.steps), 'plugins/kindle-dedrm/skills/manage/tests') shell: pwsh run: | $config = New-PesterConfiguration diff --git a/docs/ci-runner-routing.md b/docs/ci-runner-routing.md index a3d15554c0..554d16a48b 100644 --- a/docs/ci-runner-routing.md +++ b/docs/ci-runner-routing.md @@ -40,10 +40,14 @@ reviewed pull request. The `ci-status` required check depends on every **required** workload lane (`scope`, `lint-repo`, `lint-shell`, `check-plugins`, `check-skills`, -`test-bash`) and requires +`test-bash`, `test-python`, `test-node`) and requires each result to be `success`, failing closed through execution (`!cancelled()`, never a success-guard, so a skipped lane cannot report -success to branch protection). On a draft pull request every lane but +success to branch protection). The one exception is an ordered skip: a test +lane the change gives no work skips as a job, and `ci-status` counts that skip +as `success` only when `scope` succeeded and its row for that lane +(`run_bash`, `run_python`, `run_node`) is `false`; any other skip stays red. +On a draft pull request every lane but `scope` carries a draft gate, so a draft run lints and tests nothing, and its `ci-status` fails (`draft: lanes not run`) and records `ci-lanes=failure` (`scripts/check-docs-only-gate.sh` pins both). A green draft would be the @@ -57,7 +61,10 @@ longest job in the run: time-to-green is measured to the run's completion, so an advisory lane was setting the number. Its `on.paths` filters repeat the `shell`, `python` and `powershell` groups of `ci.yml`'s change-detection table, so a diff that touches none of them starts no Windows run; the two are kept in step by -hand. The +hand. Inside a run, its `scope-windows` job runs the same planner over the same +diff, and a Windows step runs only when the change selects its suite +(`scripts/test-windows-plan.txt`); a schedule run (09:17 and 16:17 UTC), a +dispatch, and a change to that workflow run every step. The metadata checks (Conventional Commits title, `do-not-merge` label, issue linkage) run as the `pr-contract` composite step inside the same `ci-status` job on the same hosted runner, so they no longer @@ -71,13 +78,15 @@ hygiene, the CI configuration, the docs conventions); `lint-shell` runs ShellCheck and the gates that read shell source; `check-plugins` holds the plugin manifests, changelogs, hook declarations, `claude plugin validate`, the counter ceilings and every shared-library sync; `check-skills` holds the skill -and eval contracts, check 25 included; `test-bash` runs the affected contract -suites, with the Node sub-projects and the disk-hygiene module as steps. A -domain is its own job only when that shortens the critical path by more than a -job's fixed cost (about 15 s bare, about 40 s with a toolchain), which is why -those last two ride `test-bash`. Every gate keeps its step name and its `id`, -which is its aggregator key, and each job carries its own -`aggregate-hygiene-results.sh` feed over exactly its own gate steps. +and eval contracts, check 25 included; `test-bash`, `test-python` and +`test-node` run the shell suites, the Python suites and the Node packages the +change selects, and each skips when it has none. A domain is its own job only +when that shortens the critical path by more than a job's fixed cost (about +15 s bare, about 40 s with a toolchain), or, for the test lanes, when it lets a +change that touches none of its ecosystem start no runner for it. Every gate +keeps its step name and its `id`, which is its aggregator key, and each job +carries its own `aggregate-hygiene-results.sh` feed over exactly its own gate +steps. ## What each event tests @@ -111,10 +120,17 @@ to the whole tree when there is no diff base or `ci.yml` changed, and the scheduled run scans everything. Replayed on 20 recent pull requests, every skipped or narrowed scan landed on a whole-tree success. -`scope` also plans `test-bash`: one leg per 25 selected suites, one to -four (four on the whole tree or an UNMAPPED file), and, per leg, whether its -slice needs the animation wheels, the inventory's parser packages or the DuckDB -CLI. A leg installs only those; the shfmt and DuckDB downloads are cached. +`scope` also plans the test lanes, once, with `scripts/plan-test-lanes.sh`: +each selected suite goes to the lane of its ecosystem (a Node suite with a +sibling `.test.sh` runs through it in `test-bash`), `test-bash` gets one to six +legs of about 120 suite-seconds each and `test-python` one to four of about +180, packed longest first from the measured seconds in +`scripts/suite-seconds.txt`, and each leg installs only the optional toolchains +(the animation wheels, the inventory's parser packages, the DuckDB CLI) its +suites need. `test-node` runs the Node packages the change reaches. An +UNMAPPED file adds the whole corpus of its language. A Python pin runs every +Python suite, a Node pin every Node package, and a change to `ci.yml` or +`.github/actions/` every suite of every lane. A suite that scans a directory never names the file that changed, so it declares what it reads in a `# test-scope:` header, and the selector's rule R8 diff --git a/scripts/affected-tests-no-suite.txt b/scripts/affected-tests-no-suite.txt index 8aac31f564..d6c4dce1f6 100644 --- a/scripts/affected-tests-no-suite.txt +++ b/scripts/affected-tests-no-suite.txt @@ -102,8 +102,9 @@ scripts/sync-plugin-options-docs.py # consumer's plugin cache (which fires only on a plugin-root lockfile) never # installs its dev toolchain. Its tests are vitest (`*.test.ts`), which is not # one of the four suite conventions this tool selects, so the whole tree needs -# this entry. The lane covering it is the miro block of the `test` job in -# .github/workflows/ci.yml, gated on `run_node` (matcher `plugins/miro/**`): +# this entry. The lane covering it is the miro block of the `test-node` job in +# .github/workflows/ci.yml, which runs whenever a file under plugins/miro/server/ +# changes (scripts/plan-test-lanes.sh): # `npm ci`, `npm run typecheck`, `npm run lint`, `npm test`, and a stdio smoke # test that starts src/launch.ts twice from an empty plugin data directory (the # first launch installs from the lockfile, the second reuses it). Manifests and prose under the diff --git a/scripts/affected-tests.test.sh b/scripts/affected-tests.test.sh index ee1f2d6144..7123b9b57e 100755 --- a/scripts/affected-tests.test.sh +++ b/scripts/affected-tests.test.sh @@ -1413,11 +1413,11 @@ fi rm -rf "$repo3" # --- LIVE repo: ci.yml actually fans the selection out ----------------------- -# `--shard` only buys anything if the workflow both PASSES it and creates more -# than one runner to pass it from. Either half alone is silently useless: a -# shard spec with no matrix runs leg 0 of 1 (everything, as before), and a -# matrix with no shard spec runs the whole selection on every leg. Pinned -# together, in the job that owns them. +# The plan's legs only buy anything if the workflow both sizes its matrix from +# them and has each leg run its own suites. Either half alone is silently +# useless: a plan read by no matrix runs one leg, and a matrix whose legs do +# not pick their own list runs nothing or everything. Pinned together, in the +# job that owns them. live_ci="$REPO_ROOT/.github/workflows/ci.yml" test_bash_block="$(awk ' /^ [A-Za-z_][A-Za-z0-9_-]*:[[:blank:]]*(#.*)?$/ { @@ -1426,13 +1426,14 @@ test_bash_block="$(awk ' job == "test-bash" ' "$live_ci")" # shellcheck disable=SC2016 # deliberate: these are workflow literals to match, not shell expansions. -if grep -q 'affected-tests\.sh --run --jobs 3 --shard "\$LEG/\$LEGS"' <<<"$test_bash_block" && - grep -q '^ strategy:' <<<"$test_bash_block" && +if grep -q 'run-plugin-tests\.sh --jobs 3 --suites-from "\$RUNNER_TEMP/leg-suites\.txt"' <<<"$test_bash_block" && + grep -q "leg: \${{ fromJSON(needs\.scope\.outputs\.bash_legs || '\[0\]') }}" <<<"$test_bash_block" && grep -q 'LEG: \${{ strategy\.job-index }}' <<<"$test_bash_block" && - grep -q 'LEGS: \${{ strategy\.job-total }}' <<<"$test_bash_block"; then - ok "ci.yml test-bash declares a matrix, runs three suites at a time, and passes the leg through to --shard" + grep -q 'PLAN: \${{ needs\.scope\.outputs\.bash_plan }}' <<<"$test_bash_block" && + grep -q "jq -r --arg leg \"\$LEG\" '\.\[\$leg\] // \[\] | \.\[\]' <<<\"\$PLAN\"" <<<"$test_bash_block"; then + ok "ci.yml test-bash sizes its matrix from the plan and runs each leg's suites three at a time" else - fail "ci.yml test-bash no longer fans the affected selection across a matrix at --jobs 3" + fail "ci.yml test-bash no longer fans the planned selection across a matrix at --jobs 3" fi # --- ci.yml never asks for more than the proven three ------------------------ diff --git a/scripts/check-docs-only-gate.sh b/scripts/check-docs-only-gate.sh index 7438298466..67c929d18e 100755 --- a/scripts/check-docs-only-gate.sh +++ b/scripts/check-docs-only-gate.sh @@ -58,7 +58,8 @@ # fail-open, because it replaces that step's real # outcome with `success` on every docs-only diff. # 7. NO JOB-LEVEL IF — a consumer carries no job-level condition other than -# the contract-only gate or its draft form, pinned +# the contract-only gate, its draft form, or a test +# lane's ordered-skip form (property 12), pinned # below as exact literals. It reaches the output through `needs`, so it # runs only when the resolver succeeded — which is what # makes the output's domain exactly {'true','false'} @@ -86,13 +87,22 @@ # .outputs. }}`, and never in a condition: its # value is not a polarity decision, and an empty one # must mean "the whole tree" to the script reading it. -# The one other read is the test-bash matrix size, -# pinned whole in MATRIX_READ with its four-leg +# The other reads are the test lanes' matrix sizes, +# each pinned whole in MATRIX_READS with its one-leg # default. # 11. A SKIP NEVER PASSES — the aggregate's `treat-skipped-as` is `fail`, a # draft's included: a draft runs no lane, and a green # draft ci-status is the newest one on the SHA from # the flip to ready until the lanes finish. +# 12. A SKIP IS CHECKED — a test lane `test-` may skip as a job on its own +# `run_` row (the ordered-skip form, the draft gate +# followed by `&& needs..outputs.run_ == +# 'true'`), and only when the aggregate reads its +# results from the step `id: lanes`, which turns a +# skip into `success` only where the resolver +# succeeded and that row is 'false' +# (scripts/ci-ordered-skips.test.sh runs that step). +# Any other skipped lane stays `skipped` and fails. # # FAIL CLOSED ON SHAPE. Like scripts/check-lane-coverage.sh, this reads the # workflow structurally rather than through a YAML library (the repo ships no @@ -145,14 +155,16 @@ DETECT_STEP_ID="detect" # # `run_full` is the root: it is the only row derived from the detector, and # every other row narrows it. The narrowing rows read `steps.match.outputs` -# (the change-detection action) through fromJSON with a `|| '{}'` default and -# compare `!= 'false'`, never `== 'true'`, so an unset group runs the lane. +# (the change-detection action) through fromJSON with a `|| '{}'` default, or +# the test-lane plan (`steps.plan.outputs`), and compare `!= 'false'`, never +# `== 'true'`, so an unset group or plan runs the lane. OUTPUT_NAME="run_full" OUTPUT_TABLE="\ run_full${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' }} run_tests${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && github.event.pull_request.draft != true }} -run_node${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['node'] != 'false' }} -run_python${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['python'] != 'false' }} +run_bash${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.bash != 'false' }} +run_python${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.python != 'false' }} +run_node${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.node != 'false' }} run_workflows${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && fromJSON(steps.match.outputs.results || '{}')['workflows'] != 'false' }} run_skill_checker${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['skill_checker'] != 'false' }} run_manifests${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && fromJSON(steps.match.outputs.results || '{}')['manifests'] != 'false' }}" @@ -160,14 +172,23 @@ run_manifests${TAB}\${{ steps.${DETECT_STEP_ID}.outputs.docs_only != 'true' && f # exact expression and read only as a whole env entry (property 10). `lane_base` # is the ref every diff-scoped step diffs against: `origin/` on a pull # request, the newest green push run's commit on a push, and empty (the whole -# tree) on a schedule, a dispatch, or a push with no usable base. +# tree) on a schedule, a dispatch, or a push with no usable base. The rest are +# the test-lane plan of scripts/plan-test-lanes.sh. DATA_TABLE="\ lane_base${TAB}\${{ steps.base.outputs.ref }} -test_legs${TAB}\${{ steps.legs.outputs.legs }} -test_needs${TAB}\${{ steps.legs.outputs.needs }}" -# The one data read that is not an env entry: test-bash sizes its matrix from -# `test_legs`, and an unset value falls back to the full four-leg fan-out. -MATRIX_READ="leg: \${{ fromJSON(needs.${RESOLVER_JOB}.outputs.test_legs || '[0,1,2,3]') }}" +bash_legs${TAB}\${{ steps.plan.outputs.bash_legs }} +bash_plan${TAB}\${{ steps.plan.outputs.bash_plan }} +bash_needs${TAB}\${{ steps.plan.outputs.bash_needs }} +python_legs${TAB}\${{ steps.plan.outputs.python_legs }} +python_plan${TAB}\${{ steps.plan.outputs.python_plan }} +python_needs${TAB}\${{ steps.plan.outputs.python_needs }} +node_packages${TAB}\${{ steps.plan.outputs.node_packages }}" +# The data reads that are not env entries: each sharded test lane sizes its +# matrix from its `_legs`. A leg reads its suites from the plan and fails +# on none, so an unset value falls back to one leg that fails loud. +MATRIX_READS="\ +leg: \${{ fromJSON(needs.${RESOLVER_JOB}.outputs.bash_legs || '[0]') }} +leg: \${{ fromJSON(needs.${RESOLVER_JOB}.outputs.python_legs || '[0]') }}" # The single required context. Everything reachable from its `needs` is a # REQUIRED lane, and that closure is what decides whether a job-level condition # is a defect (check 5c) and whether a lane may opt out of coverage (check 8). @@ -185,6 +206,13 @@ JOB_GATE="\${{ !(${CONTRACT_ONLY_PREDICATE}) }}" # The same gate with the draft term every lane carries: a draft pull request # runs no lane at all, and `ci-status` fails on the skipped lanes. JOB_GATE_DRAFT="\${{ !(${CONTRACT_ONLY_PREDICATE}) && github.event.pull_request.draft != true }}" +# A test lane's ordered skip (property 12): the draft gate and the lane's own +# row. `job_gate_skip ` is the one literal job `test-` may carry. +job_gate_skip() { printf '%s' "\${{ !(${CONTRACT_ONLY_PREDICATE}) && github.event.pull_request.draft != true && needs.${RESOLVER_JOB}.outputs.run_$1 == 'true' }}"; } +# The aggregate step that checks every skip against the resolver, by id, and +# the results input the aggregate must then read from it. +SKIP_CHECK_STEP_ID="lanes" +SKIP_CHECKED_RESULTS="\${{ steps.${SKIP_CHECK_STEP_ID}.outputs.results }}" REFERENCE_PREFIX="needs.${RESOLVER_JOB}.outputs." REFERENCE="${REFERENCE_PREFIX}${OUTPUT_NAME}" @@ -453,6 +481,11 @@ parsed="$( sub(/^[[:blank:]]+treat-skipped-as:[[:blank:]]*/, "", sk) print "SKIPAS\t" job "\t" uncommented(sk) } + if ($0 ~ /^[[:blank:]]+results:/) { + rv = $0 + sub(/^[[:blank:]]+results:[[:blank:]]*/, "", rv) + print "RESULTSIN\t" job "\t" uncommented(rv) + } if ($0 ~ /^ if:/ && mentions_output(uncommented($0))) { rest = $0 @@ -490,6 +523,7 @@ REC_STEPOUT="" REC_JOBIF="" REC_LANEOK="" REC_SKIPAS="" +REC_RESULTSIN="" while IFS= read -r line; do [[ -n "$line" ]] || continue case "$line" in @@ -508,6 +542,7 @@ while IFS= read -r line; do "JOBIF$TAB"*) REC_JOBIF+="${line#*"$TAB"}"$'\n' ;; "LANEOK$TAB"*) REC_LANEOK+="${line#*"$TAB"}"$'\n' ;; "SKIPAS$TAB"*) REC_SKIPAS+="${line#*"$TAB"}"$'\n' ;; + "RESULTSIN$TAB"*) REC_RESULTSIN+="${line#*"$TAB"}"$'\n' ;; # The awk pass emits no other record kind; a line that reaches here means the # two halves have drifted, which is not something to guess past. *) @@ -814,7 +849,7 @@ while IFS="$TAB" read -r refjob reford kind text; do *) # A data output, read as a whole env entry or as the matrix size # (property 10). - if is_data_read "$text" || [[ "$text" == "$MATRIX_READ" ]]; then + if is_data_read "$text" || has_line "$MATRIX_READS"$'\n' "$text"; then continue fi # An aggregator feed entry, for some table output X: @@ -972,16 +1007,46 @@ is_required() { [[ "$required_closure" == *$'\n'"$1"$'\n'* ]]; } # subtracts is a draft pull request, where every lane is skipped and the # aggregate fails. +# +# AND A TEST LANE'S ORDERED SKIP, `job_gate_skip ` on job `test-` only, +# whose `run_` row the table names: the draft form followed by +# `&& needs..outputs.run_ == 'true'`. Still no status-check +# function, so the needs edge governs; what it subtracts is a change that gives +# the lane no work. Check 12 holds the other half: the aggregate turns that +# skip into `success` only against the same row. + +skip_lanes="" while IFS= read -r refjob; do [[ -n "$refjob" ]] || continue is_required "$refjob" || continue while IFS="$TAB" read -r cjob ctext; do [[ "$cjob" == "$refjob" ]] || continue [[ "$ctext" == "$JOB_GATE" || "$ctext" == "$JOB_GATE_DRAFT" ]] && continue - report "NO JOB-LEVEL CONDITION ON A REQUIRED CONSUMER: job '$refjob' reads $OUTPUT_NAME, is reachable from ${AGGREGATE_JOB}.needs, and carries a job-level condition that is not the contract-only gate: if: $ctext. If that condition ever lets the job run when '$RESOLVER_JOB' did not succeed, $OUTPUT_NAME is the empty string, both sanctioned forms are false, and the lane reports success having run nothing. Gate the steps and let the needs edge decide whether the job runs at all. The only sanctioned job-level conditions here are exactly: $JOB_GATE, or $JOB_GATE_DRAFT" + if [[ "$refjob" == test-* ]] && table_has "run_${refjob#test-}" && + [[ "$ctext" == "$(job_gate_skip "${refjob#test-}")" ]]; then + skip_lanes+="$refjob"$'\n' + continue + fi + report "NO JOB-LEVEL CONDITION ON A REQUIRED CONSUMER: job '$refjob' reads $OUTPUT_NAME, is reachable from ${AGGREGATE_JOB}.needs, and carries a job-level condition that is not the contract-only gate: if: $ctext. If that condition ever lets the job run when '$RESOLVER_JOB' did not succeed, $OUTPUT_NAME is the empty string, both sanctioned forms are false, and the lane reports success having run nothing. Gate the steps and let the needs edge decide whether the job runs at all. The only sanctioned job-level conditions here are exactly: $JOB_GATE, or $JOB_GATE_DRAFT, or on a job test- whose run_ row the table names: $(job_gate_skip '')" done <<<"$REC_JOBIF" done <<<"$refjobs" +# --- 12. A SKIP IS CHECKED ----------------------------------------------------- +# +# A lane that may skip on its own row is only safe if the aggregate reads its +# result through the step that compares the skip with that row. Reading the raw +# `needs.*.result` instead would fail the skip (fail-closed, but every such +# change red); reading anything else could pass a skip nothing checked. +if [[ -n "$skip_lanes" ]]; then + agg_results="$(awk -F '\t' -v j="$AGGREGATE_JOB" '$1 == j { print $2 }' <<<"$REC_RESULTSIN")" + if [[ "$agg_results" != "$SKIP_CHECKED_RESULTS" ]]; then + report "A SKIP IS CHECKED: lane(s) [$(uniq_list "$skip_lanes")] skip on their own row, but job '$AGGREGATE_JOB' aggregates results: ${agg_results:-}. It must read exactly $SKIP_CHECKED_RESULTS, the step that passes a skip only where '$RESOLVER_JOB' said the lane had no work." + fi + if ! awk -F '\t' -v j="$AGGREGATE_JOB" -v id="$SKIP_CHECK_STEP_ID" '$1 == j && $3 == id { f = 1 } END { exit !f }' <<<"$REC_STEPID"; then + report "A SKIP IS CHECKED: lane(s) [$(uniq_list "$skip_lanes")] skip on their own row, but job '$AGGREGATE_JOB' has no step 'id: $SKIP_CHECK_STEP_ID' to check those skips against '$RESOLVER_JOB'." + fi +fi + # --- 11. A SKIPPED LANE NEVER PASSES ----------------------------------------- # # The draft gate skips every lane on a draft, and the aggregate fails it. A diff --git a/scripts/check-docs-only-gate.test.sh b/scripts/check-docs-only-gate.test.sh index 29c46ffc91..c3fd578c09 100755 --- a/scripts/check-docs-only-gate.test.sh +++ b/scripts/check-docs-only-gate.test.sh @@ -82,14 +82,20 @@ jobs: outputs: run_full: ${{ steps.detect.outputs.docs_only != 'true' }} run_tests: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true }} - run_node: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['node'] != 'false' }} - run_python: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['python'] != 'false' }} + run_bash: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.bash != 'false' }} + run_python: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.python != 'false' }} + run_node: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.node != 'false' }} run_workflows: ${{ steps.detect.outputs.docs_only != 'true' && fromJSON(steps.match.outputs.results || '{}')['workflows'] != 'false' }} run_skill_checker: ${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['skill_checker'] != 'false' }} run_manifests: ${{ steps.detect.outputs.docs_only != 'true' && fromJSON(steps.match.outputs.results || '{}')['manifests'] != 'false' }} lane_base: ${{ steps.base.outputs.ref }} - test_legs: ${{ steps.legs.outputs.legs }} - test_needs: ${{ steps.legs.outputs.needs }} + bash_legs: ${{ steps.plan.outputs.bash_legs }} + bash_plan: ${{ steps.plan.outputs.bash_plan }} + bash_needs: ${{ steps.plan.outputs.bash_needs }} + python_legs: ${{ steps.plan.outputs.python_legs }} + python_plan: ${{ steps.plan.outputs.python_plan }} + python_needs: ${{ steps.plan.outputs.python_needs }} + node_packages: ${{ steps.plan.outputs.node_packages }} # A comment INSIDE the job body, between two mapping keys. steps: - name: Check out @@ -486,6 +492,48 @@ expect "a draft-gated lane with no treat-skipped-as on the aggregate is rejected # The aggregate's own job-level condition is not a consumer's, and must stand. expect "the aggregate's own job-level condition is untouched" 0 "scope resolved once" --check "$base" +# --- 12. A SKIP IS CHECKED ---------------------------------------------------- +# +# A test lane may skip as a job on its own row, and only when the aggregate +# reads its results through the step that checks every skip against that row. +skip_form="${draft_gate% \}\}} && needs.scope.outputs.run_python == 'true' }}" +ordered="$scratch/ordered-skip.yml" +xform_insert_after "$skip_as_fail" " - gamma" " - test-python" "$scratch/o1.yml" +xform_replace_line "$scratch/o1.yml" " - name: Aggregate lane results" \ + " - name: Check each skipped lane against scope\n id: lanes\n run: echo checked\n - name: Aggregate lane results" "$scratch/o2.yml" +xform_insert_after "$scratch/o2.yml" " treat-skipped-as: fail" " results: \${{ steps.lanes.outputs.results }}" "$scratch/o3.yml" +xform_append "$scratch/o3.yml" " + test-python: + needs: [scope] + if: $skip_form + runs-on: ubuntu-24.04 + steps: + - name: Run the selected Python suites + run: echo python" "$ordered" +expect "a test lane skipping on its own row, checked by the aggregate, is allowed" 0 "scope resolved once" --check "$ordered" + +f="$scratch/ordered-skip-other-row.yml" +xform_replace_line "$ordered" " if: $skip_form" " if: ${skip_form/run_python/run_bash}" "$f" +expect "a test lane skipping on another lane's row is rejected" 1 "NO JOB-LEVEL CONDITION ON A REQUIRED CONSUMER" --check "$f" + +f="$scratch/ordered-skip-not-a-test-lane.yml" +xform_insert_after "$scratch/o3.yml" " gamma:" " if: $skip_form" "$f" +expect "the ordered skip on a job that is not test- is rejected" 1 "NO JOB-LEVEL CONDITION ON A REQUIRED CONSUMER" --check "$f" + +f="$scratch/ordered-skip-flipped.yml" +want_true="run_python == 'true'" +want_false="run_python == 'false'" +xform_replace_line "$ordered" " if: $skip_form" " if: ${skip_form/"$want_true"/"$want_false"}" "$f" +expect "the ordered skip with its polarity flipped is rejected" 1 "NO JOB-LEVEL CONDITION ON A REQUIRED CONSUMER" --check "$f" + +f="$scratch/ordered-skip-raw-results.yml" +xform_replace_line "$ordered" " results: \${{ steps.lanes.outputs.results }}" " results: \${{ join(needs.*.result, ' ') }}" "$f" +expect "an ordered skip the aggregate reads raw is rejected" 1 "A SKIP IS CHECKED" --check "$f" + +f="$scratch/ordered-skip-no-check-step.yml" +xform_delete "$ordered" " id: lanes" "$f" +expect "an ordered skip with no checking step is rejected" 1 "has no step 'id: lanes'" --check "$f" + # --- 6. EDGE DECLARED ------------------------------------------------------- f="$scratch/missing-edge.yml" @@ -580,11 +628,15 @@ xform_replace_line "$base" "DIFF_BASE: \${{" " diff_ref: \${{ needs.sco expect "a data output read under a non-env key is rejected" 1 "outside the aggregator feed template" --check "$f" f="$scratch/data-matrix.yml" -xform_replace_line "$sharded" " matrix: \${{" " matrix:\n leg: \${{ fromJSON(needs.scope.outputs.test_legs || '[0,1,2,3]') }}" "$f" -expect "the matrix sized from test_legs with its four-leg default is allowed" 0 "scope resolved once" --check "$f" +xform_replace_line "$sharded" " matrix: \${{" " matrix:\n leg: \${{ fromJSON(needs.scope.outputs.bash_legs || '[0]') }}" "$f" +expect "the matrix sized from bash_legs with its one-leg default is allowed" 0 "scope resolved once" --check "$f" + +f="$scratch/data-matrix-python.yml" +xform_replace_line "$sharded" " matrix: \${{" " matrix:\n leg: \${{ fromJSON(needs.scope.outputs.python_legs || '[0]') }}" "$f" +expect "the matrix sized from python_legs with its one-leg default is allowed" 0 "scope resolved once" --check "$f" f="$scratch/data-matrix-no-default.yml" -xform_replace_line "$sharded" " matrix: \${{" " matrix:\n leg: \${{ fromJSON(needs.scope.outputs.test_legs) }}" "$f" +xform_replace_line "$sharded" " matrix: \${{" " matrix:\n leg: \${{ fromJSON(needs.scope.outputs.bash_legs) }}" "$f" expect "the matrix read without its default is rejected" 1 "outside the aggregator feed template" --check "$f" f="$scratch/data-in-job-if.yml" @@ -599,7 +651,7 @@ expect "a step gated on an output outside the table is rejected" 1 "which the re # against 'true' rather than 'false' skips the lane when detection is unset, # which is the direction that hides a regression. f="$scratch/inverted-narrowing.yml" -xform_replace_line "$base" " run_node:" " run_node: \${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && fromJSON(steps.match.outputs.results || '{}')['node'] == 'true' }}" "$f" +xform_replace_line "$base" " run_node:" " run_node: \${{ steps.detect.outputs.docs_only != 'true' && github.event.pull_request.draft != true && steps.plan.outputs.node == 'true' }}" "$f" expect "a narrowing row comparing against 'true' is rejected" 1 "FAIL-CLOSED DEFAULT" --check "$f" # The whole-corpus check-25 row keeps the draft term run_tests carries, so a diff --git a/scripts/ci-ordered-skips.test.sh b/scripts/ci-ordered-skips.test.sh new file mode 100755 index 0000000000..dce7a75e34 --- /dev/null +++ b/scripts/ci-ordered-skips.test.sh @@ -0,0 +1,86 @@ +#!/usr/bin/env bash +# Runs the `Check each skipped lane against scope` step body of +# .github/workflows/ci.yml's ci-status under the shell Actions gives a step with +# no `shell:` (bash -e), against `toJSON(needs)` fixtures. +# +# A test lane `test-` that skipped turns into `success` only where `scope` +# succeeded and its `run_` row is exactly 'false'. Every other result passes +# through unchanged, so a skip that hid work stays `skipped` and the aggregate +# (treat-skipped-as: fail) reds it. +# test-scope: .github/workflows/ci.yml +set -uo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +WORKFLOW="$ROOT/.github/workflows/ci.yml" + +# shellcheck source=lib/test-harness.sh +. "$ROOT/scripts/lib/test-harness.sh" + +TMP_ROOT="$(mktemp -d)" +trap 'rm -rf "$TMP_ROOT"' EXIT + +step="$(awk '/^ - name: Check each skipped lane against scope$/ { on = 1; print; next } + on && /^ - / { exit } + on { print }' "$WORKFLOW")" +awk '/^ run: \|$/ { on = 1; next } + on && /^ / { print substr($0, 11); next } + on && /^[[:blank:]]*$/ { print ""; next } + on { exit }' <<<"$step" >"$TMP_ROOT/body.sh" + +if [[ ! -s "$TMP_ROOT/body.sh" ]]; then + fail "no 'Check each skipped lane against scope' step with a run: | body in $WORKFLOW" + test_harness::report + exit 1 +fi +for line in "id: lanes" "LANES: \${{ toJSON(needs) }}" "if: \${{ !cancelled() }}"; do + if [[ "$step" == *"$line"* ]]; then ok "step carries '$line'"; else fail "step lost '$line'"; fi +done +# shellcheck disable=SC2016 # a workflow literal to match, not a shell expansion +if grep -q 'results: \${{ steps.lanes.outputs.results }}' "$WORKFLOW"; then + ok "the aggregate reads the checked results" +else + fail "the aggregate does not read steps.lanes.outputs.results" +fi + +# lanes +lanes() { + printf '{"scope":{"result":"%s","outputs":{"run_bash":"%s","run_python":"%s","run_node":"%s"}},' "$1" "$2" "$3" "$4" + printf '"lint-repo":{"result":"success","outputs":{}},' + printf '"test-bash":{"result":"%s","outputs":{}},"test-python":{"result":"%s","outputs":{}},' "$5" "$6" + printf '"test-node":{"result":"%s","outputs":{}}}' "$7" +} + +# expect