From 3915cdda0a3d9516c762d2be719699d3969d7e73 Mon Sep 17 00:00:00 2001 From: Thomas Juul Dyhr Date: Wed, 30 Sep 2026 16:28:03 +0200 Subject: [PATCH 1/2] fix(ci): promote performance baseline only on successful runs The "Save new baseline" and "Cache updated baseline for next run" steps in performance.yml ran under always(), so a weekly run that failed the >20% regression check still cached its slower numbers as the next baseline. Each regression lowered the bar for the following run: the 2026-09-13 run compared `apps` against 2.732s, the exact figure from the failed 2026-09-06 run, failed again, and cached 4.257s -- which let the 2026-09-20 run pass trivially. Gate both steps on success() instead. A deliberate re-baseline is done by deleting the perf-baseline-macOS-* caches: compare_performance.py exits 0 when no baseline exists, so the next run becomes the baseline. Add test_performance_baseline_promoted_only_on_success to TestProjectConsistency to guard the step conditions and their ordering after the regression check. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/performance.yml | 10 ++++++++-- CHANGELOG.md | 9 +++++++++ tests/test_project_consistency.py | 20 ++++++++++++++++++++ 3 files changed, 37 insertions(+), 2 deletions(-) diff --git a/.github/workflows/performance.yml b/.github/workflows/performance.yml index ac79e91e..5ac4b8e9 100644 --- a/.github/workflows/performance.yml +++ b/.github/workflows/performance.yml @@ -117,12 +117,18 @@ jobs: # results for just that one benchmark, silently disabling # regression checks for the other benchmarks until the next # full run happens to succeed. - if: always() && (github.event.inputs.benchmark_target == 'all' || github.event.inputs.benchmark_target == '') + # success(), not always(): a run that fails the regression check must + # not become the next baseline, or each regression lowers the bar for + # the run after it. To accept a slowdown deliberately, delete the + # perf-baseline-macOS-* caches (`gh cache list --key perf-baseline-`, + # then `gh cache delete `); the next run has no baseline, passes, + # and becomes the new one. + if: success() && (github.event.inputs.benchmark_target == 'all' || github.event.inputs.benchmark_target == '') run: cp performance_results.json performance_baseline.json - name: Cache updated baseline for next run uses: actions/cache/save@v6 - if: always() && (github.event.inputs.benchmark_target == 'all' || github.event.inputs.benchmark_target == '') + if: success() && (github.event.inputs.benchmark_target == 'all' || github.event.inputs.benchmark_target == '') with: path: performance_baseline.json key: perf-baseline-${{ runner.os }}-${{ github.run_id }} diff --git a/CHANGELOG.md b/CHANGELOG.md index ebadb965..1f29ed91 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Fixed +- **Weekly performance benchmark promoted failing runs to baseline**: the `Save new baseline` and + `Cache updated baseline for next run` steps in `.github/workflows/performance.yml` ran under `always()`, so a + run that failed the >20% regression check still cached its slower numbers as the next run's baseline — each + regression lowered the bar for the following week (e.g. the 2026-09-13 run compared `apps` against 2.732s, + the exact figure from the failed 2026-09-06 run). Both steps are now gated on `success()`; to accept a + slowdown deliberately, delete the `perf-baseline-macOS-*` caches and the next run re-baselines. Guarded by + `test_performance_baseline_promoted_only_on_success`. + ## [1.2.0] - 2026-08-15 ### Security diff --git a/tests/test_project_consistency.py b/tests/test_project_consistency.py index 1817a468..9f1eaa71 100644 --- a/tests/test_project_consistency.py +++ b/tests/test_project_consistency.py @@ -181,6 +181,26 @@ def test_ci_python_versions(self): # Validate all supported versions are tested self._validate_ci_versions(supported_versions, ci_versions) + def test_performance_baseline_promoted_only_on_success(self): + """Test that a run failing the regression check never becomes the next baseline. + + With ``always()``, each failing run's slower numbers were cached as the + next baseline, so every regression lowered the bar for the following run. + """ + perf_path = get_project_root() / ".github" / "workflows" / "performance.yml" + with open(perf_path, encoding="utf-8") as f: + perf_config = yaml.safe_load(f) + + step_names = [step.get("name") for step in perf_config["jobs"]["performance-test"]["steps"]] + steps = dict(zip(step_names, perf_config["jobs"]["performance-test"]["steps"], strict=True)) + compare_index = step_names.index("Compare with baseline (fail on >20% regression)") + + for name in ("Save new baseline", "Cache updated baseline for next run"): + condition = steps[name]["if"] + assert step_names.index(name) > compare_index, f"'{name}' must run after the regression check" + assert "always()" not in condition, f"'{name}' must not run when the regression check fails" + assert condition.startswith("success()"), f"'{name}' should be gated on success()" + def _assert_version_in_ci(self, version: str, ci_versions: list[str]) -> None: """Assert that a specific version is tested in CI.""" error_msg = f"Python {version} is supported but not tested in CI" From 5c0d70c3e728d4aa65c7b714fe40a70e1c3a74bc Mon Sep 17 00:00:00 2001 From: Thomas Juul Dyhr Date: Wed, 30 Sep 2026 19:46:53 +0200 Subject: [PATCH 2/2] Update .github/workflows/performance.yml Co-authored-by: sourcery-ai[bot] <58596630+sourcery-ai[bot]@users.noreply.github.com> --- .github/workflows/performance.yml | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/.github/workflows/performance.yml b/.github/workflows/performance.yml index 5ac4b8e9..5bfd17f7 100644 --- a/.github/workflows/performance.yml +++ b/.github/workflows/performance.yml @@ -120,8 +120,7 @@ jobs: # success(), not always(): a run that fails the regression check must # not become the next baseline, or each regression lowers the bar for # the run after it. To accept a slowdown deliberately, delete the - # perf-baseline-macOS-* caches (`gh cache list --key perf-baseline-`, - # then `gh cache delete `); the next run has no baseline, passes, + # perf-baseline-macOS-* caches (`gh cache list --limit 100 --json key --jq '.[] | select(.key | startswith("perf-baseline-macOS-")) | .key' | while read -r key; do gh cache delete "$key"; done`); the next run has no baseline, passes, # and becomes the new one. if: success() && (github.event.inputs.benchmark_target == 'all' || github.event.inputs.benchmark_target == '') run: cp performance_results.json performance_baseline.json