diff --git a/.github/bench/README.md b/.github/bench/README.md index ca456f0b..a9e4d742 100644 --- a/.github/bench/README.md +++ b/.github/bench/README.md @@ -51,9 +51,15 @@ entries, and the unit tests pin the semantics. 1. Dispatch **Benchmark A/B** with `mode=aa` (same ref twice) at least 3 times; each posts a noise report to the run summary and artifact. -2. B/op floor must be ~0. If it isn't, the reduced profile has too little - warmup for C2/escape analysis — raise `-wi` in `JMH_FLAGS` / - `JMH_PROFILE` in bench-pr.yml and recalibrate. +2. B/op floor must be ~0. If it isn't: first pin the fork JVM's + allocation ergonomics — adaptive TLAB sizing makes `gc.alloc.rate.norm` + count environment-dependent retire waste (measured on PR #116: + ±78-81% A/A swings on `PlatedBench.visitorUniverseJson n=4096` + unpinned, ±0.0% with `-jvmArgsAppend -XX:-ResizeTLAB` plus a fixed + `-Xms`/`-Xmx`; fork counts do NOT fix it — `-f 3` unpinned still + swung ±11-41%). Then raise `-wi` in `JMH_FLAGS` / `JMH_PROFILE` in + bench-pr.yml and recalibrate. Changing the flags changes the profile — + see step 4. 3. Commit `.github/bench/thresholds.json`, e.g. `{"bop_regression_pct": 1.0, "bop_min_delta_bytes": 16}` — quantile data from the reports, not guesses. diff --git a/.github/bench/bench_tools.py b/.github/bench/bench_tools.py index ecefaf40..b78b2063 100644 --- a/.github/bench/bench_tools.py +++ b/.github/bench/bench_tools.py @@ -488,8 +488,11 @@ def q(vals, frac): ) lines += [ "", - "B/op max should be ~0. If it isn't, the reduced profile has too " - "little warmup for C2/escape analysis — bump `-wi` before any gate.", + "B/op max should be ~0. If it isn't: pin the fork JVM's allocation " + "ergonomics first (-jvmArgsAppend -XX:-ResizeTLAB plus a fixed " + "-Xms/-Xmx — adaptive TLAB sizing makes B/op count " + "environment-dependent retire waste), then bump `-wi` before any " + "gate.", ] return "\n".join(lines) diff --git a/.github/workflows/bench-pr.yml b/.github/workflows/bench-pr.yml index 93cb4fec..aac5578c 100644 --- a/.github/workflows/bench-pr.yml +++ b/.github/workflows/bench-pr.yml @@ -54,8 +54,16 @@ env: # The measurement profile. thresholds.json is calibrated against exactly # this string — change it and the gate reverts to advisory until an A/A # recalibration (delete thresholds.json in the same PR). - JMH_PROFILE: "pr:-i3-wi2-f1-t1-gc" - JMH_FLAGS: "-i 3 -wi 2 -f 1 -t 1 -foe true -prof gc -rf json" + # + # Pinned allocation ergonomics (-XX:-ResizeTLAB + fixed heap): adaptive + # TLAB sizing makes gc.alloc.rate.norm count environment-dependent retire + # waste — PR #116 measured A/A swings of +78-81% B/op on + # PlatedBench.visitorUniverseJson n=4096 across identical-source runs; + # pinning took the same A/A to ±0.0%. Forks don't fix it (measured: -f 3 + # unpinned still ±11-41% — it is per-invocation environment state, not + # per-fork sampling noise). + JMH_PROFILE: "pr:-i3-wi2-f1-t1-gc-rt1g" + JMH_FLAGS: "-i 3 -wi 2 -f 1 -t 1 -foe true -prof gc -rf json -jvmArgsAppend -XX:-ResizeTLAB -jvmArgsAppend -Xms1g -jvmArgsAppend -Xmx1g" jobs: bench-ab: diff --git a/.github/workflows/bench-sweep.yml b/.github/workflows/bench-sweep.yml index c2d025d1..5b74994a 100644 --- a/.github/workflows/bench-sweep.yml +++ b/.github/workflows/bench-sweep.yml @@ -58,8 +58,11 @@ concurrency: cancel-in-progress: false env: - JMH_PROFILE: "sweep:-i5-wi3-f3-t1-gc" - JMH_FLAGS: "-i 5 -wi 3 -f 3 -t 1 -foe true -prof gc -rf json" + # Pinned allocation ergonomics — see bench-pr.yml: adaptive TLAB sizing + # makes B/op count environment-dependent retire waste; -XX:-ResizeTLAB + + # a fixed heap made the flagged families' A/A B/op deterministic. + JMH_PROFILE: "sweep:-i5-wi3-f3-t1-gc-rt1g" + JMH_FLAGS: "-i 5 -wi 3 -f 3 -t 1 -foe true -prof gc -rf json -jvmArgsAppend -XX:-ResizeTLAB -jvmArgsAppend -Xms1g -jvmArgsAppend -Xmx1g" jobs: sweep: