From 457e2d6e161649be83c4bed33edb849295553e3e Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Fri, 2 Oct 2026 09:43:50 +0800 Subject: [PATCH 1/4] docs(verdicts): RIDE displacement arm separated-negative - residual closed on evidence --- .../05-ride-sft-residual-extrapolation.md | 33 ++++++++++++++++--- docs/RESULTS.md | 30 +++++++++++++++++ docs/paper.adoc | 2 +- 3 files changed, 59 insertions(+), 6 deletions(-) diff --git a/TODO.sota-2026/05-ride-sft-residual-extrapolation.md b/TODO.sota-2026/05-ride-sft-residual-extrapolation.md index 746343b..2b698bc 100644 --- a/TODO.sota-2026/05-ride-sft-residual-extrapolation.md +++ b/TODO.sota-2026/05-ride-sft-residual-extrapolation.md @@ -1,6 +1,6 @@ # 05 — RIDE-style SFT-residual extrapolation: probe-first arm -Status: PROBE PASSED (2026-10-01) — training arm spec'd below, launch pending owner call +Status: CLOSED — arm measured SEPARATED-NEGATIVE (2026-10-02) Literature basis: RIDE (arXiv 2609.36484) — extrapolate the teacher-over-base residual directly in representation space: student hidden states regressed toward @@ -71,10 +71,7 @@ Artefact: rababa-checkpoints:/ride_probe_r6_r7.json. ## Training-arm spec (gated on this probe; launch = owner decision) -The closure rule (TODO 03) permits this arm: it is a mechanism novel -to the ledger (representation-space displacement; all prior student -levers were loss/data/optimizer-side) and now has a measured transfer -premise. +*(Executed 2026-10-01 under "Proceed all" — run-016-ride, PR #233.)* - Infra: feature-regression aux loss in modal_distill — teacher/base hidden states must be cached per layer subset. Restrict to L0–L8 @@ -87,3 +84,29 @@ premise. Muon, seed 42); adopt gate ≥ 0.3pp DER improvement (E4-style bar). - Est. build: teacher/base hidden-state dump (one-off Modal job, ~1h A100) + trainer loss path + spec; run cost ≈ one 2.1-recipe arm. + +## Arm result (measured 2026-10-02) — SEPARATED-NEGATIVE, arm CLOSED + +run-016-ride: the 2.1 recipe verbatim (r7 labels, Muon, 6 epochs, +seed 42) + encoder-hidden regression toward ridge-projected displaced +targets (λ=1.0, layers 0–8, β auto-calibrated 3.576e-05 = 10% of CE +at start, ridge fit on 16 batches of mask-flattened positions; +ride.pt checkpointed). Training converged normally (CE 0.57→0.29 over +13,026 steps); teacher reproduced at 2.2921 on the same eval. + +| measure | value | +|---|---| +| student DER-CE (full 1,200) | **5.8627** | +| vs 2.1 rung 4.5701 | **+1.3926pp [1.156, 1.646], p=0.0** — separated | +| delta vs teacher | 3.5038 [3.222, 3.792] | + +**Verdict: NEGATIVE — decisively.** The direction-transferability +probe passed (max cos 0.9513), the premise was mechanistically sound, +training was healthy — and the outcome still hurt by 1.4pp. Reading: +domain-general direction is necessary but not sufficient; regressing +a 300M byte student's encoder toward ridge-projected 580M targets +DISPLACES representations the decoder relies on, competing with the +CE objective rather than sharpening it. This is ledger row #9 and the +strongest test of the closure rule to date: the one student-side arm +with a measured mechanistic premise still failed. The student-side +residual is closed on evidence, not exhaustion. No further arms. diff --git a/docs/RESULTS.md b/docs/RESULTS.md index 5845a7b..de150ce 100644 --- a/docs/RESULTS.md +++ b/docs/RESULTS.md @@ -972,3 +972,33 @@ seeds. The headwise-Muon separated-negative (+0.2267pp, p=0.017) is directionally consistent for its size class, but the seed axis is unmeasured. Future arms run multi-seed or carry this caveat. Ship decisions are unaffected — the base recipe shipped on its own merits. + +## RIDE displacement arm — SEPARATED-NEGATIVE; student-side ledger closed on evidence (2026-10-02) + +The one student-side lever that passed a mechanistic probe still +failed at the outcome. Probe (2026-10-01): the r7−r6 SFT residual +direction is domain-general in encoder layers 0–8 (cos 0.94/0.95/0.92 +… decaying to noise L11+; max 0.9513 vs the 0.5 kill bar). Arm +(run-016-ride): the 2.1 recipe verbatim + encoder-hidden regression +toward ridge-projected displaced targets +h_t + λ(h_t − h_b) (λ=1.0, layers 0–8, frozen r7 teacher + r6 base +both resident, β auto-calibrated to 10% of CE at start — 3.576e-05). +Training converged normally (CE 0.57→0.29, 13,026 steps); teacher +reproduced at 2.2921. + +| measure | value | +|---|---| +| run-016 DER-CE (full 1,200) | **5.8627** | +| paired Δ vs 2.1 (4.5701) | **+1.3926pp [1.156, 1.646], p=0.0** | +| Δ vs teacher | 3.5038 [3.222, 3.792] | + +Reading: a domain-general direction is necessary but not sufficient — +regressing a 300M byte student's encoder toward ridge-projected 580M +targets displaces representations the decoder relies on, competing +with the CE objective instead of sharpening it. Ledger row #9; the +residual is now closed **on evidence, not exhaustion**: corpus scale, +register mix, on-policy GKD, PKM memory, epochs, headwise Muon, +Sinkhorn embeddings, engram memory, and representation displacement +(the only arm with a measured mechanistic premise) have all been run +to verdict. The frontier mover remains teacher-side data only — +run-009-yallamorph (TODO.sota-2026/01) is the active lever. diff --git a/docs/paper.adoc b/docs/paper.adoc index 7bb9e0c..f839c49 100644 --- a/docs/paper.adoc +++ b/docs/paper.adoc @@ -303,7 +303,7 @@ zip through the runtime itself — sha-pinned to the index) agree on re-derives exactly by the same tooling from the published prediction files, which is how the discrepancy was caught. -Two structural findings sit on it, one now scoped by a cross-lingual counterexample. First, *pretrained width is load-bearing*: SVD width-stitching of the pretrained ByT5-small fails at both tested ratios (3.8× narrow: 82.96 — worse than the from-scratch collapse; 2× narrow: 78.23). *Depth, by contrast, was compressible under our Arabic recipe*: a verbatim layer copy of encoder layers 12→6 trained to 5.78 full-set — a shippable rung at 63% of the parameters whose 1.21pp depth cost at 6 epochs is CI-separated from the full-depth peer. The Hebrew replication of that depth cut, single-variable against its own full-depth lineage (logit-KD recipe), collapsed instead: 77.48 DER versus the full-depth 30.38 (+53.77pp [51.64, 55.92]) despite normal training convergence. Depth-compressibility is therefore *not* a universal property of pretrained ByT5-small — it held under sequence-KD with Muon on Arabic and catastrophically failed under logit-KD on Hebrew; whether the boundary is the distillation regime or the language is open. Width surgery destroys the pretrained representation; depth surgery spends it — but only where the training regime lets it. Second, the epochs lever is real but secondary: doubling 3→6 epochs moves the rung 4.82→4.57 (−0.25pp) — most of the residual is not undertraining. Two pre-registered causal tests closed the domain-coverage attribution. The swap direction — 8k news-domain units replaced by classical-register Tashkeela at constant 30k budget — came back *negative* (5.81, −0.98pp vs control). The add direction — 48k total with the full cleaned Tashkeela corpus (5× classical coverage, all other levers held, 39,018 steps) — came back *flat-negative*: 4.8231 full-set, delta vs teacher 2.3717 [2.194, 2.554], statistically indistinguishable from the 2.1 rung (4.5701 [1.91, 2.35]) with the point estimate 0.25pp worse. A third lever, on-policy distillation (GKD — training on student-generated mistakes scored by the teacher), also came back negative: 6.0036 [3.109, 3.743], 1.43pp *worse* than the off-policy rung it was meant to improve. The residual has now resisted every lever tested — corpus scale, register mix, on-policy correction, memory layers (real but 0.70pp), epochs (0.25pp), and two optimizer-recipe arms imported from the 2025 frontier-LLM literature — and we report it as a property of the compression itself rather than a shortfall of any single method. The optimizer-recipe arms are instructive because they transfer negatively at our scale: head-wise Muon on Q/K projections (reported positive at 671B scale in DeepSeek-V4.1-Flash) scores 4.8164, separated-worse than its vanilla peer by +0.2267pp [0.016, 0.411] under a paired between-students bootstrap on identical data; the Sinkhorn-balanced embedding update is statistically indistinguishable from AdamW (−0.0903pp [−0.267, 0.069]). One measurement caveat travels with all such arm verdicts: each is a single training seed, and a recent small-model distillation audit shows per-seed variance large enough to swallow sub-point deltas — with bimodal collapse in some KD variants — so our paired bootstrap's protection extends to prediction resampling but not the seed axis (Sumit et al. 2026, arXiv 2608.27729); future arms run multi-seed or carry this caveat. Both directions of the classical-corpus lever fail: the residual is *not* a classical-domain coverage deficit, and it is not an optimizer artifact. It reframes as a teacher–student interaction the corpus cannot reach — the remaining lever on our record is teacher-side: every frontier move in this table's upper rows came from the teacher's data, not the student's training. The next teacher rung is in flight on exactly that axis: systematic morphological paradigm coverage from the YallaMorph/CamelMorph resource (Reda et al. 2026, arXiv 2609.10153 — 663,804 controlled morphological-generation instances) added as an auxiliary stream to the news-domain mix that produced the current rung. +Two structural findings sit on it, one now scoped by a cross-lingual counterexample. First, *pretrained width is load-bearing*: SVD width-stitching of the pretrained ByT5-small fails at both tested ratios (3.8× narrow: 82.96 — worse than the from-scratch collapse; 2× narrow: 78.23). *Depth, by contrast, was compressible under our Arabic recipe*: a verbatim layer copy of encoder layers 12→6 trained to 5.78 full-set — a shippable rung at 63% of the parameters whose 1.21pp depth cost at 6 epochs is CI-separated from the full-depth peer. The Hebrew replication of that depth cut, single-variable against its own full-depth lineage (logit-KD recipe), collapsed instead: 77.48 DER versus the full-depth 30.38 (+53.77pp [51.64, 55.92]) despite normal training convergence. Depth-compressibility is therefore *not* a universal property of pretrained ByT5-small — it held under sequence-KD with Muon on Arabic and catastrophically failed under logit-KD on Hebrew; whether the boundary is the distillation regime or the language is open. Width surgery destroys the pretrained representation; depth surgery spends it — but only where the training regime lets it. Second, the epochs lever is real but secondary: doubling 3→6 epochs moves the rung 4.82→4.57 (−0.25pp) — most of the residual is not undertraining. Two pre-registered causal tests closed the domain-coverage attribution. The swap direction — 8k news-domain units replaced by classical-register Tashkeela at constant 30k budget — came back *negative* (5.81, −0.98pp vs control). The add direction — 48k total with the full cleaned Tashkeela corpus (5× classical coverage, all other levers held, 39,018 steps) — came back *flat-negative*: 4.8231 full-set, delta vs teacher 2.3717 [2.194, 2.554], statistically indistinguishable from the 2.1 rung (4.5701 [1.91, 2.35]) with the point estimate 0.25pp worse. A third lever, on-policy distillation (GKD — training on student-generated mistakes scored by the teacher), also came back negative: 6.0036 [3.109, 3.743], 1.43pp *worse* than the off-policy rung it was meant to improve. The residual has now resisted every lever tested — corpus scale, register mix, on-policy correction, memory layers (real but 0.70pp), epochs (0.25pp), and two optimizer-recipe arms imported from the 2025 frontier-LLM literature — and we report it as a property of the compression itself rather than a shortfall of any single method. The optimizer-recipe arms are instructive because they transfer negatively at our scale: head-wise Muon on Q/K projections (reported positive at 671B scale in DeepSeek-V4.1-Flash) scores 4.8164, separated-worse than its vanilla peer by +0.2267pp [0.016, 0.411] under a paired between-students bootstrap on identical data; the Sinkhorn-balanced embedding update is statistically indistinguishable from AdamW (−0.0903pp [−0.267, 0.069]). One measurement caveat travels with all such arm verdicts: each is a single training seed, and a recent small-model distillation audit shows per-seed variance large enough to swallow sub-point deltas — with bimodal collapse in some KD variants — so our paired bootstrap's protection extends to prediction resampling but not the seed axis (Sumit et al. 2026, arXiv 2608.27729); future arms run multi-seed or carry this caveat. Both directions of the classical-corpus lever fail: the residual is *not* a classical-domain coverage deficit, and it is not an optimizer artifact. A final arm tested the residual in representation space, on the one mechanism whose premise we could measure before spending training compute: the teacher's news-domain fine-tune shifts its encoder representations in a direction that is domain-general in early and mid layers (cosine 0.59–0.95 between the shift measured on classical vs news text, layers 0–8), so we regressed the student's encoder hidden states toward ridge-projected, extrapolated teacher targets along that direction — and it was the *most* harmful lever of all (5.8627, +1.3926pp [1.156, 1.646] worse than its vanilla peer), despite healthy training and a pre-registered loss budget. A domain-general direction is necessary but not sufficient: pulling a 300M byte student's representations toward projected 580M targets displaces what its decoder relies on. With that, the residual is closed on evidence rather than exhaustion. It reframes as a teacher–student interaction the corpus cannot reach — the remaining lever on our record is teacher-side: every frontier move in this table's upper rows came from the teacher's data, not the student's training. The next teacher rung is in flight on exactly that axis: systematic morphological paradigm coverage from the YallaMorph/CamelMorph resource (Reda et al. 2026, arXiv 2609.10153 — 663,804 controlled morphological-generation instances) added as an auxiliary stream to the news-domain mix that produced the current rung. == The decode protocol is part of the measurement [[section-decode]] From 7e1104f04b695d7ba23bb8d92ef0b2510f6326db Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Fri, 2 Oct 2026 09:46:40 +0800 Subject: [PATCH 2/4] fix(tests): guard test_ride torch import with importorskip - CI runners have no torch --- tests/test_ride.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/test_ride.py b/tests/test_ride.py index 14a0c0b..bb90c9d 100644 --- a/tests/test_ride.py +++ b/tests/test_ride.py @@ -4,7 +4,10 @@ from pathlib import Path import sys -import torch +import pytest + +pytest.importorskip("torch") +import torch # noqa: E402 sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src")) From 31b88070670b627c06b94d3a29b35a2062f343a2 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Fri, 2 Oct 2026 09:49:05 +0800 Subject: [PATCH 3/4] fix(lint): ride test/module ruff - import order, line length --- src/gpu/ride.py | 4 +++- tests/test_ride.py | 5 +++-- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/src/gpu/ride.py b/src/gpu/ride.py index 5eef8c6..99cedbe 100644 --- a/src/gpu/ride.py +++ b/src/gpu/ride.py @@ -22,7 +22,9 @@ def displaced_targets(h_teacher: torch.Tensor, h_base: torch.Tensor, lam: float) return h_teacher + lam * (h_teacher - h_base) -def masked_mse(pred: torch.Tensor, target: torch.Tensor, attention_mask: torch.Tensor) -> torch.Tensor: +def masked_mse( + pred: torch.Tensor, target: torch.Tensor, attention_mask: torch.Tensor +) -> torch.Tensor: """Mean squared error over kept positions and all dims. pred/target: (B, T, D); attention_mask: (B, T), 1 = kept. diff --git a/tests/test_ride.py b/tests/test_ride.py index bb90c9d..2bd2ea1 100644 --- a/tests/test_ride.py +++ b/tests/test_ride.py @@ -1,8 +1,8 @@ """Tests for the RIDE displacement-arm math (TODO.sota-2026/05).""" +import sys import unittest from pathlib import Path -import sys import pytest @@ -23,7 +23,8 @@ def test_lambda_zero_is_teacher(self): def test_lambda_one_extrapolates(self): h_t = torch.tensor([[2.0, 4.0]]) h_b = torch.tensor([[0.0, 0.0]]) - self.assertTrue(torch.equal(displaced_targets(h_t, h_b, lam=1.0), torch.tensor([[4.0, 8.0]]))) + got = displaced_targets(h_t, h_b, lam=1.0) + self.assertTrue(torch.equal(got, torch.tensor([[4.0, 8.0]]))) def test_fractional_lambda(self): h_t = torch.tensor([[3.0]]) From 29c1fd84c75057b646a744d09f37d25d3d2bc567 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Fri, 2 Oct 2026 12:03:05 +0800 Subject: [PATCH 4/4] fix(lint): drop unused fit_ridge import in ride setup --- src/gpu/modal_distill.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/gpu/modal_distill.py b/src/gpu/modal_distill.py index dfc90d6..2c9df11 100644 --- a/src/gpu/modal_distill.py +++ b/src/gpu/modal_distill.py @@ -797,7 +797,7 @@ def distill_sequence(spec_id: str, epochs: int = 3) -> dict: # Regress student encoder hiddens toward ridge-projected # h_t + lam*(h_t - h_b) from frozen teacher (r7) and base (r6). _ensure_src_path() - from gpu.ride import displaced_targets, fit_ridge, masked_mse + from gpu.ride import displaced_targets, masked_mse base_path = str( Path(VOLUME_MOUNTS[ride_cfg.get("base_volume", teacher_vol)]) / ride_cfg["base"]