Repository navigation
neural-parity #7
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: neural-parity | |
| # Cross-runtime parity on the neural golden corpus (golden-v2+). | |
| # Quantized contract: per-row agreement >= 99.5%, corpus >= 99.9%. | |
| # Python is the reference (golden generator); ruby and ts run the same | |
| # sets. | |
| on: | |
| workflow_dispatch: | |
| inputs: | |
| model: | |
| description: "model id (plane family)" | |
| required: true | |
| default: ara-diac-plane-1.0 | |
| corpus_bound: | |
| description: "ruby corpus bound override (K=3 base artifacts skew; see RESULTS.md WO18)" | |
| required: false | |
| default: "" | |
| permissions: | |
| contents: read | |
| jobs: | |
| ruby: | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: ruby/setup-ruby@v1 | |
| with: | |
| ruby-version: "3.4" | |
| bundler-cache: false | |
| - run: git clone --depth 1 https://github.com/interscript/interscript-ruby /tmp/rt | |
| - name: run the golden set | |
| working-directory: /tmp/rt | |
| run: | | |
| bundle install --jobs 4 | |
| mkdir -p /tmp/golden && curl -sL "https://github.com/interscript/interscript-models/releases/download/golden-v2/${MODEL}.jsonl" -o /tmp/golden/set.jsonl | |
| curl -sL "https://github.com/interscript/interscript-models/releases/download/${MODEL}/${MODEL}.zip" -o /tmp/model.zip | |
| SECRYST_E2E_ZIP=/tmp/model.zip SECRYST_GOLDEN=/tmp/golden/set.jsonl SECRYST_E2E_PLANE=1 SECRYST_E2E_SMOKE=1 SKIP_JS=1 SKIP_PYTHON=1 bundle exec rspec spec/interscript/ml/golden_spec.rb | |
| env: | |
| MODEL: ${{ inputs.model }} | |
| SECRYST_E2E_CORPUS_BOUND: ${{ inputs.corpus_bound }} | |
| python: | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-python@v6 | |
| with: | |
| python-version: "3.12" | |
| - run: git clone --depth 1 https://github.com/interscript/interscript-py /tmp/py | |
| - name: regenerate byte-exact from the reference | |
| working-directory: /tmp/py | |
| run: | | |
| pip install -e ".[ml]" | |
| curl -sL "https://github.com/interscript/interscript-models/releases/download/${MODEL}/${MODEL}.zip" -o /tmp/model.zip | |
| curl -sL "https://github.com/interscript/interscript-models/releases/download/golden-v2/${MODEL}.jsonl" -o /tmp/set.jsonl | |
| python - << 'PY' | |
| import json | |
| from pathlib import Path | |
| from interscript.ml.plane import PlaneModel | |
| m = PlaneModel.from_zip(Path("/tmp/model.zip").read_bytes()) | |
| rows = [json.loads(l) for l in open("/tmp/set.jsonl") if l.strip()] | |
| total, diffs = 0, 0 | |
| for r in rows: | |
| out = m.translate(r["input"]) | |
| total += len(r["output"]) | |
| if out != r["output"]: | |
| diffs += sum(1 for a, b in zip(out, r["output"]) if a != b) | |
| # K-pass decoding amplifies single numeric flips into clustered | |
| # per-row divergence; the bound is corpus-level by design. | |
| assert diffs / total < 0.02, (diffs, total) | |
| PY | |
| env: | |
| MODEL: ${{ inputs.model }} | |
| ts: | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-node@v6 | |
| with: | |
| node-version: "24" | |
| - run: git clone --depth 1 https://github.com/interscript/interscript-ts /tmp/ts | |
| - name: run the golden set | |
| working-directory: /tmp/ts | |
| run: | | |
| npm ci | |
| mkdir -p /tmp/golden && curl -sL "https://github.com/interscript/interscript-models/releases/download/golden-v2/${MODEL}.jsonl" -o /tmp/golden/set.jsonl | |
| curl -sL "https://github.com/interscript/interscript-models/releases/download/${MODEL}/${MODEL}.zip" -o /tmp/model.zip | |
| INTERSCRIPT_PLANE_ZIP=/tmp/model.zip INTERSCRIPT_PLANE_GOLDEN=/tmp/golden/set.jsonl INTERSCRIPT_PARITY_SMOKE=1 npx vitest run test/ml/plane-parity.test.ts | |
| env: | |
| MODEL: ${{ inputs.model }} |