diff --git a/.dockerignore b/.dockerignore index 81c6d67e..071d3abe 100644 --- a/.dockerignore +++ b/.dockerignore @@ -1,13 +1,81 @@ +# --------------------------------------------------------------------------- +# #498 — build-context hygiene. +# +# Everything listed here is excluded from the context the Docker daemon +# receives. This is the control that makes "no secrets in the image" true by +# construction rather than by review: a file that is not in the context cannot +# be captured in a layer, and cannot survive in a layer cache either. +# +# Only `package*.json`, `package-lock.json`, `nest-cli.json`, `tsconfig*.json`, +# `prisma/`, and `src/` are actually needed to build. Everything else is +# dead weight in the context and a liability if it leaks. +# --------------------------------------------------------------------------- + +# --- Version control & CI ------------------------------------------------- +# Git history is large and can contain credentials in past commits. +.git +.gitignore +.gitattributes +.github + +# --- Dependencies & build output ----------------------------------------- +# Never ship a host node_modules or a stale dist into the image: both are +# rebuilt inside the builder stage against the pinned lockfile. node_modules -coverage dist +build +coverage +.nyc_output +*.tsbuildinfo + +# --- Environment files ---------------------------------------------------- +# A real .env (or any variant) must never reach the daemon, a layer, or a +# cache entry. `.env.example` is the deliberate exception: it is checked-in +# documentation of the variable NAMES the image expects, contains no values, +# and is what an operator copies to .env on first run. .env +.env.* +!.env.example + +# --- Local databases / dev state ----------------------------------------- +# Local SQLite files are developer state, never part of a runtime image. +database.sqlite +dev.db +*.sqlite +*.sqlite-journal +*.sqlite-wal +*.sqlite-shm +prisma/*.db + +# --- Tests, fixtures & scratch output ------------------------------------- +test +coverage-reports +*.log +npm-debug.log* +yarn-debug.log* +yarn-error.log* +run-tests.js +test-ip-security.js +commit-fix.sh +git_automate.py +*.pid +pids + +# --- Editor / OS cruft --------------------------------------------------- +.vscode +.idea .DS_Store -.git -.gitignore +Thumbs.db +*.swp +*.swo + +# --- Docker control files (not needed inside the image) ------------------ Dockerfile +.dockerignore docker-compose.yml -test -.vscode -npm-debug.log +docker-compose.*.yml + +# --- Documentation -------------------------------------------------------- +# Large, and irrelevant at runtime. The contract-relevant content lives in +# docs/CONTAINER_IMAGE.md in the repository, not baked into the image. *.md diff --git a/.github/codeql/config.yml b/.github/codeql/config.yml new file mode 100644 index 00000000..03e9b367 --- /dev/null +++ b/.github/codeql/config.yml @@ -0,0 +1,68 @@ +# --------------------------------------------------------------------------- +# CodeQL configuration for the TruthBounty API (issue #497). +# +# Referenced by both: +# .github/workflows/ci.yml (push/pull_request — owned by a +# parallel PR, not edited here) +# .github/workflows/codeql-schedule.yml (cron + workflow_dispatch) +# +# CodeQL validates this file against a strict schema. Keys are limited to: +# name, disable-default-queries, queries (+ include/exclude per query), +# paths, paths-ignore, packs, and disable-feature-flags. Adding an unknown key +# makes the action fail closed with a configuration error, so resist the urge +# to add "just one more option". +# --------------------------------------------------------------------------- + +name: "TruthBounty API CodeQL config" + +# Keep the default suite. This is the baseline of correctness/security queries +# that ship with the analyser; disabling it would silently drop coverage and +# there is no compensating control in this repository. +disable-default-queries: false + +queries: + # Wider analysis than the default suite. `security-extended` adds the + # precise, lower-signal/lower-severity variants of security queries (e.g. + # the SSRF and injection variants that are usually false positives on an + # indexer that legitimately talks to an RPC provider). `security-and-quality` + # adds maintainability and reliability queries. Both are additive to the + # default suite, not a replacement for it. + # + # No `exclude` filters are declared here on purpose. A query exclusion is a + # blanket suppression by another name, and this repository's suppression + # policy (see docs/STATIC_ANALYSIS.md) requires a documented, expiring + # justification for any suppressed alert. Alert-level suppression belongs in + # the alert itself, where the justification and expiry can be reviewed. + - uses: security-extended + - uses: security-and-quality + +# JavaScript/TypeScript is an interpreted language, so `paths` narrows the set +# of extracted source files rather than acting as a compiler include path. +# `src` is the entire shipped application: NestJS modules, services, +# controllers, entities, and the V2 projectors. +paths: + - src + +# Build output and generated artifacts. These are excluded because analysing +# them produces noise with no security value: `dist` is a transpilation of +# files already covered by `src`, `src/generated` is the machine-written +# Prisma client, and the rest is test/coverage output that is never shipped. +# +# Note what is NOT here: no `**/*.spec.ts` and no source directory. Test files +# and application source are both analysed. Narrowing this list to dodge +# findings would be a coverage regression dressed up as hygiene. +paths-ignore: + - dist + - dist/** + - build + - build/** + - coverage + - coverage/** + - node_modules + - node_modules/** + # Machine-generated Prisma client — never hand-edited, never a place to fix. + - src/generated + - src/generated/** + # Source maps point back into `dist`, which is already ignored, and analyzing + # them double-counts every file. + - "**/*.js.map" diff --git a/.github/workflows/codeql-schedule.yml b/.github/workflows/codeql-schedule.yml new file mode 100644 index 00000000..240417b6 --- /dev/null +++ b/.github/workflows/codeql-schedule.yml @@ -0,0 +1,142 @@ +name: CodeQL Scheduled Analysis + +# --------------------------------------------------------------------------- +# Issue #497 — "Enforce CodeQL and Static Security Analysis". +# +# WHAT THIS CLOSES +# `.github/workflows/ci.yml` already runs CodeQL, but only on: +# push: branches: [main] +# pull_request: branches: [main] +# That means a CodeQL *query* or *scanner* update published by GitHub after +# the last PR is never applied to the default branch. A new +# `js/injection`-style query, or a new taint-mode model for a library this +# repo uses, silently produces zero alerts until the next unrelated PR lands. +# For a project that indexes a financial protocol, an unnoticed new finding on +# the default branch is exactly the failure mode you cannot detect by +# reviewing PRs. +# +# THIS WORKFLOW FILLS THAT GAP. It does not replace or duplicate the ci.yml +# job; both run, and they upload to the same SARIF stream. ci.yml remains +# owned by a parallel PR (#497 follow-on) which is responsible for pinning +# and least-privileging that job — see the "Known gaps" section of +# docs/STATIC_ANALYSIS.md. +# --------------------------------------------------------------------------- + +on: + schedule: + # Weekly, Monday 03:17 UTC. Deliberately off the hour: GitHub's shared + # scheduler queue is heavily oversubscribed on the hour, and a scheduled + # job that lands in that queue can be delayed by 10+ minutes or dropped. + - cron: '17 3 * * 1' + workflow_dispatch: + inputs: + target_ref: + description: 'Branch, tag, or SHA to analyze (defaults to the default branch)' + required: false + type: string + +# Least privilege at the workflow level. `security-events: write` is the only +# permission CodeQL genuinely needs -- it is what allows the SARIF upload -- +# and `contents: read` is what allows checkout. Nothing here grants write +# access to the repository, packages, deployments, or issues. +permissions: + contents: read + security-events: write + +# Never cancel an in-flight analysis. Two overlapping CodeQL runs uploading +# SARIF for the same ref race, and the loser's alerts can be dropped from the +# security view. Queueing is the correct behaviour for a scheduled job. +concurrency: + group: codeql-schedule-${{ github.ref }} + cancel-in-progress: false + +jobs: + analyze: + name: Analyze JavaScript/TypeScript + runs-on: ubuntu-latest + timeout-minutes: 60 + + # Restated per-job, because a workflow-level block can be widened by a + # future edit and the job is what actually runs. + permissions: + contents: read + security-events: write + + steps: + - name: Resolve target ref + id: target + env: + # Passed through the environment rather than interpolated directly + # into the shell, so a hostile ref string cannot become script. + INPUT_TARGET_REF: ${{ inputs.target_ref }} + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + run: | + set -eu + ref="${INPUT_TARGET_REF:-${DEFAULT_BRANCH:-}}" + if [ -z "$ref" ]; then + echo "::error::No default branch available in the event payload; re-run with an explicit target_ref." + exit 1 + fi + echo "Analyzing ref: $ref" + echo "ref=$ref" >> "$GITHUB_OUTPUT" + + - name: Checkout code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ steps.target.outputs.ref }} + + - name: Initialize CodeQL + uses: github/codeql-action/init@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4.38.2 + with: + languages: javascript-typescript + config-file: ./.github/codeql/config.yml + # JavaScript/TypeScript is analysed from source; there is no + # compiler step to trace. `none` avoids invoking `npm ci` + + # `npm run build` inside the analysis job, which is both slow and a + # second, redundant copy of whatever `build-and-test` in ci.yml + # already gates on. It also means an unrelated build failure cannot + # mask or fake a static-analysis result. + build-mode: 'none' + # Analysis starts from a clean slate: a partial previous run's cache + # must never be able to suppress an alert. + clean: 'true' + # + # `fail-on-errors` is deliberately left at its default (true). That + # flag governs *analysis errors* — a malformed config file, an + # extractor crash, an upload failure — and those must fail the run + # loudly. A red scheduled job saying "the scanner broke" is a + # correct, actionable result. Alert counts are a separate question: + # this workflow does not turn new alerts into a build failure, + # because alert triage is a review process, not a compile step. + + - name: Perform CodeQL Analysis + uses: github/codeql-action/analyze@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4.38.2 + with: + category: '/language:javascript-typescript' + # Explicit rather than implicit, so that a future bump of the action + # cannot quietly flip the default and leave a green run with no + # SARIF in the security tab. + upload-sarif: 'true' + # Block until the results are committed to the security view. Without + # this the job can go green before the upload lands, and a + # "successful" scan whose alerts never appeared is worse than a + # visible failure. + wait-for-processing: true + + - name: Summarise run + if: always() + env: + ANALYZED_REF: ${{ steps.target.outputs.ref }} + run: | + { + echo "### CodeQL scheduled analysis" + echo + echo "- **Ref analyzed:** \`${ANALYZED_REF}\`" + echo "- **Suites:** default + \`security-extended\` + \`security-and-quality\`" + echo "- **Config:** \`.github/codeql/config.yml\`" + echo "- **Trigger:** ${{ github.event_name }}" + echo + echo "Alert counts and per-alert detail are in the repository's" + echo "**Security → Code scanning** tab. Triage and suppression policy:" + echo "\`docs/STATIC_ANALYSIS.md\`." + } >> "$GITHUB_STEP_SUMMARY" diff --git a/Dockerfile b/Dockerfile index 05661faf..408d1fca 100644 --- a/Dockerfile +++ b/Dockerfile @@ -22,13 +22,53 @@ FROM node:20-alpine AS runner WORKDIR /app ENV NODE_ENV=production +# --------------------------------------------------------------------------- +# #498 — runtime-stage hardening. Only this stage is modified by this change; +# the builder stage above is owned by a separate PR (#516) so the two do not +# conflict. See docs/CONTAINER_IMAGE.md for the full runtime contract. +# --------------------------------------------------------------------------- + +# Dedicated service account. A named user/group (rather than a bare numeric +# UID) is used deliberately: it gives the account a real passwd entry, a +# nologin shell, and an ownership bit that survives `docker exec` debugging. +# UID/GID 1001 is chosen because it is in the unprivileged range and does not +# collide with the `node` account baked into the base image. +RUN addgroup -S -g 1001 truthbounty \ + && adduser -S -u 1001 -G truthbounty -h /app -s /sbin/nologin truthbounty + +# Only runtime artifacts cross the stage boundary: production dependencies, +# the compiled bundle, and the generated Prisma client. No source, no config, +# no test fixtures, no .env — .dockerignore guarantees none of those are even +# present in the build context. COPY package*.json ./ COPY --from=builder /app/node_modules ./node_modules COPY --from=builder /app/dist ./dist COPY --from=builder /app/src/generated ./src/generated +# Hand the two writable-capable artifact trees to the service account. Applied +# as a targeted RUN rather than `COPY --chown` so the stage stays mergeable +# with the parallel builder-stage change and so the COPY lines remain exactly +# as they were. +RUN chown -R truthbounty:truthbounty /app/dist /app/src/generated + +# Drop root. Everything after this point runs unprivileged, including anything +# a future layer might execute. +USER truthbounty + # Expose port EXPOSE 3000 +# Liveness probe against the route the application actually serves: +# `HealthController` is `@Controller('health')` + `@Public()` with +# `@Get('live')` (src/health/health.controller.ts), and no global prefix is set +# in src/main.ts — so the real path is GET /health/live. It is unauthenticated, +# so no credential is baked into the image. Implemented with the interpreter +# already present in the image rather than a wget/curl dependency. +# Deliberately NOT /health/ready: that endpoint fails closed when Postgres, +# the job queue, or the indexer is down, which would make Docker kill a +# container that is alive but correctly refusing traffic. +HEALTHCHECK --interval=30s --timeout=5s --start-period=40s --retries=3 \ + CMD node -e "require('http').get({host:'127.0.0.1',port:process.env.PORT||3000,path:'/health/live',timeout:4000},r=>{process.exit(r.statusCode===200?0:1)}).on('error',()=>process.exit(1))" + # Start app CMD ["npm", "run", "start:prod"] diff --git a/docs/CONTAINER_IMAGE.md b/docs/CONTAINER_IMAGE.md new file mode 100644 index 00000000..466978b5 --- /dev/null +++ b/docs/CONTAINER_IMAGE.md @@ -0,0 +1,202 @@ +# Production Container Runtime Contract + +> **Scope.** This document describes the *runtime* contract of the TruthBounty +> API production image: who it runs as, how its health is determined, what it +> listens on, and which environment variables it expects **by name only**. +> It never contains a value, a secret, or a production address. +> +> Related: [`STATIC_ANALYSIS.md`](./STATIC_ANALYSIS.md) (CI-side security +> gates), [`DEPLOYMENT.md`](./DEPLOYMENT.md) (operator-facing rollout steps). + +--- + +## 1. Why the runner stage is hardened + +The image is an **indexer and read surface for a financial protocol**. The +deployed Optimism/EVM contracts are the only authority for protocol truth, and +the API is deliberately not authoritative for settlement, rewards, or dispute +outcomes. That design is only trustworthy if the process that serves it is +itself a minimal, non-privileged, well-observed artifact. Three specific +risks are addressed here: + +| Risk | Control | +| ---- | ------- | +| A container escape or RCE lands with root inside the container | Dedicated non-root `USER`, no login shell, UID/GID 1001 | +| A developer's local `.env` (or any env variant) is baked into a published layer | `.dockerignore` excludes `.env` / `.env.*` from the **build context** | +| A silently wedged process keeps "running" while serving nothing | `HEALTHCHECK` against the route the app actually serves | + +--- + +## 2. Runtime contract + +| Property | Value | +| -------- | ----- | +| Base image | `node:20-alpine` | +| Working directory | `/app` | +| Runtime user | `truthbounty` (group `truthbounty`, UID/GID `1001`, shell `/sbin/nologin`) | +| Privileged | No. `USER truthbounty` is declared before `HEALTHCHECK` and `CMD`, so every subsequent instruction — and anything a future layer executes — runs unprivileged. | +| Exposed port | `3000` (override with `PORT`) | +| Healthcheck | `GET http://127.0.0.1:${PORT:-3000}/health/live` | +| Entrypoint | `npm run start:prod` → `node dist/main` | +| Files copied in | `package.json`, `package-lock.json`, `node_modules`, `dist`, `src/generated` | +| Files **not** copied in | source tree, `prisma/`, `contracts/`, tests, docs, `.env*` | + +### 2.1 Why the UID is not a bare number + +The account is created with `adduser`/`addgroup` and a name, not with a bare +`USER 1001`. A numeric `USER` has no passwd entry, no group ownership, and no +shell, which makes `docker exec` debugging and file-ownership reasoning +needlessly hard, and can silently collide with whatever UID the host has +mapped into the container. `1001` is chosen because it sits in the +unprivileged range and does not collide with the `node` account already +present in `node:20-alpine`. + +`node_modules` is intentionally **not** `chown`ed. The application only ever +reads from it; leaving it root-owned with default `0755` permissions means a +compromised process cannot drop a shim into its own resolution path. + +### 2.2 The real health endpoint + +The `HEALTHCHECK` is wired to the route the application genuinely serves, read +off the source rather than assumed: + +- `src/health/health.controller.ts` declares + `@Controller('health')` + `@Public()` with `@Get('live')`. +- `src/main.ts` sets **no** global routing prefix (`app.setGlobalPrefix` does + not appear anywhere in `src/`), so the effective path is `/health/live`. +- `@Public()` is honoured by `src/auth/global-auth.guard.ts` via + `IS_PUBLIC_KEY`, so **no bearer token is required** — which is what makes it + safe to probe from inside the container without embedding a credential in + the image. (`GlobalAuthGuard` would also allow a `GET` through + unauthenticated, but the explicit `@Public()` is the contract we rely on.) + +`/health/live` is a **liveness** probe: `HealthService.getLiveness()` returns +`{ status: 'alive', timestamp, uptime }` from memory only, with no dependency +checks. + +The probe deliberately does **not** use `/health/ready` or `/health` (the +aggregated report). Those run the full dependency set in +`HealthService.runChecks()` — Postgres, the BullMQ `jobs-queue`, IPFS upload, +and the blockchain/indexer state — and return `503` when a *critical* +dependency is down. Wiring `HEALTHCHECK` to a readiness endpoint would let +Docker's restart policy kill a process that is alive and behaving exactly as +designed, turning a database blip into a crash loop and discarding in-flight +work. Use `/health/ready` as a **Kubernetes readinessProbe** and +`/health/live` as a **livenessProbe**; that separation is the point of having +both routes. + +The probe is implemented with `node -e` rather than `wget`/`curl` so it adds +no package to the image and runs on the exact interpreter serving traffic. +`--start-period=40s` covers cold start (module init, TypeORM connection). + +### 2.3 The healthcheck must pass with no environment file + +`.dockerignore` removes every `.env*` from the build context, so the container +starts with an empty environment unless the operator supplies one. The +`HEALTHCHECK` reads only `PORT` and defaults it, so the probe is functional in +that state. Any *other* missing variable is the application's business, not +the probe's — see §3. + +--- + +## 3. Expected environment variables (names only) + +Copy `.env.example` to `.env` locally, or supply the variables through your +orchestrator's secret store. **Values are never committed, never baked into +the image, and never printed here.** The table lists names and purpose only. + +| Variable | Purpose | Notes | +| -------- | ------- | ----- | +| `NODE_ENV` | Runtime mode. Set to `production` in the image. | Already baked in; do not override to `development`. | +| `PORT` | HTTP listen port. Image exposes `3000`. | Healthcheck follows this value. | +| `DATABASE_URL` | PostgreSQL connection string. **Required** in production. | Without it the datasource falls back to a local SQLite file, which a non-root user cannot create. Fail fast is the intended outcome. | +| `DATABASE_SYNCHRONIZE` | Schema auto-sync. | Leave `false`; migrations are the supported path. | +| `DATABASE_LOGGING` | SQL logging toggle. | `false` in production. | +| `DB_HOST`, `DB_PORT`, `DB_USERNAME`, `DB_PASSWORD`, `DB_NAME` | Individual PG settings, as an alternative to `DATABASE_URL`. | Prefer `DATABASE_URL`. | +| `DB_SSL`, `DATABASE_SSL`, `DATABASE_SSL_REJECT_UNAUTHORIZED` | TLS to PostgreSQL. | Enable for managed PG. | +| `DB_POOL_MAX`, `DB_POOL_IDLE_TIMEOUT`, `DB_POOL_ACQUIRE_TIMEOUT`, `DB_POOL_RETRIES`, `DB_POOL_RETRY_DELAY` | Connection-pool tuning. | Defaults documented in `src/config/data-source.ts`. | +| `JWT_SECRET` | Signs wallet-auth JWTs. **Required.** | Never ship the `.env.example` placeholder. | +| `JWT_EXPIRATION` | Token lifetime. | e.g. `7d`. | +| `TRUSTED_PROXIES` | Comma-separated proxy IPs. | Empty ⇒ `trust proxy` disabled. | +| `REDIS_ENABLED`, `REDIS_HOST`, `REDIS_PORT`, `REDIS_PASSWORD`, `REDIS_DB`, `REDIS_TLS` | Redis / BullMQ connection. | Required for the notification outbox relay. | +| `BLOCKCHAIN_RPC_URL`, `OPTIMISM_RPC_URL`, `CHAIN_ID` | Optimism RPC endpoint and chain id. | Carry provider credentials; do not bake them in. | +| `REWARD_CONTRACT_ADDRESS` | Deployed contract to index. | Real address from the deployment record, never a dummy. | +| `START_BLOCK`, `REQUIRED_CONFIRMATIONS`, `CONFIRMATIONS_REQUIRED`, `BLOCK_RANGE_PER_BATCH`, `MAX_RETRY_ATTEMPTS`, `POLLING_INTERVAL_MS` | Indexer pacing and finality lag. | Finality lag is what keeps projections off unfinalized chain state. | +| `INDEXED_CONTRACTS` | JSON array of contract/event index configuration. | Empty array ⇒ indexer idle. | +| `SMTP_HOST`, `SMTP_PORT`, `SMTP_USER`, `SMTP_PASS`, `SMTP_FROM` | Notification email delivery. | Secrets via the orchestrator, not the image. | +| `NOTIFICATION_QUEUE_DELAY`, `NOTIFICATION_MAX_RETRIES`, `NOTIFICATION_RETRY_DELAY` | Outbox relay pacing. | | +| `CACHE_CLAIMS_TTL`, `CACHE_VERSION` | Redis cache tuning. | Bump `CACHE_VERSION` when cache shape changes. | +| `RATE_LIMIT_*` | Per-route throttling. | | +| `REALTIME_*` | SSE/outbox publisher tuning. | | +| `SYBIL_MIN_CLAIMS_FOR_ACCORACY_SCORE` | Sybil resistance floor. | | +| `BLOCKCHAIN_MAX_BLOCKS`, `BLOCKCHAIN_MAX_EVENTS`, `BLOCKCHAIN_MAX_REORG_HISTORY` | In-memory state caps (OOM guard). | | + +`src/config/data-source.ts` and `src/config/blockchain.config.ts` are the +authoritative readers; this table is a convenience summary and can drift. +When in doubt, follow the code. + +--- + +## 4. What is deliberately **not** in the image + +| Excluded | Why | +| -------- | --- | +| Source (`src/**` except `src/generated`) | Only `dist` is executed. Source would be dead weight at best and a map of internals at worst. | +| `prisma/` schema + migrations | Migrations are run by the deploy pipeline against the live database, not from inside the app image. | +| `.env`, `.env.*` | Excluded from the **build context** by `.dockerignore`, so it cannot enter a layer or a layer cache. `.env.example` is kept deliberately: names only, no values. | +| `node_modules` from the host | Always rebuilt in the builder stage against the pinned lockfile. | +| `dist` from the host | Same — a host build may not match the image's platform or libc. | +| `database.sqlite`, `dev.db` | Developer state. Not runtime input. | +| Tests, coverage, docs | Not runtime input. | + +--- + +## 5. Known gaps and follow-ups + +These are **not** fixed by this change and are recorded so they are not lost. + +1. **Builder stage still uses `npm install`, not `npm ci`.** Owned by the + parallel PR (#516). It was deliberately left untouched here so the two + changes merge cleanly. Until it lands, the builder can drift from + `package-lock.json` and produce a non-reproducible `node_modules`. +2. **Builder stage `prunes` in place.** `npm prune --production` mutates the + builder's `node_modules`, which is then copied wholesale. `--omit=dev` + during a dedicated install step would be cleaner, but that is also a + builder-stage change. +3. **No digest pinning of the base image.** `node:20-alpine` is a floating + tag. Pin to a digest (e.g. `node:20-alpine@sha256:...`) for reproducible + and auditable builds; this needs a real digest and a Dependabot-compatible + update path, so it is tracked rather than guessed. +4. **No read-only root filesystem declared.** `docker run --read-only` plus a + writable tmpfs for `/tmp` is achievable because the app writes nothing to + `/app`, but that is an orchestrator-level decision, not a Dockerfile one. +5. **`ContractArtifactsLoader` reads `config/contracts/release-artifacts.json` + from the working directory**, a path that is neither present in the + repository nor copied into the image. Today this is harmless: neither + `ContractArtifactsLoader` nor `HealthDiagnosticsController` is registered as + a provider or controller anywhere in `src/app.module.ts`, so it is dead + code. **If either is ever wired up, the image will fail to start** until + the release artifacts are mounted at that path. +6. **Startup requires a reachable PostgreSQL.** With no `DATABASE_URL`, the + datasource falls back to SQLite at a relative path. Under a non-root user + in a read-only `/app` that fails — intentionally, since silently serving + from local state would contradict the non-authoritative-database rule in + `docs/DEPLOYMENT.md`. + +--- + +## 6. Verifying an image + +```bash +# Runtime user and healthcheck are present +docker inspect --format '{{ .Config.User }} {{ json .Config.Healthcheck }}' truthbounty-api: + +# Nothing that looks like an env file made it into any layer +docker history --no-trunc truthbounty-api: | grep -E '\.env' || echo "clean" + +# The probe actually passes against a running container +docker exec truthbounty-api: node -e "require('http').get('http://127.0.0.1:3000/health/live',r=>console.log(r.statusCode))" +``` + +None of the above was executed as part of the change that introduced this +document; the maintainer runs all image and CI verification. diff --git a/docs/PROJECTION_REBUILD.md b/docs/PROJECTION_REBUILD.md new file mode 100644 index 00000000..5147ca42 --- /dev/null +++ b/docs/PROJECTION_REBUILD.md @@ -0,0 +1,351 @@ +# Projection Rebuild Pipeline (V2-BE-019) + +> **Scope.** How to deterministically rebuild every V2 read model from the +> persisted canonical event log, starting at a configured deployment block; +> how to resume an interrupted rebuild; how to read the reconciliation report; +> and the exact procedure for cutting a rebuilt read model over to live. +> +> Related: [`indexer-runbook.md`](./indexer-runbook.md) (lag/finality +> alerting), [`load-budgets.md`](./load-budgets.md) (read-path budgets), +> [`V2_REWARD_ALLOCATION_AUDIT.md`](./V2_REWARD_ALLOCATION_AUDIT.md) (what the +> rebuilt reward projection contains). + +--- + +## 0. What already existed, and what this adds + +The repository already had three partial capabilities. None of them was a +rebuild, and this section is explicit about the difference so the diff reads +honestly. + +| Prior art | Path | What it did | Why it was not a rebuild | +| --------- | ---- | ----------- | ------------------------ | +| Per-claim reprojection | `ClaimProjectorService.reproject` — `src/claims/claim-projector.service.ts` | Replays one claim's stored lifecycle events into its read model | Per-claim, not full-schema; driven by the `ClaimLifecycleEvent` table, not the canonical event log; no checkpoint, no report, no cross-projection coordination | +| Cursor rewind | `EventIndexerService.backfillFromBlock` — `src/indexer/event-indexer.service.ts` | Moves `lastProcessedBlockNumber` back so the RPC poller re-fetches a range | Rewinds an *ingestion* cursor, not a projection cursor. Re-fetches from RPC, applies into a **non-empty** live schema, and is neither resumable nor reportable | +| Outbox idempotency | `OutboxService` — `src/outbox/outbox.service.ts` | Deterministic idempotency keys, at-least-once relay, dead-lettering | Establishes the *pattern* this pipeline follows (deterministic identity + explicit terminal state); it is a relay, not a projection | + +What was genuinely missing: a **full**, **deterministic**, **resumable**, +**report-producing** re-derivation of every V2 read model, with a guard that +keeps a partial rebuild from ever being observable as authoritative. + +--- + +## 1. Scope boundary — what a rebuild consumes + +The rebuild replays **`v2_canonical_events`**, the normalized/decoded event +log. It does **not** re-scan the chain. + +This matches the position already recorded in +[`indexer-runbook.md`](./indexer-runbook.md): *"Projections are rebuildable +from raw, persisted events."* The canonical event log is produced by the +ingestion path (`CanonicalEventsService.ingest`); re-fetching it from RPC is +the indexer's job, and building a second RPC re-scan path here would duplicate +ingestion logic and create a second source of indexing truth. + +Consequence: **a rebuild into a genuinely empty schema needs the canonical log +populated first.** Procedure: + +1. Run the indexer against the target database until + `GET /health/indexer` reports `projectionLag: 0` and a non-null + `finalizedBlock` (see `indexer-runbook.md`). +2. Confirm the log: `SELECT COUNT(*), MIN("blockNumber"), MAX("blockNumber") + FROM v2_canonical_events WHERE "chainId" = ;` +3. Only then run the rebuild. + +### Projections in scope + +| Projection | Tables | Events | +| ---------- | ------ | ------ | +| `v2-evidence` | `v2_project_evidence`, `v2_project_evidence_version` | `EvidenceRegistered`, `EvidenceReplaced`, `EvidenceRemoved` | +| `v2-verification` | `v2_project_verification_round`, `v2_project_participant_position` | `VerificationRoundOpened`, `PositionCommitted` | +| `v2-disputes` | `v2_project_dispute` | `DisputeRaised`, `DisputeResolved`, `DisputeExpired` | +| `v2-rewards` | `v2_project_reward_allocation`, `v2_project_reward_pool` | `RewardPoolSettled`, `RewardAllocated`, `RewardClaimed` | + +The order is fixed in `src/v2/rebuild/projection-registry.ts` and is the same +on every run. The registry is composed in one file rather than contributed by +each projector module, precisely so the effective order is reviewable rather +than an accident of Nest module wiring. + +### Explicitly **out** of scope + +- `v2_canonical_events` itself, `v2_event_checkpoints`, `v2_event_quarantine` + — the input, not a projection. +- `v2_indexing_anomalies` — the projector's audit log; the rebuild does not + clear it, so anomaly history survives a rebuild. +- Legacy `src/rewards` (`reward_claims`, `reward_distributions`), the + `IndexedEvent` indexer tables, and the realtime `projection_events` outbox — + different pipelines, different sources. See + [`V2_REWARD_ALLOCATION_AUDIT.md`](./V2_REWARD_ALLOCATION_AUDIT.md) §"Legacy + paths". +- `ReorgSafeCursorService` (`src/indexer/reorg-safe-cursor.service.ts`) — see + §7, item 2. + +--- + +## 2. Running a rebuild + +The entry point is a script, **not** an HTTP endpoint. A rebuild truncates and +re-derives every read model; exposing it over HTTP would put a destructive, +long-running, cluster-wide operation behind a request any authenticated caller +could repeat. + +```bash +# Shadow rebuild — the normal case +export REBUILD_SCHEMA=truthbounty_shadow +export DATABASE_URL=postgresql://.../truthbounty_shadow +npx ts-node src/v2/rebuild/projection-rebuild.cli.ts \ + --chain-id 10 \ + --deployment-block 126000000 \ + --reset +``` + +| Flag | Meaning | +| ---- | ------- | +| `--chain-id` | Required. Numeric chain id (Optimism is `10`). | +| `--deployment-block` | Required. First block in scope. A **string**; never parse it into a JS number. | +| `--reset` | Clear every registered projection's tables and cursor first. Required for a from-scratch rebuild. | +| `--event-batch-size` | Events per projector per drain iteration. Default `500`. Affects speed only, never the report. | +| `--max-batches N` | Stop after N iterations and leave a resumable checkpoint. Default: drain to completion. | +| `--resume-from-block B` | Resume from a previous run's `fromBlock`. | +| `--resume-digest D` | The previous run's `inputDigest`. **Required** with `--resume-from-block`. | +| `--allow-in-place` | Acknowledge rebuilding the **live** schema. Discouraged; see §4. | + +stdout carries the deterministic report; logs go to stderr. So the report +pipes cleanly: + +```bash +npx ts-node src/v2/rebuild/projection-rebuild.cli.ts --chain-id 10 \ + --deployment-block 126000000 --reset > rebuild-a.json +``` + +--- + +## 3. The reconciliation report + +```jsonc +{ + "anomalies": 0, // events a projector refused and logged + "batchesProcessed": 2, // drain iterations (NOT part of determinism) + "chainId": 10, + "complete": true, // the drain ran to exhaustion + "deploymentBlock": "126000000", + "eventsApplied": 184213, // projector outcomes that wrote a row + "eventsConsumed": 190442, // canonical events folded into inputDigest + "eventsSkipped": 6229, // consumed but deliberately not written + "fromBlock": "126000001", // first block NOT yet folded (the resume point) + "inputDigest": "9f2c…", // rolling SHA-256 over consumed event identities + "logIndex": 41, + "perProjection": { + "v2-disputes": { + "anomalies": 0, "eventsApplied": 1204, "eventsConsumed": 1204, + "eventsSkipped": 0, "rowsInTable": 1180 + } + // … + }, + "safeToCutover": true, // the cutover precondition + "toBlock": "131500207" +} +``` + +Every count is concrete and checkable against SQL. For example: + +```sql +-- perProjection[...].rowsInTable +SELECT COUNT(*) FROM v2_project_dispute; +-- eventsApplied is in v2_projection_rebuild_runs.eventsApplied +SELECT "batchesProcessed", "eventsConsumed", "eventsApplied", "anomalies", + "inputDigest", "safeToCutover" +FROM v2_projection_rebuild_runs ORDER BY "startedAt" DESC LIMIT 1; +``` + +### 3.1 What makes it deterministic + +`RebuildCheckpoint` (`src/v2/rebuild/rebuild-checkpoint.ts`) contains **no +timestamp, no duration, no run id, and no host detail**. It is a pure function +of `(chainId, deploymentBlock, the ordered canonical events in range)`. All +non-deterministic material (status transitions, timings, errors) lives in the +`v2_projection_rebuild_runs` row instead. + +`inputDigest` is a **left fold** of SHA-256 over +`chainId:txHash:logIndex:eventName` for every event consumed, in +`(blockNumber, logIndex)` order — the protocol's own order. Properties: + +- **Order-dependent.** A reordered log yields a different digest, so a + mis-ordered replay is detectable rather than invisible. +- **Batch-size independent.** One batch of 500 and 500 batches of 1 fold to the + same accumulator, because each drain iteration runs every projection to + exhaustion before the next begins, so the folded ranges are disjoint and + ordered. +- **Resume-safe.** The accumulator is the only state carried across a + checkpoint boundary, and it is persisted in the run row. Resuming folds the + remainder onto the stored value and lands on the same digest as an + uninterrupted run. + +Serialisation sorts keys at every level, so two structurally equal reports +are byte-identical and can be diffed directly: + +```bash +diff <(jq -S . rebuild-a.json) <(jq -S . rebuild-b.json) # must be empty +``` + +### 3.2 `unclaimedEvents` — the registry drift alarm + +`unclaimedEvents` counts canonical events in the drained range that **no** +registered projection claims. It is normally `0`. + +A non-zero value means the registry's `eventNames` are stale: a new projector +exists in the code but not in `buildDefaultRegistry`, so a cutover would swap +in a read model that is silently missing a projection. It therefore blocks +`safeToCutover` on its own. + +--- + +## 4. Cutover safety + +**The pipeline never performs a cutover.** It has no method that swaps schemas, +renames tables, or repoints a connection. Two independent guards enforce that a +partial rebuild is never observable as authoritative: + +1. **Refusal to run in place by default.** Without `REBUILD_SCHEMA` set, or + with `--allow-in-place` absent, `rebuild()` throws before touching a single + row. With `--allow-in-place`, it logs a warning naming the exact risk. +2. **`safeToCutover`, not `cutover`.** The report *states* whether a rebuild is + eligible; acting on it is a separate, deliberate, documented act. The CLI + exits `2` when `safeToCutover` is false, so a pipeline that chains the + command fails rather than continuing. + +`safeToCutover === true` requires **all** of: + +- `complete` — the drain reached exhaustion, not a `maxBatches` stop; +- `anomalies === 0` — no projector refused an event; +- `unclaimedEvents === 0` — the registry accounts for the whole log. + +### 4.1 The exact swap procedure (PostgreSQL) + +The rebuild targets a shadow schema. The swap is a rename inside a single +transaction, so readers see either the whole old read model or the whole new +one, never a mixture. + +```sql +BEGIN; + +-- 0. Preconditions. Abort if any is false. +-- - the latest v2_projection_rebuild_runs row for this chain has +-- safeToCutover = true and status = 'completed' +-- - no application instance is still writing to the shadow schema +-- - a backup of the live read model exists (see docs/DISASTER_RECOVERY.md) + +-- 1. Quiesce writers. The read models are written only by the projectors, so +-- stop the indexer/projector workers. Readers are unaffected and keep +-- serving from the live schema throughout. + +-- 2. Capture the pre-swap row counts for the after-the-fact comparison. +CREATE TEMP TABLE rebuild_pre_counts AS +SELECT 'v2_project_dispute' AS t, COUNT(*) AS n FROM v2_project_dispute +UNION ALL SELECT 'v2_project_participant_position', COUNT(*) FROM v2_project_participant_position +UNION ALL SELECT 'v2_project_reward_allocation', COUNT(*) FROM v2_project_reward_allocation +UNION ALL SELECT 'v2_project_reward_pool', COUNT(*) FROM v2_project_reward_pool; + +-- 3. The swap. Each V2 read-model table moves in a single statement; the +-- transaction is what makes it atomic. There is no window in which a +-- reader can observe a half-rebuilt projection. +ALTER TABLE v2_project_dispute RENAME TO v2_project_dispute_old; +ALTER TABLE truthbounty_shadow.v2_project_dispute RENAME TO v2_project_dispute; +-- … repeat for every table in scope, plus its indexes and constraints. + +-- 4. Compare counts and digests against the report before committing. +-- A difference means STOP: ROLLBACK and investigate. Do not "correct" the +-- counts — a divergence between the emitted chain data and the projection +-- is a fact to understand, not a number to reconcile away. + +-- 5. Only after the comparison passes: +COMMIT; + +-- 6. After commit: drop the _old tables, restart the projector workers, +-- re-point the shadow connection string away from truthbounty_shadow. +``` + +**On a failed or rolled-back swap:** the live read model is unchanged, because +the whole swap was one transaction. Drop the shadow schema, fix, re-run the +rebuild, compare again. + +**Never** "repair" a divergence by editing rows to make counts line up. The +backend may project what the chain emitted; it may not synthesise what the +chain did not. + +--- + +## 5. Resuming + +```bash +# Bounded run, keeps a resumable checkpoint +npx ts-node src/v2/rebuild/projection-rebuild.cli.ts \ + --chain-id 10 --deployment-block 126000000 --reset --max-batches 200 \ + > partial.json + +# Resume from the checkpoint's own fields +npx ts-node src/v2/rebuild/projection-rebuild.cli.ts \ + --chain-id 10 --deployment-block 126000000 \ + --resume-from-block "$(jq -r .fromBlock partial.json)" \ + --resume-digest "$(jq -r .inputDigest partial.json)" \ + > final.json +``` + +`--resume-digest` is mandatory alongside `--resume-from-block`: the digest is +the fold accumulator, and resuming without it would silently produce a +different final report. + +A resumed run **does not** touch the projector cursors — they are exactly where +the interrupted run left them, which is the resume point. + +--- + +## 6. Monitoring and alerting + +| Signal | Source | Alert when | +| ------ | ------ | ---------- | +| Rebuild failed | `v2_projection_rebuild_runs.status = 'failed'` | any | +| Anomalies during rebuild | `v2_projection_rebuild_runs.anomalies` | `> 0` | +| Registry drift | a rebuild report with `unclaimedEvents > 0` | any | +| Digest drift | `inputDigest` differs between two runs of the same range | any | +| Row-count drift | `perProjection[].rowsInTable` differs between two runs of the same range | any | + +The two drift checks are the ones worth wiring first. They turn "the projection +looks wrong" into "the projection changed for a reason the report can name". + +--- + +## 7. Known gaps and residual risks + +1. **The projectors are chain-agnostic.** `CanonicalEventQueryService.findAfter` + filters by event name, not by `chainId`, and so do all four projectors. The + rebuild's own event slicing *is* chain-scoped. On a single-chain deployment + (Optimism, `chainId` 10) this is invisible; on a multi-chain deployment the + projectors would mix chains. Pre-existing, not introduced here, and it + affects the live incremental path identically. +2. **`ReorgSafeCursorService` is dead code that would fail at runtime.** It + writes to `v2_indexer_cursors` and `v2_projections`, and **neither table is + created by any migration** in `src/migrations/`. It is not registered as a + provider anywhere, so nothing calls it. It also duplicates the role the + canonical event log and `ProjectorCursor` already play. **Recommendation: + delete it** rather than migrate it; the rebuild pipeline does not use it. +3. **`EventCheckpoint.lastFinalizedBlock` is never written.** + `CanonicalEventsService.ingest` only advances `lastSafeBlock`, so + `DisputesQueryService.calculateDataState` can never return + `DataState.FINALIZED` and every row is reported as `SAFE` or `OBSERVED`. The + rebuild does not touch ingestion, so a rebuilt read model inherits this. It + is a read-model labelling bug, not a data-loss bug. +4. **Anomalies are not cleared by `--reset`.** That is intentional (the anomaly + log is audit history), but it means `recentRuns` anomalies and + `v2_indexing_anomalies` row counts will disagree after a reset rebuild. +5. **No integration test exercises a multi-table `ALTER TABLE … SET SCHEMA` + swap.** The swap procedure in §4.1 is written out and reviewed, but it has + not been executed. Treat the first cutover as a rehearsed change: run the + whole sequence against a restored production snapshot first. +6. **A rebuild is O(canonical events) in wall time** and holds no back-pressure + against the live indexer, because the two write to different schemas. On a + large log, plan the window; `maxBatches` exists so it can be spread across + several invocations. +7. **The scheduled CodeQL lane and this pipeline are unrelated** but share a + theme: both are about a check that must actually run. See + [`STATIC_ANALYSIS.md`](./STATIC_ANALYSIS.md). diff --git a/docs/STATIC_ANALYSIS.md b/docs/STATIC_ANALYSIS.md new file mode 100644 index 00000000..e74f8aa9 --- /dev/null +++ b/docs/STATIC_ANALYSIS.md @@ -0,0 +1,252 @@ +# Static Analysis: CodeQL Coverage, Schedule, and Suppression Policy + +> **Scope.** What CodeQL analyses in this repository today, what the scheduled +> run adds on top of CI, how to read an alert, and the rules for suppressing +> one. Related: [`CONTAINER_IMAGE.md`](./CONTAINER_IMAGE.md) (image hardening), +> [`DEPENDENCY_SECURITY.md`](./DEPENDENCY_SECURITY.md) (dependency-layer +> scanning, owned by a parallel change). + +--- + +## 1. What is enforced today + +| Layer | Where | When | What it does | +| ----- | ----- | ---- | ------------ | +| CodeQL (PR + push) | `.github/workflows/ci.yml` → `security-scans` | `push` to `main`, `pull_request` to `main` | `github/codeql-action/init@v4` + `analyze@v4`, `languages: javascript, typescript` | +| CodeQL (schedule) | `.github/workflows/codeql-schedule.yml` | `schedule` (weekly, Mon 03:17 UTC) + `workflow_dispatch` | Same two actions, same config file, same SARIF destination | +| Dependency audit | `ci.yml` → `security-scans` | same | `npm audit --audit-level=high` | +| Secret scanning | `ci.yml` → `security-scans` | same | TruffleHog against the PR diff | +| Container scan | `ci.yml` → `container-scan` | same | Trivy, `CRITICAL,HIGH`, `os,library`, `ignore-unfixed` | +| Lint / types | `ci.yml` → `build-and-test` | same | `eslint` (non-mutating), `nest build`, `jest --coverage`, Prisma migration reset/deploy | + +**CodeQL is genuinely enforced already.** This change does not introduce it, +and it would be dishonest to claim otherwise. What it introduces is the +scheduled lane and the shared configuration file. + +### 1.1 The configuration file + +`.github/codeql/config.yml` is the single place that defines *what* is +analysed, for **both** workflows: + +```yaml +disable-default-queries: false # baseline suite is kept +queries: + - uses: security-extended + - uses: security-and-quality +paths: + - src # the entire shipped application +paths-ignore: + - dist, build, coverage, node_modules + - src/generated # machine-generated Prisma client + - "**/*.js.map" +``` + +The `security-extended` and `security-and-quality` suites are **additive** to +the default suite. `disable-default-queries: false` is spelled out explicitly +so that a future edit cannot quietly drop the baseline. + +`paths-ignore` is limited to build output and generated artifacts. It +deliberately does **not** exclude `**/*.spec.ts` or any source directory. A +`paths-ignore` entry is a blind spot that a future contributor cannot see the +consequence of, and widening it to make an alert count look better is a +coverage regression disguised as hygiene. + +--- + +## 2. What the schedule adds (the actual gap being closed) + +`ci.yml` triggers on `push`/`pull_request` **into `main`**. Between two such +events, CodeQL analyses nothing. During that window, GitHub can ship: + +- a **new query** (a newly authored `js/...` security or quality check), +- an **updated query** (a refined dataflow that newly matches existing code), +- a **new taint-mode model** for a library this repo uses, +- a **new CWE mapping** that reclassifies an already-reported alert. + +None of these are visible in a code review, and none of them re-trigger +`ci.yml`. On a repository whose purpose is to index a financial protocol, an +undetected injection or SSRF introduced on `main` by an unrelated commit is +precisely the class of failure that code review does not catch. + +`codeql-schedule.yml` therefore re-analyses the default branch weekly, plus on +demand: + +- `schedule: '17 3 * * 1'` — Monday 03:17 UTC. Off-the-hour on purpose; + GitHub's shared scheduler is oversubscribed at `:00` and on-hour jobs are + routinely delayed or dropped. +- `workflow_dispatch` with an optional `target_ref`, so a specific branch, tag, + or SHA can be scanned during an incident. + +Both lanes upload to the **same SARIF stream** (the repository's Code scanning +alerts). Scheduled results therefore *add* findings to the security tab rather +than creating a parallel set that nobody reads. + +### 2.1 Operational properties of the scheduled lane + +| Property | Value | Reason | +| -------- | ----- | ------ | +| `permissions` | `contents: read`, `security-events: write` | The minimum CodeQL needs. No write access to the repo, packages, deployments, or issues. Restated at job level as well. | +| Action pinning | Full 40-char commit SHAs | See §6. | +| `build-mode` | `none` | JavaScript/TypeScript is analysed from source; there is no compiler step to trace. Avoids a redundant `npm ci` + `npm run build`, and prevents an unrelated build failure from faking or masking a scan result. | +| `clean` | `true` | A stale partial cache must never be able to suppress an alert. | +| `upload-sarif` | `true` (explicit on `analyze`) | A green run with no SARIF is indistinguishable from a silently broken scanner. | +| `wait-for-processing` | `true` | Block until results land in the security view; otherwise a "successful" scan whose alerts never appeared is worse than a visible failure. | +| `fail-on-errors` | default (`true`) | Analysis *errors* — bad config, extractor crash, upload failure — must fail loudly. | +| `concurrency.cancel-in-progress` | `false` | Two overlapping runs uploading SARIF for the same ref race; the loser's alerts can vanish. Queue instead. | +| `timeout-minutes` | `60` | A hung analysis is a failure, not a job that runs until the 6-hour cap. | +| Injection safety | Dispatch input passed via `env`, not interpolated into the shell | A hostile `target_ref` cannot become shell script. | + +The scheduled lane does **not** fail the build on new alerts. Alert triage is +a review process, and a scheduled gate that goes red on the first new +quality-query hit trains people to re-run instead of to triage. Enforcement of +*analysis errors* is a different thing and is kept strict. + +--- + +## 3. Reading an alert + +**Security → Code scanning → an alert.** Useful fields: + +| Field | How to use it | +| ----- | ------------- | +| Rule ID | The query identity. `js/...` is a security query; `js/quality/...` or `js/maintainability/...` is from `security-and-quality` and is advisory, not a vulnerability. | +| Severity | `critical`/`high` → triage within the current sprint. `medium`/`low` and quality alerts → normal backlog. Severity is CodeQL's confidence-weighted estimate, not a CVSS score and not a statement of exploitability here. | +| Tags | The CWE mapping and the suite the query came from. | +| "Introduced by" | The commit/PR that created the path. If it says a merge commit from a long-ago PR, it is pre-existing, not a regression. | +| Location | File plus the exact tainted source/sink pair CodeQL traced. | +| "Show more" | The full dataflow. Read it end to end before triaging — CodeQL traces are precise about *where* and frequently imprecise about *whether*. | + +### 3.1 Triage order + +1. **Reproduce the flow.** Follow the reported source → intermediate → sink. + Ask: does untrusted input actually reach the sink in a reachable path? +2. **Classify the exposure.** For an indexer, the realistic untrusted inputs + are HTTP request bodies/params (including spoofable headers) and on-chain + event `payload` JSON. Anything sourced from `process.env` or from a + hard-coded, approved contract address is not attacker-controlled. +3. **Decide.** Fix, document-and-accept, or suppress — never "ignore". +4. **For a real finding**, fix it in the same PR that triaged it. Do not let + a triaged `critical` sit in the backlog. + +### 3.2 Findings that will legitimately recur + +Expect recurring alerts in these areas, and triage them as a class rather than +one at a time: + +- `js/sql-injection` on `createQueryBuilder` string fragments. TypeORM + parameterises `:named` binds; raw `.where(\`... ${x}\`)` fragments are real + findings and should be fixed. +- Path/URL handling in the IPFS, notification, and blockchain RPC clients. +- `js/insecure-randomness` for anything that is *not* a security token + (correlation ids, cursor nonces). A finding on a JWT secret or a + wallet-challenge nonce *is* real. +- `security-and-quality` maintainability queries (unused variables, empty + blocks). Advisory. Note the repository already has known dead code in this + category — see §5. + +--- + +## 4. Suppression policy + +> **A suppression is a documented, time-boxed assertion that a finding is not +> a vulnerability. It is not a way to make a number go down.** + +Rules: + +1. **No blanket suppressions.** Never suppress a whole rule, a whole file, or + a whole directory. Never add an `exclude:` to `.github/codeql/config.yml` + to silence a query. The config file's coverage is reviewable; a query + exclusion there is permanent and invisible to whoever reads an alert. +2. **Suppress at the alert, in the security tab.** Use the dismiss reason + `Used in tests`, `False positive`, or `Won't fix`. The reason and author are + recorded in the alert's audit trail, where they are reviewable. +3. **Every suppression carries an expiry.** Record the expiry date in the + dismissal comment. A suppression without an expiry date is rejected at + review. +4. **Justification must name the reason, not the conclusion.** "`js/x` is a + false positive here" is not a justification. "`process.env.DATABASE_URL` is + set from the orchestrator's secret store at pod start and is never + derived from request data; there is no path from a request parameter to + this sink" is. +5. **`Wont't fix` is for ≤ 1 sprint.** Anything longer is `False positive` with + an honest note, or it gets fixed. +6. **A high/critical alert may not be suppressed without a second reviewer's + sign-off recorded in the dismissal comment.** +7. **Re-open on material change.** If the code at the alert's location + changes, the alert reopens automatically. Do not suppress the replacement + on the basis of the old suppression. +8. **Suppressions are reviewed on a cadence.** Whoever owns + `.github/codeql/config.yml` re-reviews open suppressions monthly. Anything + past its expiry is either fixed or re-justified in writing. + +--- + +## 5. Known gaps + +Stated plainly, because a security document that claims completeness is worse +than no document. + +1. **`ci.yml` is not yet pinned or least-privileged.** Its `security-scans` + job already declares `contents: read` + `security-events: write` (correct), + but its actions are referenced by floating major tags + (`github/codeql-action/init@v4`, `trufflesecurity/trufflehog@main`, + `aquasecurity/trivy-action@master`, `actions/checkout@v7`). `trufflehog@main` + and `trivy-action@master` are mutable *branches*, not tags — they can be + repointed at arbitrary content at any time by anyone with push access to + those repositories. **A parallel PR owns `ci.yml` and is responsible for + pinning all of them to full SHAs and setting an explicit + `permissions:` block on every job.** This change deliberately does not + touch `ci.yml`, so the two PRs merge without conflict. +2. **`trufflesecurity/trufflehog@main` and `aquasecurity/trivy-action@master`** + are the highest-value pinning targets in the whole workflow, and are out + of scope here for the same reason. +3. **No SARIF is archived as a build artifact.** If a run's alerts are + dismissed wholesale, the run history in the security tab is the only + record. Uploading the SARIF as a workflow artifact would give a durable, + diffable history. Left for the `ci.yml` owner. +4. **CodeQL covers `javascript, typescript` only.** No compiled-language + analysis, and no `actions` language analysis of the workflow files + themselves — so workflow-level issues (e.g. a script-injection in a `run:` + block, which is a real class of GitHub Actions vulnerability) are not + covered by CodeQL here. Enabling `languages: actions` is a one-line change + but is owned by the `ci.yml` PR. +5. **Nothing enforces the "backend never decides protocol outcomes" invariant + statically.** The project's core rule — deployed Optimism/EVM contracts and + finalized canonical events are the only authority — is enforced by review + and by unit tests today, not by a query. A custom CodeQL query for it would + be a genuinely valuable follow-up. +6. **There is no SAST beyond CodeQL in the pull-request path.** `eslint` with + `eslint-plugin-security`/`eslint-plugin-no-unsanitized` is not configured. + That is a real gap and is a separate change. +7. **The scheduled lane has not been observed running.** The workflow file is + new; its first scheduled execution is up to seven days after merge. Until a + run completes, treat the schedule as unproven. + +--- + +## 6. Action pinning + +Both actions in `codeql-schedule.yml` are pinned to full 40-character commit +SHAs, with the human-readable version in a trailing comment: + +| Action | SHA | Version | +| ------ | --- | ------- | +| `actions/checkout` | `3d3c42e5aac5ba805825da76410c181273ba90b1` | `v7.0.1` | +| `github/codeql-action` | `2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2` | `v4.38.2` | + +Both were resolved from the GitHub API (the `codeql-action` SHA corresponds to +a PGP-signed merge commit dated 2026-09-24). To re-derive, or to refresh to a +newer release: + +```bash +gh api repos/github/codeql-action/commits/v4 --jq .sha +gh api repos/actions/checkout/commits/v7 --jq .sha +``` + +A fabricated or mistyped SHA fails the workflow at action-resolution time with +an opaque error, so an unresolvable SHA is always left as an explicit TODO +rather than guessed. **No TODO placeholders remain in this workflow** — both +were resolved for real. + +When bumping, update the SHA *and* the `# vX.Y.Z` comment together; a bare SHA +with no version comment is unauditable in a diff. diff --git a/docs/V2_REWARD_ALLOCATION_AUDIT.md b/docs/V2_REWARD_ALLOCATION_AUDIT.md new file mode 100644 index 00000000..db75343b --- /dev/null +++ b/docs/V2_REWARD_ALLOCATION_AUDIT.md @@ -0,0 +1,266 @@ +# V2-BE-017 — Reward Allocation & Claim Audit + +> **Scope.** V2-BE-017 asks for projection of the five reward-allocation +> beneficiary classes — submitter, verifier, challenger, treasury, refund — +> with claimable/claimed tracking and reconciliation against emitted source +> pools, **without** the backend deciding any outcome. The issue also requires +> the contributor to audit overlapping existing code first and state, per +> allocation kind, what was already projected. +> +> This document is that audit. It is the primary deliverable of the change, +> not an appendix to it. + +--- + +## 1. Headline finding + +**None of the five allocation kinds was projected anywhere in the repository +before this change.** The audit below is per-kind, but the summary is uniform: +there was no submitter allocation, no verifier allocation, no challenger +allocation, no treasury allocation, and no refund allocation in any read model. + +What *did* exist was a **generic, allocation-blind** reward sync path +(`src/rewards/`) that records a whole distribution as an unordered +`recipients[]` / `amounts[]` pair. It cannot answer "how much is the treasury +owed", "how much has this verifier claimed", or "do the allocations reconcile +against the pool", because it never records a beneficiary *class*, never +records a per-beneficiary running total, and never records a pool to reconcile +against. So this is not a case of "the scope was already satisfied"; the +allocation projection is genuinely new. + +--- + +## 2. Per-allocation-kind audit + +Legend for the **Where** column: the file that projects the kind, if any. + +| # | Allocation kind | Already projected before this change? | Where | Added by this change | +| - | --------------- | ------------------------------------ | ----- | -------------------- | +| 1 | **Submitter** | ❌ No | — | `src/v2/rewards/rewards-projector.service.ts` → `v2_project_reward_allocation`, `kind = 'submitter'` | +| 2 | **Verifier** | ❌ No | — | same projector, `kind = 'verifier'` | +| 3 | **Challenger** | ❌ No | — | same projector, `kind = 'challenger'` | +| 4 | **Treasury** | ❌ No | — | same projector, `kind = 'treasury'`, `beneficiary` recorded as `null` (a sink, not an EOA) | +| 5 | **Refund** | ❌ No | — | same projector, `kind = 'refund'` | +| + | **Source pool** (reconciliation anchor) | ❌ No | — | `ProjectRewardPool`, from `RewardPoolSettled` | +| + | **Claim progress** (claimable → claimed) | ⚠️ Partial, allocation-blind | `src/rewards/services/reward-sync.service.ts` | Rebuilt as a per-allocation running total; see §4 | + +The five kinds are handled by **one** code path, not five. A beneficiary class +is a value read verbatim from the event's `kind` field; an event whose `kind` is +not one of the five is rejected and recorded as an indexing anomaly rather than +being coerced into a bucket. That is what keeps the five rows of the table above +honest — they are five *labels*, not five bespoke code paths to drift apart. + +--- + +## 3. What already existed, and how it is treated here + +### 3.1 Reused unchanged + +| Path | Why it was reused | +| ---- | ----------------- | +| `src/v2/events/canonical-events.service.ts` | Ingestion + normalization + quarantine. Unchanged. The reward projector consumes its output rather than touching RPC or ABIs. | +| `src/v2/events/canonical-event-query.service.ts` | Deterministic `(blockNumber, logIndex)`-ordered reads. Unchanged. | +| `src/v2/events/event-schema-registry.ts` | **Appended to** — three new event names added. See §5. | +| `src/v2/common/entities/projector-cursor.entity.ts` | Per-projector resumption cursor. Unchanged; the reward projector uses the same upsert-per-event pattern as its three siblings. | +| `src/v2/common/entities/indexing-anomaly.entity.ts` | Shared anomaly log. Unchanged — the reward projector writes `out_of_order`, `duplicate_event`, and `invalid_transition` using the existing kinds rather than extending the enum. | +| `src/indexer/reorg-safe-cursor.service.ts` — *pattern only* | The "advance state and record the write atomically" idea. See §6. | +| `src/outbox/outbox.service.ts` | The deterministic-identity + explicit-terminal-state pattern. `RewardClaimed` is idempotent by construction. | + +### 3.2 The projector pattern, copied not reinvented + +`rewards-projector.service.ts` is a deliberate structural clone of +`disputes-projector.service.ts` and `verification-projector.service.ts`: + +- same `processNewEvents(batchSize)` entry point and `ProjectorRunSummary`; +- same cursor read → `findAfter` → apply → cursor upsert loop; +- same `isUniqueViolation` check that recognises both Postgres `23505` and + `SQLITE_CONSTRAINT` (so the fast in-memory integration test and production + behave identically); +- same "replay vs. real duplicate" discrimination — on a unique violation, look + the row up by `(eventTxHash, eventLogIndex)`; if it is there, it is a safe + replay, otherwise it is a protocol-level fact worth an anomaly; +- same `readString` payload reader, same `recordAnomaly` swallow-on-duplicate. + +This is intentional. Four projectors that share one shape are four fewer places +for a correctness argument to have to be re-derived. + +### 3.3 Replaced / superseded — and what is *not* being deleted + +| Legacy path | Status | Rationale | +| ----------- | ------ | --------- | +| `src/rewards/services/reward-sync.service.ts` | **Superseded, left in place** | It records whole distributions as `recipients[]`/`amounts[]` with no beneficiary class and no per-beneficiary total. The V2 projection is strictly richer. Removing it is a separate change with its own migration and its own consumers to check (`src/analytics/analytics.service.ts` reads `bounty`/`treasury` rows through a raw query path). | +| `src/rewards/entities/reward-claim.entity.ts` | **Superseded, left in place** | Same reason. It is also the only place in the repo that uses `decimal(78,0)` columns; the V2 tables use `varchar(100)` decimal strings to match every other V2 read model. | +| `src/rewards/entities/reward-distribution.entity.ts` | **Superseded, left in place** | Its `amounts: string[]` / `recipients: string[]` shape cannot express "treasury got 12%", only "this unordered set got these amounts". | +| `src/rewards/entities/reward.entity.ts` | **Dead code** | The file is literally `export class Reward {}`. See §6. | +| `src/rewards/rewards.service.ts` | **Dead code** | Returns the string `'This action returns all rewards'`. See §6. | +| `src/rewards/services/blockchain-listener.service.ts` | **Bypassed, left in place** | Polls `RewardClaimed`/`RewardDistributed` via `ethers` against `INDEXED_CONTRACTS`. Not wired to canonical events. | + +**Nothing was deleted.** The issue asked for reused/replaced/deprecated to be +*identified in the pull request*, not for a teardown. Deleting the legacy paths +requires knowing their consumers, and that is a different, larger change. + +### 3.4 Deprecated by this change, in the architectural sense + +The **design** of the legacy distribution table is deprecated even though the +code is not removed: + +- An unordered `amounts[]` array is deprecated in favour of one row per + `(pool, kind, beneficiary)`. You cannot reconcile an array against a pool, + and you cannot answer "how much has *this* beneficiary claimed". +- A claim total is deprecated in favour of a running `claimedAmount` per + allocation, because a claim that cannot be attributed to an allocation must be + refused rather than summed into a global figure that hides the unattributable + one. + +--- + +## 4. The one real semantic decision: refusing an over-claim + +The projector **rejects** a `RewardClaimed` that would push an allocation's +`claimedAmount` above its `allocatedAmount`, records an +`invalid_transition` anomaly, and moves on. It does not record the claim. + +This is the load-bearing decision in the whole change, so it is worth stating +plainly why: + +- The alternative — recording it and letting `claimed > allocated` — would put + a row in the read model asserting that more was claimed than the contract ever + allocated. That is backend-authored protocol truth, which is precisely what + this repository must never produce. A reader querying the API would have no + way to tell a projected fact from an invented one. +- The alternative of *clamping* — recording `claimed = allocated` and dropping + the excess — is worse: it silently discards a chain fact and makes the + projection agree with itself while disagreeing with the chain. +- Rejecting keeps the invariant `claimed ≤ allocated` true in the table, and + keeps the violation **visible** in `v2_indexing_anomalies`, where an operator + will actually look. + +A `RewardClaimed` that cannot be attributed to a known allocation is treated the +same way. The projector will not infer a beneficiary, because inferring one +means the backend deciding whose balance moved. + +`rewards-reconciliation.service.ts` still *reports* an over-claim if one exists, +so a divergence arriving by any other route is loud rather than invisible. It +has no write path at all — it cannot correct anything. + +--- + +## 5. Files added or changed for #354 + +### Added + +| File | Purpose | +| ---- | ------- | +| `src/v2/rewards/reward-allocation-kind.enum.ts` | The five kinds + `parseAllocationKind`, which returns `null` for anything unrecognised. | +| `src/v2/rewards/entities/project-reward-allocation.entity.ts` | One row per emitted allocation. `allocatedAmount` verbatim, `claimedAmount` tracked. | +| `src/v2/rewards/entities/project-reward-pool.entity.ts` | The emitted pool total — the only thing allocations can be reconciled *against*. | +| `src/v2/rewards/reward-reconciliation.ts` | Pure `bigint` arithmetic. No clock, no DB, no network. | +| `src/v2/rewards/rewards-projector.service.ts` | The projector. | +| `src/v2/rewards/rewards-reconciliation.service.ts` | Read-side report. No write path. | +| `src/v2/rewards/v2-rewards.module.ts` | Module. **No controller** — see §7. | +| `src/v2/rewards/reward-reconciliation.spec.ts` | Pure unit tests. | +| `src/v2/rewards/rewards-projector.service.integration.spec.ts` | SQLite integration tests, modelled on `disputes-projector.service.integration.spec.ts`. | +| `src/migrations/1788100000000-CreateV2RewardAllocationTables.ts` | The two tables. | + +### Modified + +| File | Change | Why it was necessary | +| ---- | ------ | -------------------- | +| `src/v2/events/event-schema-registry.ts` | Appended `RewardPoolSettled`, `RewardAllocated`, `RewardClaimed` | Without a mapping, `EventDecoderService.normalize` returns `artifact_drift` and the events are **quarantined** — they would never reach `CanonicalEvent`, and the projector would have nothing to read. | +| `src/app.module.ts` | Registered `V2RewardsModule` | Otherwise the projector is never instantiated. | + +--- + +## 6. Findings outside this issue's scope + +Recorded here so they are not lost. **None of these was changed by this PR.** + +1. **`src/rewards/entities/reward.entity.ts` is `export class Reward {}`.** + Empty. Registered nowhere. +2. **`src/rewards/rewards.service.ts` returns placeholder strings** + (`'This action returns all rewards'`). `RewardsService` is registered in + `RewardsModule`, which `app.module.ts` imports, so it is live — and the + endpoints behind it return strings, not data. +3. **`ClaimResolutionService.computeConfidenceScore` derives a verdict from + vote weights and `resolveClaim` persists it** + (`src/claims/claim-resolution.service.ts`). This is backend-authoritative + settlement: the API decides `verdict` and `confidenceScore` from vote + totals. Under the project's own stated invariant — the chain is the only + authority for protocol truth — this is the single most significant + contradiction found during this audit. It is **not** touched here; it needs + its own issue, and removing it is a behavioural change, not a refactor. +4. **`src/health/health.service.ts` contains unresolved merge-conflict + remnants** — bare expression statements ` feat/be-016-monitoring-api` and + ` main` at lines 230, 241, 266 and 274 (and the same pattern in + `health.service.spec.ts` at lines 88, 95, 106 and 112), introduced by commit + `5d2af21 "Merge branch 'main' into feat/be-016-monitoring-api"`. These sit + inside function bodies, so the files will not type-check cleanly. Pre-existing + on `main`, unrelated to this work, and deliberately left alone — but it means + `npm run build` / `npm test` may currently fail for reasons that have nothing + to do with any open PR. **Check this before attributing a red build to this + change.** +5. **`src/contracts/contract-artifacts.loader.ts` and + `src/health/health-diagnostics.controller.ts` are dead code.** The loader + `process.exit(1)`s when `config/contracts/release-artifacts.json` is missing, + and that path is neither in the repository nor copied into the container + image. Harmless only because neither is registered as a provider or + controller in `src/app.module.ts`. If either is ever wired up, the + production image will refuse to start. Cross-referenced in + [`CONTAINER_IMAGE.md`](./CONTAINER_IMAGE.md) §5. +6. **`src/indexer/reorg-safe-cursor.service.ts` targets two tables that no + migration creates** (`v2_indexer_cursors`, `v2_projections`). It is not + registered anywhere. Cross-referenced in + [`PROJECTION_REBUILD.md`](./PROJECTION_REBUILD.md) §7. + +--- + +## 7. What this change deliberately does not do + +- **No endpoint that writes, adjusts, or approves an allocation.** The module + registers no controller. A `POST /rewards/...` that creates or changes an + allocation would be exactly the backend-authoritative settlement this + repository forbids. A read-only controller is a separate, additive change. +- **No `getTotalClaimedByWallet` shortcut across kinds.** The legacy + `RewardClaimRepository.getTotalClaimedByWallet` sums claims globally. A + per-kind, per-allocation report is provided instead, so a caller cannot + mistake a refund for a verifier reward. +- **No recomputation of a stake share, a slashing ratio, or a split.** Every + amount is either the event's `amount` or a `bigint` sum of amounts from events + the contract emitted. +- **No floating point anywhere.** Amounts are `varchar(100)` decimal strings; + all arithmetic is `bigint`. `reward-reconciliation.ts` throws on anything + that is not a base-10 integer string, so a malformed amount surfaces as a + defect instead of coercing to `0` — and `0` would read as "this beneficiary + was allocated nothing", which is a materially different claim from "we could + not read this event". + +--- + +## 8. The one assumption flagged for review + +Carried forward from every sibling projector, and stated in the same terms: + +> **V2-BE-008 (approved artifact import) has not landed, so there is no frozen +> ABI to read real argument names from.** The payload keys read here — `kind`, +> `beneficiary`, `sourcePoolId`, `allocationId`, `poolId` — follow the +> vocabulary of the V2-BE-017 issue text and the convention documented in +> `src/v2/events/event-schema-registry.ts`. They are expected to be reconciled +> against the real approved ABI once V2-BE-008 exists. Nothing in this change +> alters protocol meaning: it only says where to look for each field's value in +> whatever the approved ABI turns out to expose. + +The consequence to watch for: if the real ABI names these arguments +differently, the projector will reject events as `out_of_order` (rather than +mis-record them), the anomaly log will show it immediately, and the fix is a +rename in one registry plus one projector. That is the intended failure mode — +loud and cheap, not silent. + +--- + +## 9. Verification status + +**No install, build, lint, or test run was performed by the author of this +change.** The maintainer runs all of that. Nothing in this document should be +read as a claim that the code compiles, that the specs pass, or that the +migrations apply. Every count, row, and path above is derived by reading the +source, not by running it. diff --git a/src/app.module.ts b/src/app.module.ts index 05eacbd8..3c46c979 100644 --- a/src/app.module.ts +++ b/src/app.module.ts @@ -47,6 +47,8 @@ import { V2EventsModule } from './v2/events/v2-events.module'; import { V2EvidenceModule } from './v2/evidence/v2-evidence.module'; import { V2VerificationModule } from './v2/verification/v2-verification.module'; import { V2DisputesModule } from './v2/disputes/v2-disputes.module'; +import { V2RewardsModule } from './v2/rewards/v2-rewards.module'; +import { V2RebuildModule } from './v2/rebuild/v2-rebuild.module'; import { V2ProjectionModule } from './v2/projection/v2-projection.module'; import { ProfilerModule } from './profiler/profiler.module'; import { ProfilerInterceptor } from './profiler/profiler.interceptor'; @@ -337,6 +339,8 @@ async function createThrottlerStorage( V2EvidenceModule, V2VerificationModule, V2DisputesModule, + V2RewardsModule, + V2RebuildModule, V2ProjectionModule, ProfilerModule, HealthModule, diff --git a/src/migrations/1788100000000-CreateV2RewardAllocationTables.ts b/src/migrations/1788100000000-CreateV2RewardAllocationTables.ts new file mode 100644 index 00000000..241ffed9 --- /dev/null +++ b/src/migrations/1788100000000-CreateV2RewardAllocationTables.ts @@ -0,0 +1,115 @@ +import { MigrationInterface, QueryRunner } from 'typeorm'; + +/** + * V2-BE-017: reward allocation / claim read model tables. + * + * Both tables are pure projections of canonical contract events. Amounts are + * `varchar(100)` decimal strings — never `float`, never `double`, never a + * `numeric` column the driver would hand back as a JS number. That matches the + * convention already used by `v2_project_participant_position.stake` and + * `v2_canonical_events.amount`, and it is what keeps a 256-bit `uint256` exact. + */ +export class CreateV2RewardAllocationTables1788100000000 + implements MigrationInterface +{ + name = 'CreateV2RewardAllocationTables1788100000000'; + + public async up(queryRunner: QueryRunner): Promise { + await queryRunner.query(` + CREATE TABLE "v2_project_reward_pool" ( + "poolId" varchar(200) PRIMARY KEY, + "chainId" integer NOT NULL, + "claimId" varchar(66) NOT NULL, + "asset" varchar(42) NOT NULL, + "poolAmount" varchar(100) NOT NULL, + "eventTxHash" varchar(66) NOT NULL, + "eventLogIndex" integer NOT NULL, + "blockNumber" bigint NOT NULL, + "createdAt" TIMESTAMP NOT NULL DEFAULT now(), + CONSTRAINT "uq_v2_reward_pool_event" UNIQUE ("eventTxHash", "eventLogIndex") + ) + `); + await queryRunner.query( + `CREATE INDEX "idx_v2_reward_pool_claim_id" ON "v2_project_reward_pool" ("claimId")`, + ); + await queryRunner.query( + `CREATE INDEX "idx_v2_reward_pool_chain_id" ON "v2_project_reward_pool" ("chainId")`, + ); + + await queryRunner.query(` + CREATE TABLE "v2_project_reward_allocation" ( + "allocationId" varchar(200) PRIMARY KEY, + "chainId" integer NOT NULL, + "claimId" varchar(66) NOT NULL, + "roundId" varchar(66) NULL, + "sourcePoolId" varchar(200) NOT NULL, + "kind" varchar(16) NOT NULL, + "beneficiary" varchar(42) NULL, + "asset" varchar(42) NOT NULL, + "allocatedAmount" varchar(100) NOT NULL, + "claimedAmount" varchar(100) NOT NULL DEFAULT '0', + "lastClaimBlockNumber" bigint NULL, + "lastClaimEvent" varchar(140) NULL, + "eventTxHash" varchar(66) NOT NULL, + "eventLogIndex" integer NOT NULL, + "blockNumber" bigint NOT NULL, + "createdAt" TIMESTAMP NOT NULL DEFAULT now(), + "updatedAt" TIMESTAMP NOT NULL DEFAULT now() + ) + `); + await queryRunner.query( + `CREATE INDEX "idx_v2_reward_allocation_claim_id" ON "v2_project_reward_allocation" ("claimId")`, + ); + await queryRunner.query( + `CREATE INDEX "idx_v2_reward_allocation_source_pool" ON "v2_project_reward_allocation" ("sourcePoolId")`, + ); + await queryRunner.query( + `CREATE INDEX "idx_v2_reward_allocation_beneficiary" ON "v2_project_reward_allocation" ("beneficiary")`, + ); + await queryRunner.query( + `CREATE INDEX "idx_v2_reward_allocation_kind" ON "v2_project_reward_allocation" ("kind")`, + ); + await queryRunner.query( + `CREATE INDEX "idx_v2_reward_allocation_chain_id" ON "v2_project_reward_allocation" ("chainId")`, + ); + // Event-identity lookup, used by the projector's replay-vs-duplicate + // discrimination. Mirrors the equivalent unique constraint on the other V2 + // read models so a replayed event is always idempotent rather than + // ambiguous. + await queryRunner.query( + `CREATE UNIQUE INDEX "uq_v2_reward_allocation_event" ON "v2_project_reward_allocation" ("eventTxHash", "eventLogIndex")`, + ); + + // One row per emitted RewardClaimed event. This is the idempotency guard for + // the allocation's `claimedAmount` counter: without a row keyed on the + // emitting event, replaying a claim would double-count it. + await queryRunner.query(` + CREATE TABLE "v2_project_reward_claim" ( + "withdrawalId" varchar(200) PRIMARY KEY, + "chainId" integer NOT NULL, + "allocationId" varchar(200) NOT NULL, + "claimId" varchar(66) NOT NULL, + "beneficiary" varchar(42) NULL, + "asset" varchar(42) NULL, + "amount" varchar(100) NOT NULL, + "claimTxHash" varchar(66) NOT NULL, + "claimLogIndex" integer NOT NULL, + "blockNumber" bigint NOT NULL, + "createdAt" TIMESTAMP NOT NULL DEFAULT now(), + CONSTRAINT "uq_v2_reward_claim_event" UNIQUE ("chainId", "claimTxHash", "claimLogIndex") + ) + `); + await queryRunner.query( + `CREATE INDEX "idx_v2_reward_claim_allocation_id" ON "v2_project_reward_claim" ("allocationId")`, + ); + await queryRunner.query( + `CREATE INDEX "idx_v2_reward_claim_claim_id" ON "v2_project_reward_claim" ("claimId")`, + ); + } + + public async down(queryRunner: QueryRunner): Promise { + await queryRunner.query(`DROP TABLE "v2_project_reward_claim"`); + await queryRunner.query(`DROP TABLE "v2_project_reward_allocation"`); + await queryRunner.query(`DROP TABLE "v2_project_reward_pool"`); + } +} diff --git a/src/migrations/1788200000000-CreateProjectionRebuildRuns.ts b/src/migrations/1788200000000-CreateProjectionRebuildRuns.ts new file mode 100644 index 00000000..642f4301 --- /dev/null +++ b/src/migrations/1788200000000-CreateProjectionRebuildRuns.ts @@ -0,0 +1,58 @@ +import { MigrationInterface, QueryRunner } from 'typeorm'; + +/** + * V2-BE-019: durable checkpoint + audit row for projection rebuilds. + * + * One row per rebuild attempt. The deterministic report is stored both as + * queryable columns (so a run can be alerted on or listed without parsing + * JSON) and as the full serialised checkpoint in `checkpointJson`. + * + * This table is the *record* of a rebuild; it is never read by a serving read + * path, so a row existing does not make a rebuild authoritative. Only the + * documented cutover procedure — an operator action — can make a rebuilt read + * model live. + */ +export class CreateProjectionRebuildRuns1788200000000 + implements MigrationInterface +{ + name = 'CreateProjectionRebuildRuns1788200000000'; + + public async up(queryRunner: QueryRunner): Promise { + await queryRunner.query(` + CREATE TABLE "v2_projection_rebuild_runs" ( + "id" uuid PRIMARY KEY DEFAULT gen_random_uuid(), + "chainId" integer NOT NULL, + "status" varchar(16) NOT NULL, + "targetSchema" varchar(128) NULL, + "deploymentBlock" varchar(32) NOT NULL, + "fromBlock" varchar(32) NOT NULL, + "toBlock" varchar(32) NULL, + "inputDigest" varchar(64) NOT NULL, + "batchesProcessed" integer NOT NULL DEFAULT 0, + "eventsConsumed" integer NOT NULL DEFAULT 0, + "eventsApplied" integer NOT NULL DEFAULT 0, + "eventsSkipped" integer NOT NULL DEFAULT 0, + "anomalies" integer NOT NULL DEFAULT 0, + "safeToCutover" boolean NOT NULL DEFAULT false, + "checkpointJson" text NOT NULL, + "error" text NULL, + "startedAt" TIMESTAMP NOT NULL DEFAULT now(), + "updatedAt" TIMESTAMP NOT NULL DEFAULT now(), + "finishedAt" TIMESTAMP NULL + ) + `); + await queryRunner.query( + `CREATE INDEX "idx_v2_rebuild_runs_chain_status" ON "v2_projection_rebuild_runs" ("chainId", "status")`, + ); + // Resuming looks up the most recent non-completed run for a chain, so the + // index is on the startedAt ordering within a chain rather than on status + // alone. + await queryRunner.query( + `CREATE INDEX "idx_v2_rebuild_runs_chain_started" ON "v2_projection_rebuild_runs" ("chainId", "startedAt")`, + ); + } + + public async down(queryRunner: QueryRunner): Promise { + await queryRunner.query(`DROP TABLE "v2_projection_rebuild_runs"`); + } +} diff --git a/src/v2/events/event-schema-registry.ts b/src/v2/events/event-schema-registry.ts index 14228ddd..107b248e 100644 --- a/src/v2/events/event-schema-registry.ts +++ b/src/v2/events/event-schema-registry.ts @@ -17,8 +17,11 @@ * common across event types) carry the same flagged assumption: * VerificationRoundOpened.{roundType, roundNumber, deadline}, * PositionCommitted.{stake, reputationInput, effectiveWeight, verdict}, - * DisputeRaised.{deadline, appealRoundId}, DisputeResolved.{outcome}. See - * verification-projector.service.ts and disputes-projector.service.ts. + * DisputeRaised.{deadline, appealRoundId}, DisputeResolved.{outcome}, + * RewardPoolSettled.{poolId}, RewardAllocated.{kind, beneficiary, + * sourcePoolId, allocationId}, RewardClaimed.{allocationId, kind, + * beneficiary}. See verification-projector.service.ts, + * disputes-projector.service.ts and rewards-projector.service.ts. */ export interface EventFieldMapping { actor?: string; @@ -52,6 +55,26 @@ export const EVENT_SCHEMA_REGISTRY: Record = { }, DisputeResolved: { claimId: 'claimId', roundId: 'roundId' }, DisputeExpired: { claimId: 'claimId', roundId: 'roundId' }, + + // Reward allocations and claim progress (V2-BE-017) + RewardPoolSettled: { + claimId: 'claimId', + asset: 'asset', + amount: 'amount', + }, + RewardAllocated: { + actor: 'beneficiary', + claimId: 'claimId', + roundId: 'roundId', + asset: 'asset', + amount: 'amount', + }, + RewardClaimed: { + actor: 'beneficiary', + claimId: 'claimId', + asset: 'asset', + amount: 'amount', + }, }; /** Every event name this pipeline currently knows how to normalize. */ diff --git a/src/v2/rebuild/entities/projection-rebuild-run.entity.ts b/src/v2/rebuild/entities/projection-rebuild-run.entity.ts new file mode 100644 index 00000000..e0e03997 --- /dev/null +++ b/src/v2/rebuild/entities/projection-rebuild-run.entity.ts @@ -0,0 +1,133 @@ +import { + Entity, + PrimaryGeneratedColumn, + Column, + CreateDateColumn, + UpdateDateColumn, + Index, +} from 'typeorm'; +import { + RebuildCheckpoint, + RebuildStatus, + serializeCheckpoint, +} from '../rebuild-checkpoint'; + +/** + * Durable record of one projection-rebuild attempt. + * + * This is the **audit and checkpoint** row. The deterministic part of a rebuild + * — the counts, the per-projection breakdown, and the input digest — is stored + * twice: as individual columns so it is queryable and alertable, and as the + * serialised {@link RebuildCheckpoint} in {@link checkpointJson} so an operator + * can reproduce the exact report without re-deriving it. + * + * The deliberately non-deterministic fields (status transitions, timings, + * errors) live here and *not* in the checkpoint, which is what allows two + * rebuilds of the same data to be compared byte for byte. + */ +@Entity('v2_projection_rebuild_runs') +@Index(['chainId', 'status']) +export class ProjectionRebuildRun { + @PrimaryGeneratedColumn('uuid') + id: string; + + @Column({ type: 'int' }) + chainId: number; + + @Column({ type: 'varchar', length: 16 }) + status: RebuildStatus; + + /** + * The schema/namespace this rebuild was pointed at. Normally a **shadow** + * schema, never the live one — see `ProjectionRebuildService`'s guard and + * `docs/PROJECTION_REBUILD.md`. Null means "the live schema", which is only + * reachable when the operator explicitly passed `allowInPlace`. + */ + @Column({ type: 'varchar', length: 128, nullable: true }) + targetSchema: string | null; + + @Column({ type: 'varchar', length: 32 }) + deploymentBlock: string; + + @Column({ type: 'varchar', length: 32 }) + fromBlock: string; + + @Column({ type: 'varchar', length: 32, nullable: true }) + toBlock: string | null; + + @Column({ type: 'varchar', length: 64 }) + inputDigest: string; + + @Column({ type: 'int', default: 0 }) + batchesProcessed: number; + + @Column({ type: 'int', default: 0 }) + eventsConsumed: number; + + @Column({ type: 'int', default: 0 }) + eventsApplied: number; + + @Column({ type: 'int', default: 0 }) + eventsSkipped: number; + + @Column({ type: 'int', default: 0 }) + anomalies: number; + + @Column({ type: 'boolean', default: false }) + safeToCutover: boolean; + + /** The full deterministic report, serialised with sorted keys. */ + @Column({ type: 'text' }) + checkpointJson: string; + + @Column({ type: 'text', nullable: true }) + error: string | null; + + @CreateDateColumn() + startedAt: Date; + + @UpdateDateColumn() + updatedAt: Date; + + @Column({ type: 'datetime', nullable: true }) + finishedAt: Date | null; +} + +/** Column projection used when writing a checkpoint onto a run row. */ +export type CheckpointColumns = Pick< + ProjectionRebuildRun, + | 'status' + | 'deploymentBlock' + | 'fromBlock' + | 'toBlock' + | 'inputDigest' + | 'batchesProcessed' + | 'eventsConsumed' + | 'eventsApplied' + | 'eventsSkipped' + | 'anomalies' + | 'safeToCutover' + | 'checkpointJson' +>; + +export function checkpointColumns( + checkpoint: RebuildCheckpoint, + status: RebuildStatus, +): CheckpointColumns { + return { + status, + deploymentBlock: checkpoint.deploymentBlock, + fromBlock: checkpoint.fromBlock, + toBlock: checkpoint.toBlock, + inputDigest: checkpoint.inputDigest, + batchesProcessed: checkpoint.batchesProcessed, + eventsConsumed: checkpoint.eventsConsumed, + eventsApplied: checkpoint.eventsApplied, + eventsSkipped: checkpoint.eventsSkipped, + anomalies: checkpoint.anomalies, + safeToCutover: checkpoint.safeToCutover, + // Sorted-key serialisation, so two structurally equal checkpoints produce + // byte-identical columns and the column can be diffed directly. + checkpointJson: serializeCheckpoint(checkpoint), + }; +} diff --git a/src/v2/rebuild/projection-rebuild.cli.ts b/src/v2/rebuild/projection-rebuild.cli.ts new file mode 100644 index 00000000..8bba727f --- /dev/null +++ b/src/v2/rebuild/projection-rebuild.cli.ts @@ -0,0 +1,181 @@ +import { Logger, Module } from '@nestjs/common'; +import { NestFactory } from '@nestjs/core'; +import { TypeOrmModule } from '@nestjs/typeorm'; +import { readFileSync } from 'fs'; +import { dataSource } from '../../config/data-source'; +import { RebuildCheckpoint } from './rebuild-checkpoint'; +import { ProjectionRebuildService } from './projection-rebuild.service'; +import { ProjectionRebuildModule } from './v2-rebuild.module'; + +/** + * Operator entry point for a projection rebuild. + * + * This is a standalone script, not an HTTP endpoint, on purpose: a rebuild + * truncates and re-derives every read model. Exposing it over HTTP would put a + * destructive, long-running, cluster-wide operation behind a request that any + * authenticated caller could repeat. An operator runs it deliberately, in a + * shell, against a database they have chosen. + * + * ## Usage + * + * ```bash + * # Shadow rebuild (the normal case) — set REBUILD_SCHEMA first + * REBUILD_SCHEMA=truthbounty_shadow \ + * npx ts-node src/v2/rebuild/projection-rebuild.cli.ts \ + * --chain-id 10 --deployment-block 126000000 --reset + * + * # Stop early and keep a resumable checkpoint + * ... --max-batches 200 > partial.json + * + * # Resume from that checkpoint file + * ... --resume-from partial.json > final.json + * + * # Deliberately rebuild the live schema (discouraged; see docs) + * ... --allow-in-place + * ``` + * + * The script prints the deterministic checkpoint as JSON on stdout and logs to + * stderr, so the report can be piped straight into `jq` or `diff`: + * + * ```bash + * diff <(jq -S . run-a.json) <(jq -S . run-b.json) # must be empty + * ``` + */ + +interface CliArgs { + chainId: number; + deploymentBlock: string; + eventBatchSize?: number; + maxBatches: number | null; + reset: boolean; + resumeFromFile?: string; + allowInPlace: boolean; +} + +function parseArgs(argv: string[]): CliArgs { + const raw: Record = {}; + for (let i = 0; i < argv.length; i += 1) { + const token = argv[i]; + if (!token.startsWith('--')) continue; + const key = token.slice(2); + const next = argv[i + 1]; + if (next === undefined || next.startsWith('--')) { + raw[key] = 'true'; + } else { + raw[key] = next; + i += 1; + } + } + + if (!raw['chain-id']) { + throw new Error('--chain-id is required'); + } + if (!raw['deployment-block']) { + throw new Error('--deployment-block is required'); + } + + return { + chainId: Number(raw['chain-id']), + // Kept as a string all the way into the service: a block number is a + // 256-bit quantity and must never be parsed into a JS number. + deploymentBlock: raw['deployment-block'], + eventBatchSize: raw['event-batch-size'] + ? Number(raw['event-batch-size']) + : undefined, + maxBatches: raw['max-batches'] ? Number(raw['max-batches']) : null, + reset: raw['reset'] === 'true', + resumeFromFile: raw['resume-from'], + allowInPlace: raw['allow-in-place'] === 'true', + }; +} + +/** + * Load a previous run's checkpoint. + * + * The whole checkpoint is needed, not just a block and a digest: the counters + * and per-projection breakdowns carry forward so a resumed report describes the + * rebuild as a whole. Resuming from a hand-picked subset of fields would + * silently produce a report that only covers the final leg. + */ +function loadCheckpoint(path: string): RebuildCheckpoint { + const parsed: unknown = JSON.parse(readFileSync(path, 'utf8')); + if (typeof parsed !== 'object' || parsed === null) { + throw new Error(`${path} does not contain a checkpoint object`); + } + const checkpoint = parsed as Partial; + for (const field of ['fromBlock', 'inputDigest'] as const) { + if (typeof checkpoint[field] !== 'string') { + throw new Error(`${path} is missing required checkpoint field "${field}"`); + } + } + return checkpoint as RebuildCheckpoint; +} + +/** + * Composition root for the CLI. It reuses the application's own + * `DataSource` options rather than re-deriving them, so the rebuild connects + * to exactly the database the migrations were applied to — no second, drifting + * copy of the connection configuration. + */ +@Module({ + imports: [TypeOrmModule.forRoot(dataSource.options), ProjectionRebuildModule], +}) +class ProjectionRebuildCliModule {} + +async function main(): Promise { + const logger = new Logger('ProjectionRebuildCli'); + const args = parseArgs(process.argv.slice(2)); + + if (args.resumeFromFile && args.reset) { + throw new Error( + '--reset and --resume-from are mutually exclusive: --reset clears the ' + + 'projections, which would discard the very state the resume point ' + + 'refers to', + ); + } + + const app = await NestFactory.createApplicationContext( + ProjectionRebuildCliModule, + { logger: ['error', 'warn', 'log'] }, + ); + + try { + const service = app.get(ProjectionRebuildService); + const checkpoint = await service.rebuild({ + chainId: args.chainId, + deploymentBlock: args.deploymentBlock, + eventBatchSize: args.eventBatchSize, + maxBatches: args.maxBatches, + resetProjections: args.reset, + resumeFrom: args.resumeFromFile + ? loadCheckpoint(args.resumeFromFile) + : null, + allowInPlace: args.allowInPlace, + }); + + // stdout is the report; keep it clean so it can be piped. + process.stdout.write(`${service.render(checkpoint)}\n`); + + if (!checkpoint.safeToCutover) { + logger.error( + 'Rebuild is NOT safe to cut over. Do not swap the read model. ' + + `complete=${checkpoint.complete} anomalies=${checkpoint.anomalies} ` + + `unclaimedEvents=${checkpoint.unclaimedEvents}`, + ); + process.exitCode = 2; + } else { + logger.log( + 'Rebuild complete and safe to cut over. Follow the swap procedure in ' + + 'docs/PROJECTION_REBUILD.md (this script never performs it).', + ); + } + } finally { + await app.close(); + } +} + +void main().catch((error: unknown) => { + // eslint-disable-next-line no-console + console.error(error instanceof Error ? error.message : String(error)); + process.exit(1); +}); diff --git a/src/v2/rebuild/projection-rebuild.service.integration.spec.ts b/src/v2/rebuild/projection-rebuild.service.integration.spec.ts new file mode 100644 index 00000000..95d92fd9 --- /dev/null +++ b/src/v2/rebuild/projection-rebuild.service.integration.spec.ts @@ -0,0 +1,529 @@ +import { Test, TestingModule } from '@nestjs/testing'; +import { TypeOrmModule } from '@nestjs/typeorm'; +import { DataSource } from 'typeorm'; +import { ProjectionRebuildService } from './projection-rebuild.service'; +import { PROJECTION_REGISTRY, RebuildableProjection, buildDefaultRegistry } from './projection-registry'; +import { ProjectionRebuildRun } from './entities/projection-rebuild-run.entity'; +import { CanonicalEvent } from '../events/entities/canonical-event.entity'; +import { CanonicalEventQueryService } from '../events/canonical-event-query.service'; +import { ProjectorCursor } from '../common/entities/projector-cursor.entity'; +import { IndexingAnomaly } from '../common/entities/indexing-anomaly.entity'; +import { ProjectEvidence } from '../evidence/entities/project-evidence.entity'; +import { ProjectEvidenceVersion } from '../evidence/entities/project-evidence-version.entity'; +import { EvidenceProjectorService } from '../evidence/evidence-projector.service'; +import { ProjectVerificationRound } from '../verification/entities/project-verification-round.entity'; +import { ProjectParticipantPosition } from '../verification/entities/project-participant-position.entity'; +import { VerificationProjectorService } from '../verification/verification-projector.service'; +import { ProjectDispute } from '../disputes/entities/project-dispute.entity'; +import { DisputesProjectorService } from '../disputes/disputes-projector.service'; +import { ProjectRewardAllocation } from '../rewards/entities/project-reward-allocation.entity'; +import { ProjectRewardPool } from '../rewards/entities/project-reward-pool.entity'; +import { ProjectRewardClaim } from '../rewards/entities/project-reward-claim.entity'; +import { RewardsProjectorService } from '../rewards/rewards-projector.service'; + +const CLAIM_ID = '0x' + '11'.repeat(32); +const ROUND_ID = '0x' + 'aa'.repeat(32); +const ASSET = '0x' + '33'.repeat(20); +const ACTOR = '0x' + '22'.repeat(20); + +const DEPLOYMENT_BLOCK = '1000'; + +describe('ProjectionRebuildService (integration)', () => { + let moduleRef: TestingModule; + let service: ProjectionRebuildService; + let dataSource: DataSource; + const originalSchema = process.env.REBUILD_SCHEMA; + + beforeEach(async () => { + // A shadow target, so the live-schema guard is satisfied by the normal path. + process.env.REBUILD_SCHEMA = 'truthbounty_shadow_test'; + + moduleRef = await Test.createTestingModule({ + imports: [ + TypeOrmModule.forRoot({ + type: 'sqlite', + database: ':memory:', + // eslint-disable-next-line @typescript-eslint/no-require-imports + driver: require('sqlite3'), + entities: [ + CanonicalEvent, + ProjectorCursor, + IndexingAnomaly, + ProjectEvidence, + ProjectEvidenceVersion, + ProjectVerificationRound, + ProjectParticipantPosition, + ProjectDispute, + ProjectRewardAllocation, + ProjectRewardPool, + ProjectRewardClaim, + ProjectionRebuildRun, + ], + synchronize: true, + }), + TypeOrmModule.forFeature([CanonicalEvent, ProjectionRebuildRun]), + ], + providers: [ + ProjectionRebuildService, + CanonicalEventQueryService, + EvidenceProjectorService, + VerificationProjectorService, + DisputesProjectorService, + RewardsProjectorService, + { + provide: PROJECTION_REGISTRY, + inject: [ + DataSource, + EvidenceProjectorService, + VerificationProjectorService, + DisputesProjectorService, + RewardsProjectorService, + ], + useFactory: ( + ds: DataSource, + evidence: EvidenceProjectorService, + verification: VerificationProjectorService, + disputes: DisputesProjectorService, + rewards: RewardsProjectorService, + ) => + buildDefaultRegistry(ds, { evidence, verification, disputes, rewards }), + }, + ], + }).compile(); + + service = moduleRef.get(ProjectionRebuildService); + dataSource = moduleRef.get(DataSource); + }); + + afterEach(async () => { + await moduleRef.close(); + if (originalSchema === undefined) { + delete process.env.REBUILD_SCHEMA; + } else { + process.env.REBUILD_SCHEMA = originalSchema; + } + }); + + async function seed(overrides: Partial): Promise { + await dataSource.getRepository(CanonicalEvent).insert({ + chainId: 10, + contractAddress: '0x' + 'aa'.repeat(20), + artifactVersion: 'v1', + txHash: '0x' + '00'.repeat(32), + logIndex: 0, + blockNumber: '1', + claimId: CLAIM_ID, + payload: {} as object, + rawArgs: {} as object, + ...overrides, + }); + } + + /** + * A canonical event log spanning blocks 900..1100, including one event + * *before* the deployment block. The pre-deployment event must be ignored. + */ + async function seedLog(): Promise { + // Before the deployment block — must be excluded by the rebuild. + await seed({ + eventName: 'EvidenceRegistered', + txHash: '0x' + 'e0'.repeat(32), + blockNumber: '900', + actor: ACTOR, + payload: { digest: '0xdeadbeef' }, + }); + + await seed({ + eventName: 'VerificationRoundOpened', + txHash: '0x' + 'e1'.repeat(32), + blockNumber: '1001', + logIndex: 0, + claimId: CLAIM_ID, + roundId: ROUND_ID, + payload: { roundType: 'first', roundNumber: '1' }, + }); + await seed({ + eventName: 'PositionCommitted', + txHash: '0x' + 'e2'.repeat(32), + blockNumber: '1002', + logIndex: 0, + claimId: CLAIM_ID, + roundId: ROUND_ID, + actor: ACTOR, + payload: { stake: '1000', verdict: 'true' }, + }); + await seed({ + eventName: 'DisputeRaised', + txHash: '0x' + 'e3'.repeat(32), + blockNumber: '1003', + logIndex: 0, + claimId: CLAIM_ID, + roundId: ROUND_ID, + actor: ACTOR, + asset: ASSET, + amount: '5000', + payload: {}, + }); + await seed({ + eventName: 'RewardPoolSettled', + txHash: '0x' + 'e4'.repeat(32), + blockNumber: '1004', + logIndex: 0, + claimId: CLAIM_ID, + asset: ASSET, + amount: '10000', + payload: { poolId: 'pool-1', amount: '10000' }, + }); + await seed({ + eventName: 'RewardAllocated', + txHash: '0x' + 'e5'.repeat(32), + blockNumber: '1005', + logIndex: 0, + claimId: CLAIM_ID, + roundId: ROUND_ID, + actor: ACTOR, + asset: ASSET, + amount: '10000', + payload: { + kind: 'verifier', + beneficiary: ACTOR, + sourcePoolId: 'pool-1', + amount: '10000', + }, + }); + } + + describe('cutover safety guard', () => { + it('refuses to run without a shadow target and without allowInPlace', async () => { + delete process.env.REBUILD_SCHEMA; + await expect( + service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + allowInPlace: false, + }), + ).rejects.toThrow(/REBUILD_SCHEMA/); + }); + + it('rejects a non-integer deployment block', async () => { + await expect( + service.rebuild({ + chainId: 10, + deploymentBlock: 'not-a-block', + }), + ).rejects.toThrow(/base-10 integer string/); + }); + + it('never marks a partial run safe to cut over', async () => { + await seedLog(); + const checkpoint = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + maxBatches: 1, + }); + + expect(checkpoint.complete).toBe(false); + expect(checkpoint.safeToCutover).toBe(false); + }); + }); + + describe('replay from the deployment block', () => { + it('rebuilds every projection and reports concrete counts', async () => { + await seedLog(); + + const checkpoint = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + + expect(checkpoint.complete).toBe(true); + expect(checkpoint.eventsConsumed).toBe(5); // 6 seeded, 1 pre-deployment + expect(checkpoint.anomalies).toBe(0); + expect(checkpoint.unclaimedEvents).toBe(0); + expect(checkpoint.safeToCutover).toBe(true); + + expect(checkpoint.perProjection['v2-verification'].eventsApplied).toBe(2); + expect(checkpoint.perProjection['v2-verification'].rowsInTable).toBe(2); + expect(checkpoint.perProjection['v2-disputes'].eventsApplied).toBe(1); + expect(checkpoint.perProjection['v2-disputes'].rowsInTable).toBe(1); + expect(checkpoint.perProjection['v2-rewards'].eventsApplied).toBe(2); + expect(checkpoint.perProjection['v2-rewards'].rowsInTable).toBe(2); + expect(checkpoint.perProjection['v2-evidence'].eventsApplied).toBe(0); + expect(checkpoint.perProjection['v2-evidence'].rowsInTable).toBe(0); + }); + + it('ignores canonical events before the deployment block', async () => { + await seedLog(); + + const checkpoint = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + + // The only pre-deployment event is an EvidenceRegistered. If the + // deployment block were ignored, evidence would have one row. + expect(checkpoint.perProjection['v2-evidence'].eventsConsumed).toBe(0); + expect( + await dataSource.getRepository(ProjectEvidence).find(), + ).toHaveLength(0); + }); + + it('is idempotent: re-running the same range re-applies nothing', async () => { + await seedLog(); + const first = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + expect(first.complete).toBe(true); + + const second = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + }); + + // The canonical log is re-read (that is what a rebuild *is*), so the fold + // counter moves. What must not move is any projection's applied count: + // every write is guarded by the emitting event's unique constraint, so a + // replay lands as a no-op. + expect(second.eventsConsumed).toBe(5); + expect(second.eventsApplied).toBe(0); + for (const name of [ + 'v2-evidence', + 'v2-verification', + 'v2-disputes', + 'v2-rewards', + ]) { + expect(second.perProjection[name].eventsApplied).toBe(0); + } + // Same events, same digest. + expect(second.inputDigest).toBe(first.inputDigest); + + // And no duplicated state. + expect( + await dataSource.getRepository(ProjectRewardAllocation).find(), + ).toHaveLength(1); + expect( + await dataSource.getRepository(ProjectParticipantPosition).find(), + ).toHaveLength(1); + expect(await dataSource.getRepository(ProjectDispute).find()).toHaveLength( + 1, + ); + }); + + it('reproduces the same digest from a full reset', async () => { + await seedLog(); + const first = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + const second = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + + expect(second.perProjection['v2-rewards'].eventsApplied).toBe(2); + expect(second.safeToCutover).toBe(true); + expect(second.inputDigest).toBe(first.inputDigest); + expect( + await dataSource.getRepository(ProjectRewardAllocation).find(), + ).toHaveLength(1); + }); + + it('produces a byte-identical report for the same input', async () => { + await seedLog(); + const first = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + eventBatchSize: 500, + resetProjections: true, + }); + const second = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + eventBatchSize: 1, + resetProjections: true, + }); + + // Different batch sizes must not change the report. `batchesProcessed` + // legitimately differs, so compare everything else. + const strip = (c: typeof first) => ({ ...c, batchesProcessed: 0 }); + expect(service.render(strip(second))).toBe(service.render(strip(first))); + expect(second.inputDigest).toBe(first.inputDigest); + }); + + it('changes the digest when the canonical log changes', async () => { + await seedLog(); + const first = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + + await seed({ + eventName: 'DisputeExpired', + txHash: '0x' + 'e6'.repeat(32), + blockNumber: '1006', + logIndex: 0, + claimId: CLAIM_ID, + roundId: ROUND_ID, + payload: {}, + }); + + const second = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + + expect(second.inputDigest).not.toBe(first.inputDigest); + expect(second.eventsConsumed).toBe(first.eventsConsumed + 1); + }); + }); + + describe('resumability', () => { + it('records a checkpoint after a bounded run and resumes from it to the same digest', async () => { + await seedLog(); + + const partial = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + maxBatches: 1, + eventBatchSize: 1, + resetProjections: true, + }); + expect(partial.complete).toBe(false); + expect(partial.eventsConsumed).toBeGreaterThan(0); + // The resume point has advanced past the deployment block. + expect(BigInt(partial.fromBlock)).toBeGreaterThan( + BigInt(DEPLOYMENT_BLOCK), + ); + + const resumed = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + eventBatchSize: 1, + resumeFrom: partial, + }); + + expect(resumed.complete).toBe(true); + // Counters carry forward, so the resumed report describes the rebuild as + // a whole and is directly comparable to an uninterrupted run. + expect(resumed.eventsConsumed).toBe(5); + expect(resumed.unclaimedEvents).toBe(0); + + // The resumed fold must land on exactly the same accumulator as an + // uninterrupted run — that is the whole determinism claim. + const uninterrupted = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + eventBatchSize: 1, + resetProjections: true, + }); + expect(resumed.inputDigest).toBe(uninterrupted.inputDigest); + const strip = (c: typeof resumed) => ({ ...c, batchesProcessed: 0 }); + expect(service.render(strip(resumed))).toBe( + service.render(strip(uninterrupted)), + ); + }); + + it('persists an auditable run row with the checkpoint and a terminal status', async () => { + await seedLog(); + await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + + const runs = await service.recentRuns(); + expect(runs).toHaveLength(1); + expect(runs[0].status).toBe('completed'); + expect(runs[0].deploymentBlock).toBe(DEPLOYMENT_BLOCK); + expect(runs[0].eventsConsumed).toBe(5); + expect(runs[0].safeToCutover).toBe(true); + expect(runs[0].finishedAt).not.toBeNull(); + expect(JSON.parse(runs[0].checkpointJson).inputDigest).toBe( + runs[0].inputDigest, + ); + }); + + it('records a failed run row when the drain throws', async () => { + await seedLog(); + const registry = moduleRef.get(PROJECTION_REGISTRY) as RebuildableProjection[]; + const broken: RebuildableProjection = { + ...registry[0], + run: async () => { + throw new Error('simulated projector failure'); + }, + }; + + const failing = new ProjectionRebuildService(dataSource, [broken]); + await expect( + failing.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }), + ).rejects.toThrow('simulated projector failure'); + + const runs = await service.recentRuns(); + expect(runs[0].status).toBe('failed'); + expect(runs[0].error).toContain('simulated projector failure'); + }); + }); + + describe('reconciliation accounting', () => { + it('counts events no registered projection claims', async () => { + await seed({ + eventName: 'SomeFutureProtocolEvent', + txHash: '0x' + 'f0'.repeat(32), + blockNumber: '1050', + logIndex: 0, + payload: {}, + }); + + const checkpoint = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + + expect(checkpoint.unclaimedEvents).toBe(1); + // An unaccounted event blocks cutover: swapping in a read model that is + // missing a projection is silent data loss. + expect(checkpoint.safeToCutover).toBe(false); + }); + + it('counts a projector refusal as an anomaly and blocks cutover', async () => { + // A RewardClaimed for an allocation that was never emitted: the projector + // refuses to guess a beneficiary and records an anomaly. + await seed({ + eventName: 'RewardClaimed', + txHash: '0x' + 'f1'.repeat(32), + blockNumber: '1050', + logIndex: 0, + claimId: CLAIM_ID, + actor: ACTOR, + amount: '1', + payload: { amount: '1' }, + }); + + const checkpoint = await service.rebuild({ + chainId: 10, + deploymentBlock: DEPLOYMENT_BLOCK, + resetProjections: true, + }); + + expect(checkpoint.anomalies).toBeGreaterThan(0); + expect(checkpoint.perProjection['v2-rewards'].anomalies).toBeGreaterThan(0); + expect(checkpoint.safeToCutover).toBe(false); + expect( + await dataSource.getRepository(IndexingAnomaly).find(), + ).not.toHaveLength(0); + }); + }); +}); diff --git a/src/v2/rebuild/projection-rebuild.service.ts b/src/v2/rebuild/projection-rebuild.service.ts new file mode 100644 index 00000000..60f9287b --- /dev/null +++ b/src/v2/rebuild/projection-rebuild.service.ts @@ -0,0 +1,577 @@ +import { Inject, Injectable, Logger, BadRequestException } from '@nestjs/common'; +import { InjectDataSource } from '@nestjs/typeorm'; +import { DataSource, Repository } from 'typeorm'; +import { + RebuildCheckpoint, + ProjectionRebuildCounter, + RebuildStatus, + canonicalEventIdentity, + emptyCounter, + foldDigest, + initialDigest, + serializeCheckpoint, +} from './rebuild-checkpoint'; +import { + PROJECTION_REGISTRY, + ProjectorRunSummary, + RebuildableProjection, +} from './projection-registry'; +import { + ProjectionRebuildRun, + checkpointColumns, +} from './entities/projection-rebuild-run.entity'; +import { CanonicalEvent } from '../events/entities/canonical-event.entity'; +import { ProjectorCursor } from '../common/entities/projector-cursor.entity'; + +/** Default event batch handed to each projector per drain iteration. */ +const DEFAULT_EVENT_BATCH = 500; + +export interface ProjectionRebuildOptions { + chainId: number; + /** + * The block the deployment began at. Everything strictly before it is out of + * scope for the read models, so the rebuild starts here rather than at + * genesis. A string, because a deployment block is a 256-bit quantity that + * must not pass through a JS `number`. + */ + deploymentBlock: string; + /** Events per projector per drain iteration. */ + eventBatchSize?: number; + /** + * Stop after this many drain iterations and persist a resumable checkpoint. + * `null` (the default) drains to completion. + */ + maxBatches?: number | null; + /** + * Clear every registered projection's tables and its cursor before starting. + * Required for a from-scratch rebuild; the guard below additionally requires + * an out-of-band shadow target unless `allowInPlace` is set. + */ + resetProjections?: boolean; + /** + * Resume from a previous run's checkpoint instead of starting at + * `deploymentBlock`. + * + * The **whole** checkpoint is required, not just a block and a digest, + * because the report describes the rebuild as a whole: counters and + * per-projection breakdowns are carried forward and added to, so a resumed + * run's report is directly comparable to an uninterrupted one. Resuming with + * only a block would produce a report covering just the final leg. + */ + resumeFrom?: RebuildCheckpoint | null; + /** + * Acknowledge rebuilding the **live** schema. Without this the service + * refuses to run against the schema it is connected to, so a half-rebuilt + * read model can never be observed as authoritative. See §"Cutover safety" + * in `docs/PROJECTION_REBUILD.md`. + */ + allowInPlace?: boolean; +} + +/** + * Deterministic, resumable, idempotent rebuild of every V2 read model from the + * persisted canonical event log (V2-BE-019). + * + * ## Scope, stated honestly + * + * This rebuilds the **V2 read models from `v2_canonical_events`**, starting at + * a configured deployment block. It does not re-scan the chain. The canonical + * event log is itself produced by the ingestion path + * (`CanonicalEventsService.ingest`), and re-fetching it from RPC is a separate + * concern that `docs/indexer-runbook.md` already scopes to the indexer + * ("projections are rebuildable from raw, persisted events"). A rebuild into a + * genuinely empty schema therefore requires the canonical log to be populated + * first; the exact procedure is in `docs/PROJECTION_REBUILD.md`. + * + * ## Determinism + * + * Projections are drained in a fixed registry order until every one reports + * `processed === 0`. Within each iteration the canonical events in the newly + * covered block range are folded — in `(blockNumber, logIndex)` order, which is + * the protocol's own order — into a rolling SHA-256. Same deployment block plus + * same events yields the same digest, regardless of batch size or where the run + * was interrupted. See `rebuild-checkpoint.ts`. + * + * ## Idempotency and replay-safety + * + * Idempotency is inherited from the projectors, not reimplemented here. Each + * already guards writes with a unique constraint on `(eventTxHash, + * eventLogIndex)` and treats a violation as either a safe replay or a recorded + * anomaly. Re-running a completed rebuild therefore reads the same events, + * applies nothing, and produces a report whose per-projection `applied` counts + * are zero — which is itself the checkable proof that the first run was + * complete. + * + * ## Cutover safety + * + * The service never performs a cutover. It refuses to start against the live + * schema unless the operator explicitly passes `allowInPlace`, and it reports + * `safeToCutover` rather than acting on it. The swap is a deliberate, + * documented operator step (see `docs/PROJECTION_REBUILD.md`); the worst + * outcome of a half-finished rebuild here is a shadow database that is wrong, + * never a live one. + */ +@Injectable() +export class ProjectionRebuildService { + private readonly logger = new Logger(ProjectionRebuildService.name); + + constructor( + @InjectDataSource() private readonly dataSource: DataSource, + @Inject(PROJECTION_REGISTRY) + private readonly registry: RebuildableProjection[], + ) {} + + /** + * Run (or resume) a rebuild. Returns the deterministic checkpoint; a durable + * audit row is written for the run. + */ + async rebuild( + options: ProjectionRebuildOptions, + ): Promise { + const eventBatchSize = options.eventBatchSize ?? DEFAULT_EVENT_BATCH; + this.assertParsableDeploymentBlock(options.deploymentBlock); + this.assertShadowTarget(options); + + const runRepo = this.dataSource.getRepository(ProjectionRebuildRun); + const run = await runRepo.save( + runRepo.create({ + chainId: options.chainId, + status: 'running' satisfies RebuildStatus, + targetSchema: options.allowInPlace ? null : this.describeTargetSchema(), + deploymentBlock: options.deploymentBlock, + fromBlock: options.deploymentBlock, + toBlock: null, + inputDigest: initialDigest(), + batchesProcessed: 0, + eventsConsumed: 0, + eventsApplied: 0, + eventsSkipped: 0, + anomalies: 0, + safeToCutover: false, + checkpointJson: '', + error: null, + finishedAt: null, + }), + ); + + try { + if (options.resumeFrom) { + // A resumed run inherits the cursors the interrupted run left behind. + // Touching them here would rewind the drain and double-fold the digest. + this.logger.log( + `Resuming rebuild from block ${options.resumeFrom.fromBlock} ` + + `(carried digest ${options.resumeFrom.inputDigest})`, + ); + } else { + if (options.resetProjections) { + await this.resetProjections(); + } + await this.seedCursor(options.deploymentBlock); + } + + const checkpoint = await this.drain(run, options, eventBatchSize); + const status: RebuildStatus = checkpoint.complete + ? 'completed' + : 'aborted'; + await this.persistCheckpoint(run.id, checkpoint, status, null); + this.logger.log( + `Rebuild ${run.id} ${status}: ${checkpoint.eventsConsumed} events, ` + + `digest ${checkpoint.inputDigest}`, + ); + return checkpoint; + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + await runRepo.update(run.id, { status: 'failed', error: message.slice(0, 2000) }); + throw err; + } + } + + /** Serialise a checkpoint for byte-comparison between two runs. */ + render(checkpoint: RebuildCheckpoint): string { + return serializeCheckpoint(checkpoint); + } + + /** Most recent run rows, newest first, for operator inspection. */ + async recentRuns(limit = 20): Promise { + return this.dataSource + .getRepository(ProjectionRebuildRun) + .find({ order: { startedAt: 'DESC' }, take: limit }); + } + + // ─── Drain loop ───────────────────────────────────────────────────────── + + private async drain( + run: ProjectionRebuildRun, + options: ProjectionRebuildOptions, + eventBatchSize: number, + ): Promise { + const deploymentBlock = BigInt(options.deploymentBlock); + const prior = options.resumeFrom ?? null; + /** First block not yet folded into `digest`. */ + let nextBlockToFold = prior + ? BigInt(prior.fromBlock) + : deploymentBlock; + let digest = prior ? prior.inputDigest : initialDigest(); + + // Counters carry forward across a resume, so the final report describes the + // rebuild as a whole and is directly comparable to an uninterrupted run. + let eventsConsumed = prior?.eventsConsumed ?? 0; + let eventsApplied = prior?.eventsApplied ?? 0; + let eventsSkipped = prior?.eventsSkipped ?? 0; + let anomalies = prior?.anomalies ?? 0; + let unclaimedEvents = prior?.unclaimedEvents ?? 0; + let batchesProcessed = prior?.batchesProcessed ?? 0; + let complete = false; + let lastCursor: { blockNumber: bigint; logIndex: number } | null = null; + + const perProjection: Record = {}; + for (const projection of this.registry) { + perProjection[projection.name] = { ...emptyCounter() }; + } + if (prior) { + for (const [name, counter] of Object.entries(prior.perProjection)) { + if (perProjection[name]) { + perProjection[name] = { ...counter }; + } else { + // A projection that was registered for the interrupted run but is not + // in the registry now. Carried forward rather than dropped, so the + // two reports remain comparable — and surfaced in the log, because + // it means the registry changed mid-rebuild. + perProjection[name] = { ...counter }; + this.logger.warn( + `Resumed checkpoint references unknown projection ${name}; its ` + + 'counters are carried forward unchanged', + ); + } + } + } + + for (;;) { + const summaries: ProjectorRunSummary[] = []; + for (const projection of this.registry) { + summaries.push(await projection.run(eventBatchSize)); + } + + const drained = summaries.every((summary) => summary.processed === 0); + const cursor = await this.highestCursor(); + + if (cursor) { + // Fold the newly covered range exactly once. The projectors consume in + // strict (blockNumber, logIndex) order and each iteration runs every + // projection to exhaustion, so successive ranges are disjoint and + // together cover [nextBlockToFold, cursor.blockNumber]. + if (cursor.blockNumber >= nextBlockToFold) { + const slice = await this.readSlice( + options.chainId, + nextBlockToFold, + cursor.blockNumber, + ); + digest = foldDigest( + digest, + slice.map(canonicalEventIdentity), + ); + eventsConsumed += slice.length; + unclaimedEvents += this.countUnclaimed(slice); + nextBlockToFold = cursor.blockNumber + 1n; + } else if (!drained) { + // A projection reported work but no cursor moved. That is a real + // stall — a projector consuming events without advancing would + // otherwise spin here forever — so abort loudly rather than loop. + throw new Error( + `Rebuild stalled: projectors reported work but no cursor advanced past ` + + `block ${(nextBlockToFold - 1n).toString()}. Refusing to loop.`, + ); + } + lastCursor = cursor; + } + + for (let i = 0; i < this.registry.length; i += 1) { + const projection = this.registry[i]; + const summary = summaries[i]; + const counter = perProjection[projection.name]; + counter.eventsConsumed += summary.processed; + counter.eventsApplied += summary.applied; + counter.eventsSkipped += summary.processed - summary.applied; + counter.anomalies += summary.anomalies ?? 0; + eventsApplied += summary.applied; + eventsSkipped += summary.processed - summary.applied; + anomalies += summary.anomalies ?? 0; + } + + batchesProcessed += 1; + + if (drained) { + complete = true; + break; + } + + // Checkpoint after every non-final iteration: this is what makes a long + // rebuild resumable, and it is also the earliest point at which a crash + // leaves an auditable record of how far the run had got. + await this.persistCheckpoint( + run.id, + this.buildCheckpoint( + options, + nextBlockToFold, + lastCursor, + digest, + eventsConsumed, + eventsApplied, + eventsSkipped, + anomalies, + unclaimedEvents, + batchesProcessed, + perProjection, + false, + ), + 'running', + null, + ); + + if ( + options.maxBatches !== null && + options.maxBatches !== undefined && + batchesProcessed >= options.maxBatches + ) { + this.logger.warn( + `Rebuild ${run.id} stopped after ${batchesProcessed} batches at block ` + + `${lastCursor?.blockNumber.toString() ?? 'unknown'}`, + ); + break; + } + } + + for (const projection of this.registry) { + perProjection[projection.name].rowsInTable = await projection.countRows(); + } + + return this.buildCheckpoint( + options, + nextBlockToFold, + lastCursor, + digest, + eventsConsumed, + eventsApplied, + eventsSkipped, + anomalies, + unclaimedEvents, + batchesProcessed, + perProjection, + complete, + ); + } + + // ─── Guards ───────────────────────────────────────────────────────────── + + /** + * Refuse to start a rebuild against the live schema unless the operator has + * said so explicitly. + * + * The point is not ceremony: a rebuild truncates and re-derives every read + * model. If it is interrupted, the live schema is left holding a partially + * rebuilt projection that looks authoritative to every reader. Requiring an + * explicit acknowledgement makes that a decision rather than an accident. + */ + private assertShadowTarget(options: ProjectionRebuildOptions): void { + if (options.allowInPlace) { + this.logger.warn( + 'Rebuild running IN PLACE against the live schema. A partial rebuild is ' + + 'observable to readers until the run completes. Prefer a shadow schema.', + ); + return; + } + const schema = process.env.REBUILD_SCHEMA?.trim(); + if (!schema) { + throw new BadRequestException( + 'Refusing to rebuild without a shadow target. Set REBUILD_SCHEMA to the ' + + 'shadow schema/namespace to rebuild into, or pass allowInPlace: true to ' + + 'accept that a partial rebuild will be observable on the live schema. ' + + 'See docs/PROJECTION_REBUILD.md.', + ); + } + this.logger.log(`Rebuild target schema: ${schema}`); + } + + private describeTargetSchema(): string | null { + const schema = process.env.REBUILD_SCHEMA?.trim(); + return schema && schema.length > 0 ? schema : null; + } + + private assertParsableDeploymentBlock(deploymentBlock: string): void { + if (!/^\d+$/.test(deploymentBlock.trim())) { + throw new BadRequestException( + `deploymentBlock must be a base-10 integer string, received ${JSON.stringify(deploymentBlock)}`, + ); + } + } + + // ─── Plumbing ─────────────────────────────────────────────────────────── + + /** + * Clear every registered projection's tables and the shared projector cursor. + * + * Only reachable against a shadow target unless `allowInPlace` was set — the + * guard above runs first. + */ + private async resetProjections(): Promise { + for (const projection of this.registry) { + await projection.reset(); + } + await this.dataSource.getRepository(ProjectorCursor).clear(); + this.logger.log( + `Reset ${this.registry.length} projections and the shared projector cursor`, + ); + } + + /** + * Park every projector cursor at `(deploymentBlock - 1, -1)`. + * + * This is the whole trick for "start from a configured deployment block" + * without touching a single projector: each projector resumes via + * `CanonicalEventQueryService.findAfter(after)`, whose predicate is + * `blockNumber > after.blockNumber OR (blockNumber = after.blockNumber AND + * logIndex > after.logIndex)`. Feeding it `deploymentBlock - 1 / -1` therefore + * yields exactly the events at or after `deploymentBlock`, and no earlier. + * It reuses the projectors' own, already-tested resumption path rather than + * introducing a second one. + * + * The upsert matters as much as the value: a projector that finds **no** + * cursor row passes `after = null` to `findAfter`, which means *genesis*, not + * the deployment block. Simply updating existing rows would therefore leave a + * fresh or reset schema scanning the entire chain. + */ + private async seedCursor(deploymentBlock: string): Promise { + const cursorRepo = this.dataSource.getRepository(ProjectorCursor); + // Drop any cursor left behind by a projector that is no longer registered, + // so a stale row cannot make the drain think it is already past a block. + await cursorRepo.clear(); + + const preceding = (BigInt(deploymentBlock) - 1n).toString(); + for (const projection of this.registry) { + await cursorRepo.upsert( + { + projectorName: projection.projectorName, + lastBlockNumber: preceding, + lastLogIndex: -1, + }, + ['projectorName'], + ); + } + this.logger.log( + `Parked ${this.registry.length} projector cursors at block ${preceding}`, + ); + } + + /** + * The furthest point any projector cursor has reached, as an exact + * `(blockNumber, logIndex)` pair so a resumed run can pick up from precisely + * where this one stopped rather than re-scanning the block. + */ + private async highestCursor(): Promise<{ + blockNumber: bigint; + logIndex: number; + } | null> { + const cursors = await this.dataSource.getRepository(ProjectorCursor).find(); + let highest: { blockNumber: bigint; logIndex: number } | null = null; + for (const cursor of cursors) { + const blockNumber = BigInt(cursor.lastBlockNumber); + if ( + highest === null || + blockNumber > highest.blockNumber || + (blockNumber === highest.blockNumber && cursor.lastLogIndex > highest.logIndex) + ) { + highest = { blockNumber, logIndex: cursor.lastLogIndex }; + } + } + return highest; + } + + /** + * Read the canonical events in `[fromBlock, toBlock]` in protocol order. + * `blockNumber` is a `bigint` column, so the bounds are passed as strings and + * compared as bigints — never as JS numbers. + */ + private async readSlice( + chainId: number, + fromBlock: bigint, + toBlock: bigint, + ): Promise { + const repo: Repository = + this.dataSource.getRepository(CanonicalEvent); + return repo + .createQueryBuilder('e') + .where('e.chainId = :chainId', { chainId }) + .andWhere('e.blockNumber >= :fromBlock', { + fromBlock: fromBlock.toString(), + }) + .andWhere('e.blockNumber <= :toBlock', { + toBlock: toBlock.toString(), + }) + .orderBy('e.blockNumber', 'ASC') + .addOrderBy('e.logIndex', 'ASC') + .getMany(); + } + + /** Canonical events in the slice that no registered projection claims. */ + private countUnclaimed(slice: CanonicalEvent[]): number { + const claimed = new Set(); + for (const projection of this.registry) { + for (const name of projection.eventNames) claimed.add(name); + } + return slice.filter((event) => !claimed.has(event.eventName)).length; + } + + private async persistCheckpoint( + runId: string, + checkpoint: RebuildCheckpoint, + status: RebuildStatus, + error: string | null, + ): Promise { + await this.dataSource.getRepository(ProjectionRebuildRun).update(runId, { + ...checkpointColumns(checkpoint, status), + error, + finishedAt: status === 'running' ? null : new Date(), + }); + } + + private buildCheckpoint( + options: ProjectionRebuildOptions, + nextBlockToFold: bigint, + lastCursor: { blockNumber: bigint; logIndex: number } | null, + inputDigest: string, + eventsConsumed: number, + eventsApplied: number, + eventsSkipped: number, + anomalies: number, + unclaimedEvents: number, + batchesProcessed: number, + perProjection: Record, + complete: boolean, + ): RebuildCheckpoint { + return { + chainId: options.chainId, + deploymentBlock: options.deploymentBlock, + // Resume semantics: `fromBlock` is the first block *not yet folded*, so + // handing it back as `resumeFrom.fromBlock` continues the digest fold + // without double-counting or skipping a single canonical event. + fromBlock: nextBlockToFold.toString(), + toBlock: lastCursor ? lastCursor.blockNumber.toString() : null, + logIndex: lastCursor ? lastCursor.logIndex : -1, + batchesProcessed, + eventsConsumed, + eventsApplied, + eventsSkipped, + anomalies, + unclaimedEvents, + inputDigest, + perProjection, + // A rebuild is only cutover-eligible when it drained to completion, left + // no anomaly behind, and accounted for every canonical event. A + // non-zero `unclaimedEvents` means the registry is stale, and swapping in + // a projection that is missing a whole read model is exactly the silent + // data loss this pipeline exists to prevent. + safeToCutover: complete && anomalies === 0 && unclaimedEvents === 0, + complete, + }; + } +} diff --git a/src/v2/rebuild/projection-registry.ts b/src/v2/rebuild/projection-registry.ts new file mode 100644 index 00000000..abdcfe00 --- /dev/null +++ b/src/v2/rebuild/projection-registry.ts @@ -0,0 +1,172 @@ +import { DataSource, EntityTarget } from 'typeorm'; +import { ProjectEvidence } from '../evidence/entities/project-evidence.entity'; +import { ProjectEvidenceVersion } from '../evidence/entities/project-evidence-version.entity'; +import { EvidenceProjectorService } from '../evidence/evidence-projector.service'; +import { ProjectVerificationRound } from '../verification/entities/project-verification-round.entity'; +import { ProjectParticipantPosition } from '../verification/entities/project-participant-position.entity'; +import { VerificationProjectorService } from '../verification/verification-projector.service'; +import { ProjectDispute } from '../disputes/entities/project-dispute.entity'; +import { DisputesProjectorService } from '../disputes/disputes-projector.service'; +import { ProjectRewardAllocation } from '../rewards/entities/project-reward-allocation.entity'; +import { ProjectRewardPool } from '../rewards/entities/project-reward-pool.entity'; +import { ProjectRewardClaim } from '../rewards/entities/project-reward-claim.entity'; +import { RewardsProjectorService } from '../rewards/rewards-projector.service'; + +/** + * Counters a V2 projector reports after a drain batch. + * + * `anomalies` and `duplicates` are both optional because the existing + * projectors do not agree on a shape: `EvidenceProjectorService` reports + * `duplicates`, while the verification, disputes, and rewards projectors report + * `anomalies`. Both are accepted rather than "fixing" three projectors and four + * specs in a change whose subject is the rebuild pipeline. They are also + * genuinely different things — a duplicate is a safe no-op, an anomaly is a + * refused event — so they are counted separately rather than conflated. + */ +export interface ProjectorRunSummary { + processed: number; + applied: number; + anomalies?: number; + duplicates?: number; +} + +/** + * One read model the rebuild pipeline knows how to re-derive. + * + * This is a thin, uniform adapter over the projectors that already exist. It + * deliberately adds no projection logic of its own: `run()` calls the projector's + * own `processNewEvents`, and `countRows()` reads the projector's own tables. + * The rebuild pipeline's job is ordering, checkpointing, and reporting — not + * re-deriving state by a second, divergent code path. + */ +export interface RebuildableProjection { + /** Stable name, used as the report key. Must be deterministic. */ + readonly name: string; + /** + * The projector's own `ProjectorCursor.projectorName`. The rebuild pipeline + * needs this to park each cursor at the deployment block, because a + * projector that finds no cursor row starts from genesis rather than from + * the deployment block. + */ + readonly projectorName: string; + /** Canonical event names this projection consumes. */ + readonly eventNames: readonly string[]; + /** Drain one batch. `processed === 0` means this projection is caught up. */ + run(batchSize: number): Promise; + /** Row count across this projection's tables, for the report. */ + countRows(): Promise; + /** Remove all projected rows. Only ever called against a shadow target. */ + reset(): Promise; +} + +/** DI token for the ordered list of rebuildable projections. */ +export const PROJECTION_REGISTRY = Symbol('PROJECTION_REGISTRY'); + +function tableCounter( + dataSource: DataSource, + targets: EntityTarget[], +): () => Promise { + return async () => { + let total = 0; + for (const target of targets) { + total += await dataSource.getRepository(target).count(); + } + return total; + }; +} + +function tableResetter( + dataSource: DataSource, + targets: EntityTarget[], +): () => Promise { + return async () => { + for (const target of targets) { + await dataSource.getRepository(target).clear(); + } + }; +} + +/** + * The default registry, in **fixed order**. + * + * Order matters for reproducibility: projections are drained in this sequence + * every time, so two rebuilds apply the same events in the same interleaving. + * It is also the order the projectors consume the canonical log in + * independently, so a projection never sees an event before the projection it + * depends on has. + * + * Note what this list is *not*: it is not a list of every table in the schema. + * Legacy `src/rewards` (the `RewardClaim`/`RewardDistribution` tables), the + * `IndexedEvent` indexer tables, and the realtime `projection_events` outbox + * are all fed by different pipelines and are out of scope for a V2 projection + * rebuild. See `docs/PROJECTION_REBUILD.md` for that boundary. + * + * `eventNames` duplicates each projector's own private `HANDLED_EVENT_NAMES`, + * because the registry needs the set for reconciliation and the projectors + * correctly keep it private. That duplication is a drift risk, so the rebuild + * report carries `unclaimedEvents`: the count of canonical events in range that + * **no** registered projection claims. If a projector is registered with stale + * event names, that number stops being zero. + */ +export function buildDefaultRegistry( + dataSource: DataSource, + projectors: { + evidence: EvidenceProjectorService; + verification: VerificationProjectorService; + disputes: DisputesProjectorService; + rewards: RewardsProjectorService; + }, +): RebuildableProjection[] { + return [ + { + name: 'v2-evidence', + projectorName: 'v2-evidence', + eventNames: ['EvidenceRegistered', 'EvidenceReplaced', 'EvidenceRemoved'], + run: (batchSize) => projectors.evidence.processNewEvents(batchSize), + countRows: tableCounter(dataSource, [ProjectEvidence, ProjectEvidenceVersion]), + reset: tableResetter(dataSource, [ProjectEvidence, ProjectEvidenceVersion]), + }, + { + name: 'v2-verification', + projectorName: 'v2-verification', + eventNames: ['VerificationRoundOpened', 'PositionCommitted'], + run: (batchSize) => projectors.verification.processNewEvents(batchSize), + countRows: tableCounter(dataSource, [ + ProjectVerificationRound, + ProjectParticipantPosition, + ]), + reset: tableResetter(dataSource, [ + ProjectVerificationRound, + ProjectParticipantPosition, + ]), + }, + { + name: 'v2-disputes', + projectorName: 'v2-disputes', + eventNames: ['DisputeRaised', 'DisputeResolved', 'DisputeExpired'], + run: (batchSize) => projectors.disputes.processNewEvents(batchSize), + countRows: tableCounter(dataSource, [ProjectDispute]), + reset: tableResetter(dataSource, [ProjectDispute]), + }, + { + name: 'v2-rewards', + projectorName: 'v2-rewards', + eventNames: [ + 'RewardPoolSettled', + 'RewardAllocated', + 'RewardClaimed', + ], + run: (batchSize) => projectors.rewards.processNewEvents(batchSize), + countRows: tableCounter(dataSource, [ + ProjectRewardAllocation, + ProjectRewardPool, + ProjectRewardClaim, + ]), + reset: tableResetter(dataSource, [ + ProjectRewardAllocation, + ProjectRewardPool, + ProjectRewardClaim, + ]), + }, + ]; +} diff --git a/src/v2/rebuild/rebuild-checkpoint.spec.ts b/src/v2/rebuild/rebuild-checkpoint.spec.ts new file mode 100644 index 00000000..1cf1bf5f --- /dev/null +++ b/src/v2/rebuild/rebuild-checkpoint.spec.ts @@ -0,0 +1,147 @@ +import { + REBUILD_DIGEST_SEED, + RebuildCheckpoint, + canonicalEventIdentity, + emptyCounter, + foldDigest, + initialDigest, + serializeCheckpoint, +} from './rebuild-checkpoint'; + +const EVENTS = [ + { chainId: 10, txHash: '0xaa', logIndex: 0, eventName: 'EvidenceRegistered' }, + { chainId: 10, txHash: '0xaa', logIndex: 1, eventName: 'PositionCommitted' }, + { chainId: 10, txHash: '0xbb', logIndex: 0, eventName: 'RewardAllocated' }, +]; + +function identities( + events: ReadonlyArray<{ + chainId: number; + txHash: string; + logIndex: number; + eventName: string; + }>, +): string[] { + return events.map(canonicalEventIdentity); +} + +describe('rebuild checkpoint determinism', () => { + describe('canonicalEventIdentity', () => { + it('includes chain, transaction, log index, and event name', () => { + expect(canonicalEventIdentity(EVENTS[0])).toBe( + '10:0xaa:0:EvidenceRegistered', + ); + }); + + it('distinguishes two logs in the same transaction', () => { + expect(canonicalEventIdentity(EVENTS[0])).not.toBe( + canonicalEventIdentity(EVENTS[1]), + ); + }); + }); + + describe('foldDigest', () => { + it('is a pure function of the seed and the identity sequence', () => { + const a = foldDigest(initialDigest(), identities(EVENTS)); + const b = foldDigest(initialDigest(), identities(EVENTS)); + expect(a).toBe(b); + expect(a).toHaveLength(64); + }); + + it('is stable across batch boundaries — one batch equals three', () => { + const one = foldDigest(initialDigest(), identities(EVENTS)); + const three = foldDigest( + foldDigest( + foldDigest(initialDigest(), identities(EVENTS.slice(0, 1))), + identities(EVENTS.slice(1, 2)), + ), + identities(EVENTS.slice(2, 3)), + ); + expect(three).toBe(one); + }); + + it('is order-dependent, so a reordered log is detectable', () => { + const inOrder = foldDigest(initialDigest(), identities(EVENTS)); + const reordered = foldDigest( + initialDigest(), + identities([EVENTS[1], EVENTS[0], EVENTS[2]]), + ); + expect(reordered).not.toBe(inOrder); + }); + + it('changes when an event is added', () => { + const base = foldDigest(initialDigest(), identities(EVENTS)); + const extended = foldDigest( + initialDigest(), + identities([ + ...EVENTS, + { chainId: 10, txHash: '0xcc', logIndex: 0, eventName: 'DisputeRaised' }, + ]), + ); + expect(extended).not.toBe(base); + }); + + it('leaves the accumulator unchanged for an empty batch', () => { + const before = foldDigest(initialDigest(), identities(EVENTS)); + expect(foldDigest(before, [])).toBe(before); + }); + + it('starts from a fixed, published seed', () => { + expect(REBUILD_DIGEST_SEED).toBe('truthbounty:v2:projection-rebuild:v1'); + expect(initialDigest()).toHaveLength(64); + expect(initialDigest()).not.toBe(REBUILD_DIGEST_SEED); + }); + }); + + describe('serializeCheckpoint', () => { + const base = (): RebuildCheckpoint => ({ + chainId: 10, + deploymentBlock: '1000', + fromBlock: '1001', + toBlock: '1099', + logIndex: 3, + batchesProcessed: 2, + eventsConsumed: 3, + eventsApplied: 3, + eventsSkipped: 0, + anomalies: 0, + unclaimedEvents: 0, + inputDigest: 'a'.repeat(64), + perProjection: { + 'v2-rewards': emptyCounter(), + 'v2-evidence': emptyCounter(), + }, + safeToCutover: true, + complete: true, + }); + + it('serialises identically regardless of key insertion order', () => { + const a = base(); + const b: RebuildCheckpoint = { + ...base(), + perProjection: { + 'v2-evidence': emptyCounter(), + 'v2-rewards': emptyCounter(), + }, + }; + expect(serializeCheckpoint(b)).toBe(serializeCheckpoint(a)); + }); + + it('contains no timestamp, so two runs of the same data are byte-identical', () => { + const rendered = serializeCheckpoint(base()); + expect(rendered).not.toMatch(/\d{4}-\d{2}-\d{2}T/); + expect(rendered).not.toContain('runId'); + expect(rendered).not.toContain('durationMs'); + }); + + it('emits keys in sorted order at every level', () => { + const rendered = serializeCheckpoint(base()); + const topLevel = Object.keys(JSON.parse(rendered) as object); + expect(topLevel).toEqual([...topLevel].sort()); + const projections = Object.keys( + (JSON.parse(rendered) as { perProjection: object }).perProjection, + ); + expect(projections).toEqual([...projections].sort()); + }); + }); +}); diff --git a/src/v2/rebuild/rebuild-checkpoint.ts b/src/v2/rebuild/rebuild-checkpoint.ts new file mode 100644 index 00000000..82012e35 --- /dev/null +++ b/src/v2/rebuild/rebuild-checkpoint.ts @@ -0,0 +1,146 @@ +import { createHash } from 'crypto'; + +/** + * The deterministic part of a projection rebuild. + * + * ## What makes a rebuild report reproducible + * + * `RebuildCheckpoint` contains **no wall-clock time, no duration, no run id, + * and no host or environment detail**. It is a pure function of: + * + * (chainId, deploymentBlock, the ordered set of canonical events in range) + * + * Given the same deployment block and the same canonical event log, two + * rebuilds — on different machines, on different days, resumed differently, in + * different batch sizes — produce byte-identical checkpoints. That is the whole + * point: an operator can diff two reports and any difference is a *real* + * difference in the projection, not noise. + * + * Anything non-deterministic (timestamps, duration, host) lives in + * `ProjectionRebuildRun`, the database row, and is explicitly excluded from + * this structure. + */ + +export type RebuildStatus = 'running' | 'completed' | 'aborted' | 'failed'; + +/** Seed for the rolling digest. A constant, so the fold is reproducible. */ +export const REBUILD_DIGEST_SEED = 'truthbounty:v2:projection-rebuild:v1'; + +export interface ProjectionRebuildCounter { + /** Events this projector consumed from the canonical log. */ + eventsConsumed: number; + /** Events this projector applied to its read model. */ + eventsApplied: number; + /** Events this projector deliberately did not apply (duplicate/no-op). */ + eventsSkipped: number; + /** Events this projector refused and recorded as an indexing anomaly. */ + anomalies: number; + /** Rows present in this projector's table when the drain finished. */ + rowsInTable: number; +} + +export interface RebuildCheckpoint { + chainId: number; + /** The block the rebuild started from, as configured. */ + deploymentBlock: string; + /** Where this (possibly resumed) run picked up. */ + fromBlock: string; + /** Highest canonical block fully drained. Null until the drain completes. */ + toBlock: string | null; + /** Log index cursor at `toBlock`, for an exact resume point. */ + logIndex: number; + batchesProcessed: number; + /** Canonical events folded into `inputDigest` for this run. */ + eventsConsumed: number; + eventsApplied: number; + eventsSkipped: number; + anomalies: number; + /** + * Canonical events in the drained range that no registered projection claims. + * A non-zero value means the registry's `eventNames` are out of date — a new + * projector exists in the code but not in the rebuild registry, so the + * rebuilt read model would silently be missing a projection. + */ + unclaimedEvents: number; + /** + * Rolling SHA-256 over the ordered identities of every canonical event + * consumed. Order-dependent, so a reordering of the log is detectable. + */ + inputDigest: string; + perProjection: Record; + /** + * True only when the drain ran to completion, produced no anomalies, and + * wrote no partial state. This is the precondition the cutover procedure + * checks; the rebuild pipeline itself never performs a cutover. + */ + safeToCutover: boolean; + complete: boolean; +} + +/** One canonical event's stable identity, in the order it was consumed. */ +export function canonicalEventIdentity(event: { + chainId: number; + txHash: string; + logIndex: number; + eventName: string; +}): string { + return `${event.chainId}:${event.txHash}:${event.logIndex}:${event.eventName}`; +} + +/** + * Fold a batch of event identities into the rolling digest. + * + * A left fold, not a set hash, so: + * - it is order-dependent (a reordered log yields a different digest); + * - it is resumable, because the accumulator is the only state carried + * across a checkpoint boundary. + * + * Folding `[]` returns the previous digest unchanged, so an empty batch cannot + * perturb the result. + */ +export function foldDigest(previous: string, identities: string[]): string { + if (identities.length === 0) return previous; + const hash = createHash('sha256'); + hash.update(previous); + for (const identity of identities) { + hash.update('\n'); + hash.update(identity); + } + return hash.digest('hex'); +} + +/** The digest a rebuild produces before consuming anything. */ +export function initialDigest(): string { + return foldDigest(REBUILD_DIGEST_SEED, []); +} + +export function emptyCounter(): ProjectionRebuildCounter { + return { + eventsConsumed: 0, + eventsApplied: 0, + eventsSkipped: 0, + anomalies: 0, + rowsInTable: 0, + }; +} + +/** + * Serialise a checkpoint with **sorted keys** so that two structurally equal + * checkpoints serialise to identical bytes. `JSON.stringify` preserves + * insertion order, which would make the output depend on the order projections + * happened to be registered in — a real source of spurious diffs. + */ +export function serializeCheckpoint(checkpoint: RebuildCheckpoint): string { + return JSON.stringify(sortKeys(checkpoint), null, 2); +} + +function sortKeys(value: unknown): unknown { + if (Array.isArray(value)) return value.map(sortKeys); + if (value === null || typeof value !== 'object') return value; + const source = value as Record; + const sorted: Record = {}; + for (const key of Object.keys(source).sort()) { + sorted[key] = sortKeys(source[key]); + } + return sorted; +} diff --git a/src/v2/rebuild/v2-rebuild.module.ts b/src/v2/rebuild/v2-rebuild.module.ts new file mode 100644 index 00000000..13d964fd --- /dev/null +++ b/src/v2/rebuild/v2-rebuild.module.ts @@ -0,0 +1,75 @@ +import { Module } from '@nestjs/common'; +import { TypeOrmModule } from '@nestjs/typeorm'; +import { DataSource } from 'typeorm'; +import { V2EventsModule } from '../events/v2-events.module'; +import { V2EvidenceModule } from '../evidence/v2-evidence.module'; +import { V2VerificationModule } from '../verification/v2-verification.module'; +import { V2DisputesModule } from '../disputes/v2-disputes.module'; +import { V2RewardsModule } from '../rewards/v2-rewards.module'; +import { CanonicalEvent } from '../events/entities/canonical-event.entity'; +import { ProjectorCursor } from '../common/entities/projector-cursor.entity'; +import { IndexingAnomaly } from '../common/entities/indexing-anomaly.entity'; +import { ProjectionRebuildRun } from './entities/projection-rebuild-run.entity'; +import { ProjectionRebuildService } from './projection-rebuild.service'; +import { + PROJECTION_REGISTRY, + RebuildableProjection, + buildDefaultRegistry, +} from './projection-registry'; +import { EvidenceProjectorService } from '../evidence/evidence-projector.service'; +import { VerificationProjectorService } from '../verification/verification-projector.service'; +import { DisputesProjectorService } from '../disputes/disputes-projector.service'; +import { RewardsProjectorService } from '../rewards/rewards-projector.service'; + +/** + * V2-BE-019 — deterministic full projection rebuild. + * + * The registry is composed here rather than contributed by each projector + * module, deliberately: the rebuild's determinism rests on the registry being + * a **fixed, reviewable list in a fixed order**, and a distributed + * `forFeature`-style registration makes the effective order an accident of + * module wiring. One file, one order, one place to review. + */ +@Module({ + imports: [ + TypeOrmModule.forFeature([ + ProjectionRebuildRun, + CanonicalEvent, + ProjectorCursor, + IndexingAnomaly, + ]), + V2EventsModule, + V2EvidenceModule, + V2VerificationModule, + V2DisputesModule, + V2RewardsModule, + ], + providers: [ + ProjectionRebuildService, + { + provide: PROJECTION_REGISTRY, + inject: [ + DataSource, + EvidenceProjectorService, + VerificationProjectorService, + DisputesProjectorService, + RewardsProjectorService, + ], + useFactory: ( + dataSource: DataSource, + evidence: EvidenceProjectorService, + verification: VerificationProjectorService, + disputes: DisputesProjectorService, + rewards: RewardsProjectorService, + ): RebuildableProjection[] => + buildDefaultRegistry(dataSource, { + evidence, + verification, + disputes, + rewards, + }), + }, + ], + exports: [ProjectionRebuildService, PROJECTION_REGISTRY], +}) +export class V2RebuildModule {} diff --git a/src/v2/rewards/entities/project-reward-allocation.entity.ts b/src/v2/rewards/entities/project-reward-allocation.entity.ts new file mode 100644 index 00000000..b701ac61 --- /dev/null +++ b/src/v2/rewards/entities/project-reward-allocation.entity.ts @@ -0,0 +1,108 @@ +import { Entity, PrimaryColumn, Column, CreateDateColumn, UpdateDateColumn } from 'typeorm'; +import { AllocationKind } from '../reward-allocation-kind.enum'; + +/** + * A single reward allocation, projected from a canonical `RewardAllocated` + * event, plus the claim progress the contract has since emitted for it. + * + * ## What this table is allowed to know + * + * - `allocatedAmount` is the amount the contract said this beneficiary was + * allocated. It is stored **verbatim** from the event's `amount` field and is + * never recomputed, apportioned, or adjusted. + * - `claimedAmount` is the running sum of amounts from `RewardClaimed` events + * that the contract has already emitted for this allocation. It is *tracked*, + * not predicted: it moves only when a `RewardClaimed` event exists. + * - `status` is a pure function of the two amounts above + * (see `reward-reconciliation.ts`). It carries no independent meaning. + * + * ## What this table must never do + * + * It must never decide *who won*, *how much anyone is owed*, or *whether a + * claim is legitimately payable*. Those are settled on-chain. If the contract + * has not emitted an allocation, this table has no row and the API has no + * opinion; if the contract has emitted a claim this projector cannot attribute, + * that is an indexing anomaly, not something to infer a beneficiary from. + * + * ## Amount representation + * + * Decimal **strings** in `varchar(100)`, matching the rest of the V2 read + * models (`ProjectParticipantPosition.stake`, `ProjectDispute.challengeBond`, + * `CanonicalEvent.amount`). Never a float, never a JS number — a 256-bit + * `uint256` amount does not survive either. All arithmetic on these values is + * `bigint`, and it happens in `reward-reconciliation.ts`. + */ +@Entity('v2_project_reward_allocation') +export class ProjectRewardAllocation { + /** + * Protocol-supplied allocation id when the event carries one, otherwise a + * deterministic derivation from the creating event's identity + * (`chainId:txHash:logIndex`). Either way the value is a pure function of + * chain data, so a replay reconstructs the identical key. + */ + @PrimaryColumn({ type: 'varchar', length: 200 }) + allocationId: string; + + @Column({ type: 'int' }) + chainId: number; + + @Column({ type: 'varchar', length: 66 }) + claimId: string; + + @Column({ type: 'varchar', length: 66, nullable: true }) + roundId: string | null; + + /** The emitted source pool this allocation draws from. Reconciliation key. */ + @Column({ type: 'varchar', length: 200 }) + sourcePoolId: string; + + @Column({ type: 'varchar', length: 16 }) + kind: AllocationKind; + + /** + * The party the allocation is for. `null` for protocol sinks such as the + * treasury, which is not an externally-owned account. Never invented: this is + * the event's `beneficiary`, lowercased, or `null` when the event has none. + */ + @Column({ type: 'varchar', length: 42, nullable: true }) + beneficiary: string | null; + + @Column({ type: 'varchar', length: 42 }) + asset: string; + + /** Verbatim amount from the `RewardAllocated` event. Decimal string. */ + @Column({ type: 'varchar', length: 100 }) + allocatedAmount: string; + + /** + * Running sum of amounts from attributed `RewardClaimed` events. Decimal + * string. The projector rejects any claim that would push this above + * `allocatedAmount`, so this column never exceeds it. + */ + @Column({ type: 'varchar', length: 100, default: '0' }) + claimedAmount: string; + + /** Block of the most recent `RewardClaimed` event attributed to this row. */ + @Column({ type: 'bigint', nullable: true }) + lastClaimBlockNumber: string | null; + + /** `txHash:logIndex` of the most recent attributed claim, for auditing. */ + @Column({ type: 'varchar', length: 140, nullable: true }) + lastClaimEvent: string | null; + + /** Identity of the `RewardAllocated` event that created this row. */ + @Column({ type: 'varchar', length: 66 }) + eventTxHash: string; + + @Column({ type: 'int' }) + eventLogIndex: number; + + @Column({ type: 'bigint' }) + blockNumber: string; + + @CreateDateColumn() + createdAt: Date; + + @UpdateDateColumn() + updatedAt: Date; +} diff --git a/src/v2/rewards/entities/project-reward-claim.entity.ts b/src/v2/rewards/entities/project-reward-claim.entity.ts new file mode 100644 index 00000000..f0a5f7c0 --- /dev/null +++ b/src/v2/rewards/entities/project-reward-claim.entity.ts @@ -0,0 +1,77 @@ +import { + Entity, + PrimaryColumn, + Column, + CreateDateColumn, + Index, + Unique, +} from 'typeorm'; + +/** + * One emitted `RewardClaimed` event — a *withdrawal* from an allocation. + * + * ## Why this table exists + * + * An allocation's `claimedAmount` is a running total, and a running total is + * the one thing in this projection that is **not** naturally idempotent: if the + * same claim event were applied twice, the total would double, and the read + * model would assert a withdrawal the chain only made once. Every other write + * in the V2 projections is guarded by inserting the event's own row, which a + * unique constraint then makes replay-proof; a counter has no such row. + * + * So the counter is backed by rows. One row per emitted claim event, unique on + * `(chainId, claimTxHash, claimLogIndex)`. The projector inserts the row first + * and only then advances the allocation's total, so a replay hits the unique + * constraint and becomes a no-op. + * + * The rows are also the reconciliation anchor the issue asks for: the sum of + * withdrawal rows for an allocation must equal its `claimedAmount`, and that + * equality is checkable with a single query. A mismatch means the counter and + * the event log disagree, which is a fact about the projection worth surfacing + * — never something to paper over by rewriting the counter. + */ +@Entity('v2_project_reward_claim') +@Unique('uq_v2_reward_claim_event', ['chainId', 'claimTxHash', 'claimLogIndex']) +@Index(['allocationId']) +@Index(['claimId']) +export class ProjectRewardClaim { + /** + * Deterministic withdrawal identity: the protocol's own when the event + * supplies one, otherwise `chainId:txHash:logIndex` of the emitting event. + */ + @PrimaryColumn({ type: 'varchar', length: 200 }) + withdrawalId: string; + + @Column({ type: 'int' }) + chainId: number; + + /** The allocation this withdrawal was attributed to. */ + @Column({ type: 'varchar', length: 200 }) + allocationId: string; + + /** The protocol claim the allocation belongs to. */ + @Column({ type: 'varchar', length: 66 }) + claimId: string; + + @Column({ type: 'varchar', length: 42, nullable: true }) + beneficiary: string | null; + + @Column({ type: 'varchar', length: 42, nullable: true }) + asset: string | null; + + /** Verbatim claimed amount from the event. Decimal string. */ + @Column({ type: 'varchar', length: 100 }) + amount: string; + + @Column({ type: 'varchar', length: 66 }) + claimTxHash: string; + + @Column({ type: 'int' }) + claimLogIndex: number; + + @Column({ type: 'bigint' }) + blockNumber: string; + + @CreateDateColumn() + createdAt: Date; +} diff --git a/src/v2/rewards/entities/project-reward-pool.entity.ts b/src/v2/rewards/entities/project-reward-pool.entity.ts new file mode 100644 index 00000000..f9047b99 --- /dev/null +++ b/src/v2/rewards/entities/project-reward-pool.entity.ts @@ -0,0 +1,47 @@ +import { Entity, PrimaryColumn, Column, CreateDateColumn } from 'typeorm'; + +/** + * A source reward pool, projected from a canonical `RewardPoolSettled` event. + * + * This exists solely so that the sum of projected allocations can be compared + * against a number the **contract** emitted. Without an emitted pool total + * there is nothing to reconcile against, and "the allocations add up" would + * only mean "the allocations agree with each other". + * + * `poolAmount` is verbatim from the event. It is never used to *derive* an + * allocation, never used to top up a beneficiary, and never used to decide who + * was paid. A mismatch between `poolAmount` and the sum of allocations is + * reported as divergence (see `RewardsReconciliationService`); it is never + * silently corrected, because correcting it would mean the backend inventing a + * distribution the chain did not emit. + */ +@Entity('v2_project_reward_pool') +export class ProjectRewardPool { + @PrimaryColumn({ type: 'varchar', length: 200 }) + poolId: string; + + @Column({ type: 'int' }) + chainId: number; + + @Column({ type: 'varchar', length: 66 }) + claimId: string; + + @Column({ type: 'varchar', length: 42 }) + asset: string; + + /** Verbatim settled amount emitted by the contract. Decimal string. */ + @Column({ type: 'varchar', length: 100 }) + poolAmount: string; + + @Column({ type: 'varchar', length: 66 }) + eventTxHash: string; + + @Column({ type: 'int' }) + eventLogIndex: number; + + @Column({ type: 'bigint' }) + blockNumber: string; + + @CreateDateColumn() + createdAt: Date; +} diff --git a/src/v2/rewards/reward-allocation-kind.enum.ts b/src/v2/rewards/reward-allocation-kind.enum.ts new file mode 100644 index 00000000..e1e97fe8 --- /dev/null +++ b/src/v2/rewards/reward-allocation-kind.enum.ts @@ -0,0 +1,44 @@ +/** + * The five reward-allocation beneficiary classes defined by V2-BE-017. + * + * These are *labels the protocol itself emits*, not a taxonomy this service + * invented: the value written to {@link AllocationKind} is read verbatim from + * the `kind` field of a canonical `RewardAllocated` event, and an event whose + * `kind` is not one of these five is rejected and recorded as an indexing + * anomaly rather than being coerced into a bucket. + * + * - SUBMITTER — the party that submitted the claim evidence + * - VERIFIER — a participant whose verification position was rewarded + * - CHALLENGER — the party that raised a dispute whose position was rewarded + * - TREASURY — the protocol treasury sink (beneficiary is not an EOA) + * - REFUND — a returned bond/stake refund to a party that lost nothing + */ +export enum AllocationKind { + SUBMITTER = 'submitter', + VERIFIER = 'verifier', + CHALLENGER = 'challenger', + TREASURY = 'treasury', + REFUND = 'refund', +} + +export const ALLOCATION_KINDS: readonly AllocationKind[] = Object.freeze([ + AllocationKind.SUBMITTER, + AllocationKind.VERIFIER, + AllocationKind.CHALLENGER, + AllocationKind.TREASURY, + AllocationKind.REFUND, +]); + +/** + * Narrow an arbitrary payload value to a known {@link AllocationKind}. + * Returns `null` for anything unrecognised — callers must treat `null` as + * "reject and record an anomaly", never as a default. + */ +export function parseAllocationKind(raw: string | null): AllocationKind | null { + if (raw === null) return null; + const normalized = raw.trim().toLowerCase(); + for (const kind of ALLOCATION_KINDS) { + if (kind === normalized) return kind; + } + return null; +} diff --git a/src/v2/rewards/reward-reconciliation.spec.ts b/src/v2/rewards/reward-reconciliation.spec.ts new file mode 100644 index 00000000..ccf23b34 --- /dev/null +++ b/src/v2/rewards/reward-reconciliation.spec.ts @@ -0,0 +1,206 @@ +import { AllocationKind, parseAllocationKind } from './reward-allocation-kind.enum'; +import { + classifyAllocationStatus, + reconcileAllocation, + reconcilePool, + sumAmounts, +} from './reward-reconciliation'; + +/** + * Pure unit tests for the reconciliation arithmetic. No Nest, no TypeORM, no + * database — these functions are total and side-effect free precisely so that + * they can be exercised this cheaply and this exhaustively. + */ +describe('reward reconciliation arithmetic', () => { + describe('parseAllocationKind', () => { + it('accepts each of the five protocol beneficiary classes', () => { + expect(parseAllocationKind('submitter')).toBe(AllocationKind.SUBMITTER); + expect(parseAllocationKind('verifier')).toBe(AllocationKind.VERIFIER); + expect(parseAllocationKind('challenger')).toBe(AllocationKind.CHALLENGER); + expect(parseAllocationKind('treasury')).toBe(AllocationKind.TREASURY); + expect(parseAllocationKind('refund')).toBe(AllocationKind.REFUND); + }); + + it('normalises case and surrounding whitespace', () => { + expect(parseAllocationKind(' Verifier ')).toBe(AllocationKind.VERIFIER); + expect(parseAllocationKind('TREASURY')).toBe(AllocationKind.TREASURY); + }); + + it('returns null for an unrecognised or absent kind rather than defaulting', () => { + expect(parseAllocationKind('slashed')).toBeNull(); + expect(parseAllocationKind('')).toBeNull(); + expect(parseAllocationKind(null)).toBeNull(); + }); + }); + + describe('classifyAllocationStatus', () => { + it('reports "allocated" when nothing has been claimed', () => { + expect(classifyAllocationStatus(1000n, 0n)).toBe('allocated'); + }); + + it('reports "partially_claimed" for a partial claim', () => { + expect(classifyAllocationStatus(1000n, 1n)).toBe('partially_claimed'); + expect(classifyAllocationStatus(1000n, 999n)).toBe('partially_claimed'); + }); + + it('reports "claimed" at exactly the allocated amount', () => { + expect(classifyAllocationStatus(1000n, 1000n)).toBe('claimed'); + }); + + it('reports "over_claimed" beyond the allocated amount, never clamping', () => { + expect(classifyAllocationStatus(1000n, 1001n)).toBe('over_claimed'); + }); + }); + + describe('reconcileAllocation', () => { + const row = (overrides: Partial[0]>) => + reconcileAllocation({ + allocationId: 'alloc-1', + kind: AllocationKind.VERIFIER, + beneficiary: '0x' + '22'.repeat(20), + allocatedAmount: '1000', + claimedAmount: '0', + ...overrides, + }); + + it('computes the remaining claimable amount with exact integer math', () => { + const result = row({ claimedAmount: '250' }); + expect(result.claimableRemaining).toBe('750'); + expect(result.divergent).toBe(false); + expect(result.status).toBe('partially_claimed'); + }); + + it('keeps a negative remainder visible instead of clamping it to zero', () => { + const result = row({ claimedAmount: '1500' }); + expect(result.claimableRemaining).toBe('-500'); + expect(result.overClaimed).toBe(true); + expect(result.divergent).toBe(true); + expect(result.status).toBe('over_claimed'); + }); + + it('is exact well past Number.MAX_SAFE_INTEGER', () => { + // 2^256-ish territory. A float-based implementation silently loses the + // low digits here; bigint arithmetic must not. + const huge = '115792089237316195423570985008687907853269984665640564039457584007913129639935'; + const result = row({ allocatedAmount: huge, claimedAmount: '1' }); + expect(result.claimableRemaining).toBe( + ( + BigInt(huge) - 1n + ).toString(), + ); + expect(result.claimed).toBe('1'); + }); + + it('throws on a non-integer amount rather than coercing it to zero', () => { + expect(() => row({ allocatedAmount: '1.5' })).toThrow( + /base-10 integer string/, + ); + expect(() => row({ claimedAmount: '1e18' })).toThrow( + /base-10 integer string/, + ); + expect(() => row({ claimedAmount: '' })).toThrow(/base-10 integer string/); + }); + }); + + describe('reconcilePool', () => { + const pool = (overrides: Partial[0]>) => + reconcilePool({ + poolId: 'pool-1', + claimId: '0x' + '11'.repeat(32), + asset: '0x' + '33'.repeat(20), + poolAmount: '10000', + allocations: [], + ...overrides, + }); + + it('reports balance when the allocations sum to the emitted pool amount', () => { + const result = pool({ + allocations: [ + { kind: AllocationKind.SUBMITTER, allocatedAmount: '6000' }, + { kind: AllocationKind.VERIFIER, allocatedAmount: '2500' }, + { kind: AllocationKind.CHALLENGER, allocatedAmount: '1000' }, + { kind: AllocationKind.TREASURY, allocatedAmount: '500' }, + ], + }); + + expect(result.allocatedTotal).toBe('10000'); + expect(result.divergence).toBe('0'); + expect(result.divergent).toBe(false); + expect(result.allocationCount).toBe(4); + }); + + it('breaks the pool down by beneficiary class', () => { + const result = pool({ + allocations: [ + { kind: AllocationKind.SUBMITTER, allocatedAmount: '6000' }, + { kind: AllocationKind.REFUND, allocatedAmount: '4000' }, + ], + }); + + expect(result.byKind[AllocationKind.SUBMITTER]).toBe('6000'); + expect(result.byKind[AllocationKind.REFUND]).toBe('4000'); + expect(result.byKind[AllocationKind.VERIFIER]).toBe('0'); + expect(result.byKind[AllocationKind.CHALLENGER]).toBe('0'); + expect(result.byKind[AllocationKind.TREASURY]).toBe('0'); + }); + + it('reports a signed divergence when allocations under-account for the pool', () => { + const result = pool({ + allocations: [ + { kind: AllocationKind.SUBMITTER, allocatedAmount: '6000' }, + ], + }); + + expect(result.allocatedTotal).toBe('6000'); + expect(result.divergence).toBe('4000'); + expect(result.divergent).toBe(true); + }); + + it('reports a negative divergence when allocations over-account for the pool', () => { + const result = pool({ + poolAmount: '1000', + allocations: [ + { kind: AllocationKind.SUBMITTER, allocatedAmount: '1500' }, + ], + }); + + expect(result.divergence).toBe('-500'); + expect(result.divergent).toBe(true); + }); + + it('treats a pool with no projected allocations as divergent, not balanced', () => { + const result = pool({ allocations: [] }); + expect(result.allocatedTotal).toBe('0'); + expect(result.divergence).toBe('10000'); + expect(result.divergent).toBe(true); + }); + + it('treats a zero-value pool with no allocations as balanced', () => { + const result = pool({ poolAmount: '0', allocations: [] }); + expect(result.divergent).toBe(false); + }); + }); + + describe('sumAmounts', () => { + it('rejects decimal fractions outright, so no float arithmetic is reachable', () => { + expect(() => sumAmounts(['0.1', '0.2'])).toThrow(/base-10 integer string/); + }); + + it('sums an empty list to zero', () => { + expect(sumAmounts([])).toBe('0'); + }); + + it('sums values beyond double precision exactly', () => { + expect( + sumAmounts([ + '9007199254740993', // 2^53 + 1, not representable as a double + '1', + ]), + ).toBe('9007199254740994'); + }); + + it('rejects a non-integer entry rather than silently dropping it', () => { + expect(() => sumAmounts(['1', 'nope'])).toThrow(/base-10 integer string/); + }); + }); +}); diff --git a/src/v2/rewards/reward-reconciliation.ts b/src/v2/rewards/reward-reconciliation.ts new file mode 100644 index 00000000..a0d34930 --- /dev/null +++ b/src/v2/rewards/reward-reconciliation.ts @@ -0,0 +1,201 @@ +import { AllocationKind, ALLOCATION_KINDS } from './reward-allocation-kind.enum'; + +/** + * Pure, deterministic reward-allocation reconciliation arithmetic. + * + * Every function here is total, side-effect free, and operates on `bigint` or + * decimal strings. There is deliberately **no** code in this file that reads a + * clock, touches a database, or reaches for the network — which is what makes + * the reconciliation report reproducible: same projection rows in, same report + * out, byte for byte. + * + * ## Why `bigint` and not a decimal library + * + * The chain's amounts are 256-bit integers. `number` loses precision above + * 2^53; `parseFloat` loses it far earlier. The repository's V2 read models + * already store these values as decimal strings in `varchar(100)` (see + * `ProjectParticipantPosition.stake`, `CanonicalEvent.amount`), so this module + * parses to `bigint`, does exact integer arithmetic, and serialises back with + * `toString()`. No float ever touches a token amount. + * + * ## What "divergent" means here + * + * Divergence is *reported*, never *repaired*. A pool whose allocations do not + * sum to the emitted pool amount, or an allocation whose claims exceed its + * allocation, is a fact about the chain that this service does not get to + * overrule. The projector is careful never to create that state in the first + * place (it rejects the offending event and records an anomaly); these + * functions exist so a divergence that arrives some other way is still visible + * rather than invisible. + */ + +/** Throw on anything that is not a base-10 integer string. */ +function toBigInt(value: string, field: string): bigint { + const trimmed = value.trim(); + if (!/^-?\d+$/.test(trimmed)) { + throw new Error( + `${field} must be a base-10 integer string, received ${JSON.stringify(value)}`, + ); + } + return BigInt(trimmed); +} + +export type AllocationStatus = 'allocated' | 'partially_claimed' | 'claimed'; + +export interface AllocationAmounts { + allocationId: string; + kind: AllocationKind; + beneficiary: string | null; + /** Verbatim amount from the allocating event. */ + allocatedAmount: string; + /** Running sum of amounts from attributed claim events. */ + claimedAmount: string; +} + +export interface AllocationReconciliation { + allocationId: string; + kind: AllocationKind; + beneficiary: string | null; + /** The contract-emitted allocation. */ + allocated: string; + /** The contract-emitted claims so far. */ + claimed: string; + /** + * `allocated - claimed`. Reported raw and **not** clamped to zero: a negative + * value is the signal that must stay visible. See {@link overClaimed}. + */ + claimableRemaining: string; + /** True when `claimed > allocated` — an invariant violation. */ + overClaimed: boolean; + status: AllocationStatus | 'over_claimed'; + /** True when anything at all is off. */ + divergent: boolean; +} + +/** + * Derive the read status from the two amounts. Pure: no event history, no + * contract semantics, just "how much of it has been claimed". + */ +export function classifyAllocationStatus( + allocated: bigint, + claimed: bigint, +): AllocationStatus | 'over_claimed' { + if (claimed < 0n) return 'over_claimed'; + if (claimed > allocated) return 'over_claimed'; + if (claimed === 0n) return 'allocated'; + if (claimed === allocated) return 'claimed'; + return 'partially_claimed'; +} + +/** + * Reconcile one allocation's allocated vs claimed amounts. + * + * @throws if either amount is not a base-10 integer string. A malformed + * amount is a data defect that must surface, not coerce to `0`. + */ +export function reconcileAllocation( + row: AllocationAmounts, +): AllocationReconciliation { + const allocated = toBigInt(row.allocatedAmount, 'allocatedAmount'); + const claimed = toBigInt(row.claimedAmount, 'claimedAmount'); + const overClaimed = claimed > allocated || claimed < 0n; + + return { + allocationId: row.allocationId, + kind: row.kind, + beneficiary: row.beneficiary, + allocated: allocated.toString(), + claimed: claimed.toString(), + // Left un-clamped on purpose: a negative remainder is the whole point. + claimableRemaining: (allocated - claimed).toString(), + overClaimed, + status: classifyAllocationStatus(allocated, claimed), + divergent: overClaimed, + }; +} + +export interface PoolReconciliationInput { + poolId: string; + claimId: string; + asset: string; + /** Verbatim settled amount emitted by the contract. */ + poolAmount: string; + /** Every projected allocation that draws from this pool. */ + allocations: ReadonlyArray< + Pick + >; +} + +export interface PoolReconciliation { + poolId: string; + claimId: string; + asset: string; + /** The contract-emitted pool total. */ + poolAmount: string; + /** Sum of the allocations projected against this pool. */ + allocatedTotal: string; + /** `poolAmount - allocatedTotal`. Un-clamped and signed. */ + divergence: string; + /** True when the projected allocations do not sum to the emitted total. */ + divergent: boolean; + /** Per-kind subtotals, keyed by {@link AllocationKind}. */ + byKind: Record; + allocationCount: number; +} + +/** + * Compare a source pool's emitted total against the allocations projected from + * it, and break the allocations down by beneficiary class. + * + * This is a *check*, never a *correction*. A `divergent: true` result means the + * database disagrees with the chain, and the correct response is to investigate + * the indexing path — not to adjust either side to make the numbers line up. + */ +export function reconcilePool( + input: PoolReconciliationInput, +): PoolReconciliation { + const poolAmount = toBigInt(input.poolAmount, 'poolAmount'); + + const byKindTotals = new Map(); + for (const kind of ALLOCATION_KINDS) { + byKindTotals.set(kind, 0n); + } + + let allocatedTotal = 0n; + for (const allocation of input.allocations) { + const amount = toBigInt(allocation.allocatedAmount, 'allocatedAmount'); + allocatedTotal += amount; + byKindTotals.set( + allocation.kind, + (byKindTotals.get(allocation.kind) ?? 0n) + amount, + ); + } + + const byKind = {} as Record; + for (const kind of ALLOCATION_KINDS) { + byKind[kind] = (byKindTotals.get(kind) ?? 0n).toString(); + } + + return { + poolId: input.poolId, + claimId: input.claimId, + asset: input.asset, + poolAmount: poolAmount.toString(), + allocatedTotal: allocatedTotal.toString(), + divergence: (poolAmount - allocatedTotal).toString(), + divergent: poolAmount !== allocatedTotal, + byKind, + allocationCount: input.allocations.length, + }; +} + +/** + * Sum a set of decimal strings exactly. Used for the report's grand totals. + */ +export function sumAmounts(values: ReadonlyArray): string { + let total = 0n; + for (const value of values) { + total += toBigInt(value, 'amount'); + } + return total.toString(); +} diff --git a/src/v2/rewards/rewards-projector.service.integration.spec.ts b/src/v2/rewards/rewards-projector.service.integration.spec.ts new file mode 100644 index 00000000..ac00cf20 --- /dev/null +++ b/src/v2/rewards/rewards-projector.service.integration.spec.ts @@ -0,0 +1,521 @@ +import { Test, TestingModule } from '@nestjs/testing'; +import { TypeOrmModule } from '@nestjs/typeorm'; +import { DataSource } from 'typeorm'; +import { RewardsProjectorService } from './rewards-projector.service'; +import { RewardsReconciliationService } from './rewards-reconciliation.service'; +import { ProjectRewardAllocation } from './entities/project-reward-allocation.entity'; +import { ProjectRewardPool } from './entities/project-reward-pool.entity'; +import { ProjectRewardClaim } from './entities/project-reward-claim.entity'; +import { ProjectorCursor } from '../common/entities/projector-cursor.entity'; +import { + IndexingAnomaly, + IndexingAnomalyKind, +} from '../common/entities/indexing-anomaly.entity'; +import { CanonicalEvent } from '../events/entities/canonical-event.entity'; +import { CanonicalEventQueryService } from '../events/canonical-event-query.service'; +import { AllocationKind } from './reward-allocation-kind.enum'; + +const CLAIM_ID = '0x' + '11'.repeat(32); +const ROUND_ID = '0x' + 'aa'.repeat(32); +const ASSET = '0x' + '33'.repeat(20); +const SUBMITTER = '0x' + '22'.repeat(20); +const VERIFIER = '0x' + '44'.repeat(20); + +describe('RewardsProjectorService (integration)', () => { + let moduleRef: TestingModule; + let projector: RewardsProjectorService; + let reconciliation: RewardsReconciliationService; + let dataSource: DataSource; + + async function seedEvent(overrides: Partial): Promise { + await dataSource.getRepository(CanonicalEvent).insert({ + chainId: 10, + contractAddress: '0x' + 'aa'.repeat(20), + artifactVersion: 'v1', + txHash: '0x' + '00'.repeat(32), + logIndex: 0, + blockNumber: '1', + claimId: CLAIM_ID, + roundId: ROUND_ID, + asset: ASSET, + payload: {} as object, + rawArgs: {} as object, + ...overrides, + }); + } + + function seedPoolSettled( + poolId: string, + poolAmount: string, + blockNumber: string, + tx: string, + logIndex = 0, + ): Promise { + return seedEvent({ + eventName: 'RewardPoolSettled', + txHash: tx, + logIndex, + blockNumber, + payload: { poolId, amount: poolAmount }, + }); + } + + function seedAllocated(opts: { + tx: string; + blockNumber: string; + kind: string; + beneficiary: string | null; + amount: string; + sourcePoolId: string; + logIndex?: number; + extra?: Record; + }): Promise { + return seedEvent({ + eventName: 'RewardAllocated', + txHash: opts.tx, + logIndex: opts.logIndex ?? 0, + blockNumber: opts.blockNumber, + actor: opts.beneficiary, + amount: opts.amount, + payload: { + kind: opts.kind, + beneficiary: opts.beneficiary, + sourcePoolId: opts.sourcePoolId, + amount: opts.amount, + ...(opts.extra ?? {}), + }, + }); + } + + beforeEach(async () => { + moduleRef = await Test.createTestingModule({ + imports: [ + TypeOrmModule.forRoot({ + type: 'sqlite', + database: ':memory:', + // eslint-disable-next-line @typescript-eslint/no-require-imports + driver: require('sqlite3'), + entities: [ + CanonicalEvent, + ProjectRewardAllocation, + ProjectRewardPool, + ProjectRewardClaim, + ProjectorCursor, + IndexingAnomaly, + ], + synchronize: true, + }), + TypeOrmModule.forFeature([ + ProjectRewardAllocation, + ProjectRewardPool, + ProjectRewardClaim, + ProjectorCursor, + IndexingAnomaly, + CanonicalEvent, + ]), + ], + providers: [ + RewardsProjectorService, + RewardsReconciliationService, + CanonicalEventQueryService, + ], + }).compile(); + + projector = moduleRef.get(RewardsProjectorService); + reconciliation = moduleRef.get(RewardsReconciliationService); + dataSource = moduleRef.get(DataSource); + }); + + afterEach(async () => { + await moduleRef.close(); + }); + + describe('allocation projection', () => { + it('projects a RewardAllocated event verbatim, with claimable but not yet claimed', async () => { + await seedAllocated({ + tx: '0x' + '01'.repeat(32), + blockNumber: '100', + kind: 'submitter', + beneficiary: SUBMITTER, + amount: '6000', + sourcePoolId: 'pool-1', + }); + + const summary = await projector.processNewEvents(); + expect(summary.processed).toBe(1); + expect(summary.applied).toBe(1); + expect(summary.anomalies).toBe(0); + + const rows = await dataSource + .getRepository(ProjectRewardAllocation) + .find(); + expect(rows).toHaveLength(1); + expect(rows[0].kind).toBe(AllocationKind.SUBMITTER); + expect(rows[0].allocatedAmount).toBe('6000'); + expect(rows[0].claimedAmount).toBe('0'); + expect(rows[0].beneficiary).toBe(SUBMITTER); + expect(rows[0].claimId).toBe(CLAIM_ID); + expect(rows[0].sourcePoolId).toBe('pool-1'); + }); + + it.each([ + [AllocationKind.SUBMITTER, 1], + [AllocationKind.VERIFIER, 2], + [AllocationKind.CHALLENGER, 3], + [AllocationKind.TREASURY, 4], + [AllocationKind.REFUND, 5], + ])( + 'projects the %s allocation kind on equal footing with the others', + async (kind, index) => { + await seedAllocated({ + tx: '0x' + String(index).repeat(32), + blockNumber: '100', + kind, + beneficiary: kind === 'treasury' ? null : VERIFIER, + amount: '1000', + sourcePoolId: `pool-${kind}`, + }); + + await projector.processNewEvents(); + + const rows = await dataSource + .getRepository(ProjectRewardAllocation) + .find(); + expect(rows).toHaveLength(1); + expect(rows[0].kind).toBe(kind); + // A treasury allocation has no externally-owned beneficiary. It is + // recorded as null rather than being given a placeholder address. + expect(rows[0].beneficiary).toBe(kind === 'treasury' ? null : VERIFIER); + }, + ); + + it('rejects an allocation whose kind is not one of the five, recording an anomaly', async () => { + await seedAllocated({ + tx: '0x' + '01'.repeat(32), + blockNumber: '100', + kind: 'slashed', + beneficiary: SUBMITTER, + amount: '6000', + sourcePoolId: 'pool-1', + }); + + const summary = await projector.processNewEvents(); + expect(summary.anomalies).toBe(1); + + const rows = await dataSource + .getRepository(ProjectRewardAllocation) + .find(); + expect(rows).toHaveLength(0); + + const anomalies = await dataSource.getRepository(IndexingAnomaly).find(); + expect(anomalies[0].kind).toBe(IndexingAnomalyKind.OUT_OF_ORDER); + expect(anomalies[0].detail).toContain('slashed'); + }); + + it('rejects a non-integer amount rather than recording it as zero', async () => { + await seedAllocated({ + tx: '0x' + '01'.repeat(32), + blockNumber: '100', + kind: 'verifier', + beneficiary: VERIFIER, + amount: '1000.5', + sourcePoolId: 'pool-1', + }); + + const summary = await projector.processNewEvents(); + expect(summary.anomalies).toBe(1); + expect( + await dataSource.getRepository(ProjectRewardAllocation).find(), + ).toHaveLength(0); + }); + + it('is replay-safe: reprocessing does not duplicate allocations', async () => { + await seedAllocated({ + tx: '0x' + '01'.repeat(32), + blockNumber: '100', + kind: 'verifier', + beneficiary: VERIFIER, + amount: '2500', + sourcePoolId: 'pool-1', + }); + + await projector.processNewEvents(); + const secondRun = await projector.processNewEvents(); + expect(secondRun.processed).toBe(0); + expect( + await dataSource.getRepository(ProjectRewardAllocation).find(), + ).toHaveLength(1); + }); + }); + + describe('claim tracking', () => { + async function seedOneAllocation(): Promise { + await seedAllocated({ + tx: '0x' + '01'.repeat(32), + blockNumber: '100', + kind: 'verifier', + beneficiary: VERIFIER, + amount: '2500', + sourcePoolId: 'pool-1', + extra: { allocationId: 'alloc-1' }, + }); + } + + it('accumulates an emitted RewardClaimed against the allocation', async () => { + await seedOneAllocation(); + await seedEvent({ + eventName: 'RewardClaimed', + txHash: '0x' + '02'.repeat(32), + blockNumber: '200', + actor: VERIFIER, + amount: '1000', + payload: { allocationId: 'alloc-1', amount: '1000' }, + }); + + await projector.processNewEvents(); + + const rows = await dataSource + .getRepository(ProjectRewardAllocation) + .find(); + expect(rows[0].claimedAmount).toBe('1000'); + expect(rows[0].lastClaimBlockNumber).toBe('200'); + expect(rows[0].lastClaimEvent).toBe(`0x${'02'.repeat(32)}:0`); + }); + + it('sums successive partial claims exactly', async () => { + await seedOneAllocation(); + for (const [index, amount] of ['1', '2', '3'].entries()) { + await seedEvent({ + eventName: 'RewardClaimed', + txHash: '0x' + '0' + index + '2'.repeat(31), + blockNumber: String(200 + index), + actor: VERIFIER, + amount, + payload: { allocationId: 'alloc-1', amount }, + }); + } + + await projector.processNewEvents(); + + const rows = await dataSource + .getRepository(ProjectRewardAllocation) + .find(); + expect(rows[0].claimedAmount).toBe('6'); + }); + + it('REFUSES a claim that would exceed the allocation, and records an anomaly', async () => { + await seedOneAllocation(); + await seedEvent({ + eventName: 'RewardClaimed', + txHash: '0x' + '02'.repeat(32), + blockNumber: '200', + actor: VERIFIER, + amount: '2501', + payload: { allocationId: 'alloc-1', amount: '2501' }, + }); + + const summary = await projector.processNewEvents(); + expect(summary.anomalies).toBe(1); + + const rows = await dataSource + .getRepository(ProjectRewardAllocation) + .find(); + // The read model must never assert that more was claimed than the + // contract allocated. The divergence stays observable in the anomaly log. + expect(rows[0].claimedAmount).toBe('0'); + + const anomalies = await dataSource.getRepository(IndexingAnomaly).find(); + expect(anomalies[0].kind).toBe(IndexingAnomalyKind.INVALID_TRANSITION); + expect(anomalies[0].detail).toContain('2501'); + }); + + it('refuses a second claim that would cross the allocation boundary', async () => { + await seedOneAllocation(); + await seedEvent({ + eventName: 'RewardClaimed', + txHash: '0x' + '02'.repeat(32), + blockNumber: '200', + actor: VERIFIER, + amount: '2000', + payload: { allocationId: 'alloc-1', amount: '2000' }, + }); + await seedEvent({ + eventName: 'RewardClaimed', + txHash: '0x' + '03'.repeat(32), + blockNumber: '300', + actor: VERIFIER, + amount: '600', + payload: { allocationId: 'alloc-1', amount: '600' }, + }); + + const summary = await projector.processNewEvents(); + expect(summary.applied).toBe(1); + expect(summary.anomalies).toBe(1); + + const rows = await dataSource + .getRepository(ProjectRewardAllocation) + .find(); + expect(rows[0].claimedAmount).toBe('2000'); + }); + + it('does not guess a beneficiary when a claim cannot be attributed', async () => { + await seedOneAllocation(); + await seedEvent({ + eventName: 'RewardClaimed', + txHash: '0x' + '02'.repeat(32), + blockNumber: '200', + actor: '0x' + '99'.repeat(20), + amount: '500', + payload: { amount: '500' }, + }); + + const summary = await projector.processNewEvents(); + expect(summary.anomalies).toBe(1); + + const rows = await dataSource + .getRepository(ProjectRewardAllocation) + .find(); + expect(rows[0].claimedAmount).toBe('0'); + + const anomalies = await dataSource.getRepository(IndexingAnomaly).find(); + expect(anomalies[0].kind).toBe(IndexingAnomalyKind.OUT_OF_ORDER); + expect(anomalies[0].detail).toContain('could not be attributed'); + }); + + it('accepts a full claim that exactly exhausts the allocation', async () => { + await seedOneAllocation(); + await seedEvent({ + eventName: 'RewardClaimed', + txHash: '0x' + '02'.repeat(32), + blockNumber: '200', + actor: VERIFIER, + amount: '2500', + payload: { allocationId: 'alloc-1', amount: '2500' }, + }); + + const summary = await projector.processNewEvents(); + expect(summary.applied).toBe(2); + expect(summary.anomalies).toBe(0); + + const report = await reconciliation.reconcileChain(10); + expect(report.allocations[0].status).toBe('claimed'); + expect(report.allocations[0].claimableRemaining).toBe('0'); + }); + }); + + describe('pool reconciliation', () => { + it('reports a balanced pool when the allocations sum to the emitted total', async () => { + await seedPoolSettled('pool-1', '10000', '50', '0x' + '05'.repeat(32)); + await seedAllocated({ + tx: '0x' + '01'.repeat(32), + blockNumber: '100', + kind: 'submitter', + beneficiary: SUBMITTER, + amount: '6000', + sourcePoolId: 'pool-1', + }); + await seedAllocated({ + tx: '0x' + '02'.repeat(32), + blockNumber: '110', + kind: 'verifier', + beneficiary: VERIFIER, + amount: '2500', + sourcePoolId: 'pool-1', + }); + await seedAllocated({ + tx: '0x' + '03'.repeat(32), + blockNumber: '120', + kind: 'treasury', + beneficiary: null, + amount: '1500', + sourcePoolId: 'pool-1', + }); + + await projector.processNewEvents(); + + const report = await reconciliation.reconcileChain(10); + expect(report.pools).toHaveLength(1); + expect(report.pools[0].poolAmount).toBe('10000'); + expect(report.pools[0].allocatedTotal).toBe('10000'); + expect(report.pools[0].divergent).toBe(false); + expect(report.summary.divergentPoolCount).toBe(0); + expect(report.summary.totalAllocated).toBe('10000'); + expect(report.summary.totalClaimed).toBe('0'); + expect(report.summary.allocatedByKind[AllocationKind.TREASURY]).toBe( + '1500', + ); + }); + + it('reports divergence when the projected allocations under-account for the pool', async () => { + await seedPoolSettled('pool-1', '10000', '50', '0x' + '05'.repeat(32)); + await seedAllocated({ + tx: '0x' + '01'.repeat(32), + blockNumber: '100', + kind: 'submitter', + beneficiary: SUBMITTER, + amount: '6000', + sourcePoolId: 'pool-1', + }); + + await projector.processNewEvents(); + + const report = await reconciliation.reconcileChain(10); + expect(report.pools[0].divergent).toBe(true); + expect(report.pools[0].divergence).toBe('4000'); + expect(report.summary.divergentPoolCount).toBe(1); + }); + + it('treats a settled pool with no projected allocations as divergent, not balanced', async () => { + await seedPoolSettled('pool-1', '10000', '50', '0x' + '05'.repeat(32)); + + await projector.processNewEvents(); + + const report = await reconciliation.reconcileChain(10); + expect(report.pools[0].divergent).toBe(true); + expect(report.pools[0].allocationCount).toBe(0); + }); + + it('rejects a second settlement of the same pool, keeping the first authoritative', async () => { + await seedPoolSettled('pool-1', '10000', '50', '0x' + '05'.repeat(32)); + await seedPoolSettled('pool-1', '99999', '400', '0x' + '06'.repeat(32)); + + const summary = await projector.processNewEvents(); + expect(summary.applied).toBe(1); + expect(summary.anomalies).toBe(1); + + const pools = await dataSource.getRepository(ProjectRewardPool).find(); + expect(pools).toHaveLength(1); + expect(pools[0].poolAmount).toBe('10000'); + + const anomalies = await dataSource.getRepository(IndexingAnomaly).find(); + expect(anomalies[0].kind).toBe(IndexingAnomalyKind.DUPLICATE_EVENT); + }); + + it('lists allocations for a claim and by kind', async () => { + await seedAllocated({ + tx: '0x' + '01'.repeat(32), + blockNumber: '100', + kind: 'verifier', + beneficiary: VERIFIER, + amount: '2500', + sourcePoolId: 'pool-1', + }); + await seedAllocated({ + tx: '0x' + '02'.repeat(32), + blockNumber: '110', + kind: 'verifier', + beneficiary: VERIFIER, + amount: '1500', + sourcePoolId: 'pool-1', + }); + + await projector.processNewEvents(); + + const forClaim = await reconciliation.listForClaim(10, CLAIM_ID); + expect(forClaim).toHaveLength(2); + + const verifiers = await reconciliation.listByKind(10, AllocationKind.VERIFIER); + expect(verifiers).toHaveLength(2); + }); + }); +}); diff --git a/src/v2/rewards/rewards-projector.service.ts b/src/v2/rewards/rewards-projector.service.ts new file mode 100644 index 00000000..427f9ce8 --- /dev/null +++ b/src/v2/rewards/rewards-projector.service.ts @@ -0,0 +1,483 @@ +import { Injectable, Logger } from '@nestjs/common'; +import { InjectDataSource } from '@nestjs/typeorm'; +import { DataSource } from 'typeorm'; +import { CanonicalEventQueryService } from '../events/canonical-event-query.service'; +import { CanonicalEvent } from '../events/entities/canonical-event.entity'; +import { ProjectRewardAllocation } from './entities/project-reward-allocation.entity'; +import { ProjectRewardPool } from './entities/project-reward-pool.entity'; +import { ProjectRewardClaim } from './entities/project-reward-claim.entity'; +import { parseAllocationKind } from './reward-allocation-kind.enum'; +import { ProjectorCursor } from '../common/entities/projector-cursor.entity'; +import { + IndexingAnomaly, + IndexingAnomalyKind, +} from '../common/entities/indexing-anomaly.entity'; + +const PROJECTOR_NAME = 'v2-rewards'; +const PG_UNIQUE_VIOLATION = '23505'; +/** Canonical event names this projector consumes. Exported for the rebuild + * pipeline's projection registry (`src/v2/rebuild/`), which needs to know the + * full event-name set to assert that a rebuild consumed everything it should. */ +export const REWARD_ALLOCATION_EVENT_NAMES = [ + 'RewardPoolSettled', + 'RewardAllocated', + 'RewardClaimed', +] as const; + +export interface ProjectorRunSummary { + processed: number; + applied: number; + anomalies: number; +} + +function readString( + payload: Record, + key: string, +): string | null { + const value = payload[key]; + return typeof value === 'string' || typeof value === 'number' + ? String(value) + : null; +} + +/** + * Parse a canonical event amount into a `bigint`. + * + * Returns `null` for anything that is not a base-10 integer string. Callers + * must treat `null` as "reject the event and record an anomaly". There is no + * `parseFloat` fallback anywhere in this file: a 256-bit token amount must + * never pass through a double, and a malformed amount must never silently + * become `0` (which would read as "this beneficiary was allocated nothing" + * and is a materially different claim from "we could not read this event"). + */ +function readAmount( + payload: Record, + key: string, +): bigint | null { + const raw = readString(payload, key); + if (raw === null) return null; + const trimmed = raw.trim(); + if (!/^\d+$/.test(trimmed)) return null; + return BigInt(trimmed); +} + +function lower(value: string | null): string | null { + return value === null ? null : value.toLowerCase(); +} + +/** + * Projects reward allocations and claim progress from canonical events + * (V2-BE-017). + * + * ## The rule this projector exists to obey + * + * The deployed Optimism/EVM contract is the only authority for who was + * allocated what and who has collected it. This projector: + * + * - **reads** an allocation amount from a `RewardAllocated` event and stores it + * verbatim; + * - **tracks** claim progress by summing amounts from `RewardClaimed` events + * the contract has already emitted; + * - **records** a source pool's emitted total from `RewardPoolSettled` so the + * two can be reconciled; + * - and does **nothing else**. It never apportions a pool, never computes a + * share, never decides a winner, and never fills a gap from a neighbouring + * event's data. + * + * Every place where it could be tempted to guess instead, it fails closed and + * writes an `IndexingAnomaly`: + * + * | Situation | Outcome | + * | --------- | ------- | + * | `RewardAllocated` with no/unknown `kind` | anomaly, no row | + * | `RewardAllocated` with a non-integer or absent `amount` | anomaly, no row | + * | `RewardClaimed` that cannot be attributed to a known allocation | `out_of_order` anomaly, nothing credited | + * | `RewardClaimed` that would push `claimed` above `allocated` | `invalid_transition` anomaly, **rejected** — `claimed` never exceeds `allocated` in this table | + * | A second `RewardPoolSettled` for the same pool | `duplicate_event` anomaly, first one wins | + * | Replay of an already-applied `RewardAllocated` | no-op (`duplicate`) | + * + * That last rejection is the load-bearing one. By refusing to record an + * over-claim, the read model cannot present an amount as "claimed" that the + * chain did not emit; the divergence stays in the anomaly log where an operator + * will see it. + * + * ## ASSUMPTION FLAGGED FOR REVIEW + * + * V2-BE-008 (approved artifact import) has not landed, so there is no frozen + * ABI to read real argument names from. The payload keys read here — `kind`, + * `beneficiary`, `sourcePoolId`, `allocationId`, `poolId` — follow the + * vocabulary of the V2-BE-017 issue text and the convention documented in + * `../events/event-schema-registry.ts`. They are expected to be reconciled + * against the real approved ABI once V2-BE-008 exists. Nothing in this file + * changes protocol meaning: it only says where to look for each field. + */ +@Injectable() +export class RewardsProjectorService { + private readonly logger = new Logger(RewardsProjectorService.name); + + constructor( + @InjectDataSource() private readonly dataSource: DataSource, + private readonly canonicalEvents: CanonicalEventQueryService, + ) {} + + async processNewEvents(batchSize = 100): Promise { + const cursorRepo = this.dataSource.getRepository(ProjectorCursor); + const cursor = await cursorRepo.findOne({ + where: { projectorName: PROJECTOR_NAME }, + }); + const after = cursor + ? { blockNumber: cursor.lastBlockNumber, logIndex: cursor.lastLogIndex } + : null; + + const events = await this.canonicalEvents.findAfter( + [...REWARD_ALLOCATION_EVENT_NAMES], + after, + batchSize, + ); + const summary: ProjectorRunSummary = { + processed: 0, + applied: 0, + anomalies: 0, + }; + + for (const event of events) { + summary.processed += 1; + const outcome = await this.applyEvent(event); + if (outcome === 'applied') summary.applied += 1; + if (outcome === 'anomaly') summary.anomalies += 1; + + await cursorRepo.upsert( + { + projectorName: PROJECTOR_NAME, + lastBlockNumber: event.blockNumber, + lastLogIndex: event.logIndex, + }, + ['projectorName'], + ); + } + + return summary; + } + + private async applyEvent( + event: CanonicalEvent, + ): Promise<'applied' | 'anomaly' | 'duplicate'> { + if (event.eventName === 'RewardPoolSettled') { + return this.applyPoolSettled(event); + } + if (event.eventName === 'RewardAllocated') { + return this.applyAllocated(event); + } + if (event.eventName === 'RewardClaimed') { + return this.applyClaimed(event); + } + return 'duplicate'; + } + + // ─── RewardPoolSettled ────────────────────────────────────────────────── + + private async applyPoolSettled( + event: CanonicalEvent, + ): Promise<'applied' | 'anomaly' | 'duplicate'> { + const poolId = readString(event.payload, 'poolId'); + const poolAmount = readAmount(event.payload, 'amount'); + + if (!event.claimId || !event.asset || !poolId || poolAmount === null) { + await this.recordAnomaly( + IndexingAnomalyKind.OUT_OF_ORDER, + poolId ?? `${event.txHash}:${event.logIndex}`, + event, + 'RewardPoolSettled rejected: missing claimId/asset/poolId or a non-integer amount', + ); + return 'anomaly'; + } + + const poolRepo = this.dataSource.getRepository(ProjectRewardPool); + try { + await poolRepo.insert({ + poolId, + chainId: event.chainId, + claimId: event.claimId!, + asset: event.asset!.toLowerCase(), + poolAmount: poolAmount.toString(), + eventTxHash: event.txHash, + eventLogIndex: event.logIndex, + blockNumber: event.blockNumber, + }); + return 'applied'; + } catch (err) { + if (!this.isUniqueViolation(err)) throw err; + + // Either a safe replay of the same event, or a genuinely second settle + // for a pool that already settled. Both are non-fatal; the first is a + // replay, the second is a protocol-level fact worth surfacing. + const existing = await poolRepo.findOne({ + where: { + eventTxHash: event.txHash, + eventLogIndex: event.logIndex, + }, + }); + if (existing) return 'duplicate'; + + await this.recordAnomaly( + IndexingAnomalyKind.DUPLICATE_EVENT, + poolId, + event, + `Pool ${poolId} was already settled; the first settlement is authoritative`, + ); + return 'anomaly'; + } + } + + // ─── RewardAllocated ──────────────────────────────────────────────────── + + private async applyAllocated( + event: CanonicalEvent, + ): Promise<'applied' | 'anomaly' | 'duplicate'> { + const kind = parseAllocationKind(readString(event.payload, 'kind')); + const amount = readAmount(event.payload, 'amount'); + const beneficiary = lower( + readString(event.payload, 'beneficiary') ?? event.actor, + ); + const sourcePoolId = + readString(event.payload, 'sourcePoolId') ?? + (event.claimId ? `claim:${event.claimId}` : null); + + if ( + !event.claimId || + !event.asset || + kind === null || + amount === null || + sourcePoolId === null + ) { + await this.recordAnomaly( + IndexingAnomalyKind.OUT_OF_ORDER, + sourcePoolId ?? `${event.txHash}:${event.logIndex}`, + event, + 'RewardAllocated rejected: missing claimId/asset/sourcePoolId, an unrecognised ' + + `kind, or a non-integer amount (kind=${readString(event.payload, 'kind') ?? ''})`, + ); + return 'anomaly'; + } + + const allocationRepo = this.dataSource.getRepository(ProjectRewardAllocation); + try { + await allocationRepo.insert({ + allocationId: this.deriveAllocationId(event), + chainId: event.chainId, + claimId: event.claimId!, + roundId: event.roundId, + sourcePoolId, + kind, + beneficiary, + asset: event.asset!.toLowerCase(), + allocatedAmount: amount.toString(), + claimedAmount: '0', + lastClaimBlockNumber: null, + lastClaimEvent: null, + eventTxHash: event.txHash, + eventLogIndex: event.logIndex, + blockNumber: event.blockNumber, + }); + return 'applied'; + } catch (err) { + if (!this.isUniqueViolation(err)) throw err; + + const existing = await allocationRepo.findOne({ + where: { eventTxHash: event.txHash, eventLogIndex: event.logIndex }, + }); + if (existing) return 'duplicate'; // safe replay + + // The primary key is the allocation id, so this branch is only reachable + // when the contract supplied an id that a *different* event already used. + await this.recordAnomaly( + IndexingAnomalyKind.DUPLICATE_EVENT, + this.deriveAllocationId(event), + event, + `Allocation id ${this.deriveAllocationId(event)} was already claimed by a ` + + 'different event; the first allocation is authoritative', + ); + return 'anomaly'; + } + } + + // ─── RewardClaimed ────────────────────────────────────────────────────── + + private async applyClaimed( + event: CanonicalEvent, + ): Promise<'applied' | 'anomaly' | 'duplicate'> { + const amount = readAmount(event.payload, 'amount'); + if (!event.claimId || amount === null) { + await this.recordAnomaly( + IndexingAnomalyKind.OUT_OF_ORDER, + `${event.txHash}:${event.logIndex}`, + event, + 'RewardClaimed rejected: missing claimId or a non-integer amount', + ); + return 'anomaly'; + } + + const allocationRepo = this.dataSource.getRepository(ProjectRewardAllocation); + const allocation = await this.resolveAllocationTarget(event); + + if (!allocation) { + // We will not infer which allocation was claimed. The contract emitted a + // claim we cannot attribute, and guessing would mean the backend deciding + // whose balance moved. + // + // Note that no withdrawal row is written either, so if the allocating + // event is projected later — a resumed run, a backfill, a re-ingest — the + // claim can still be attributed on a later pass. Refusing to record it is + // what makes that possible; recording it as unattributed would not. + await this.recordAnomaly( + IndexingAnomalyKind.OUT_OF_ORDER, + readString(event.payload, 'allocationId') ?? + `${event.claimId}:${lower(event.actor) ?? ''}`, + event, + `RewardClaimed could not be attributed to a known allocation on claim ${event.claimId}`, + ); + return 'anomaly'; + } + + const allocated = BigInt(allocation.allocatedAmount); + const claimed = BigInt(allocation.claimedAmount); + const next = claimed + amount; + + if (next > allocated) { + // Fail closed. Recording this would make the read model assert that more + // was claimed than the contract ever allocated, which is exactly the + // kind of backend-authored protocol truth this service must not produce. + await this.recordAnomaly( + IndexingAnomalyKind.INVALID_TRANSITION, + allocation.allocationId, + event, + `RewardClaimed rejected: would take claimed to ${next.toString()} against ` + + `an allocation of ${allocated.toString()} for ${allocation.allocationId}`, + ); + return 'anomaly'; + } + + // The withdrawal row goes in *first*. It is the idempotency guard for the + // only counter in this projection: a replay of this exact event violates + // its unique constraint and returns before the total is touched. Writing + // the counter first would double-count on every replay. + const withdrawalRepo = this.dataSource.getRepository(ProjectRewardClaim); + try { + await withdrawalRepo.insert({ + withdrawalId: this.deriveWithdrawalId(event), + chainId: event.chainId, + allocationId: allocation.allocationId, + claimId: allocation.claimId, + beneficiary: allocation.beneficiary, + asset: allocation.asset, + amount: amount.toString(), + claimTxHash: event.txHash, + claimLogIndex: event.logIndex, + blockNumber: event.blockNumber, + }); + } catch (err) { + if (!this.isUniqueViolation(err)) throw err; + return 'duplicate'; // safe replay of a withdrawal already recorded + } + + allocation.claimedAmount = next.toString(); + allocation.lastClaimBlockNumber = event.blockNumber; + allocation.lastClaimEvent = `${event.txHash}:${event.logIndex}`; + await allocationRepo.save(allocation); + return 'applied'; + } + + /** + * Resolve which allocation a `RewardClaimed` belongs to. + * + * Resolution is by protocol-supplied `allocationId` when present — that is + * exact. Otherwise it falls back to the `(claim, kind, beneficiary)` triple + * the event carries, taking the *earliest* matching allocation. Earliest is + * chosen because it is a pure function of already-projected state, so a + * replay resolves to the same row it resolved to the first time; a "most + * recent" or "largest" rule would make the result depend on projection order. + * + * There is no third fallback. If the event carries neither an allocation id + * nor enough of a triple to match, this returns `null` and the caller records + * an anomaly rather than picking the most plausible row. + */ + private async resolveAllocationTarget( + event: CanonicalEvent, + ): Promise { + const repo = this.dataSource.getRepository(ProjectRewardAllocation); + const allocationId = readString(event.payload, 'allocationId'); + if (allocationId) { + return repo.findOne({ where: { allocationId } }); + } + + const kind = parseAllocationKind(readString(event.payload, 'kind')); + const beneficiary = lower( + readString(event.payload, 'beneficiary') ?? event.actor, + ); + if (!kind || !beneficiary) return null; + + const candidates = await repo.find({ + where: { + chainId: event.chainId, + claimId: event.claimId!, + kind, + beneficiary, + }, + order: { blockNumber: 'ASC', eventLogIndex: 'ASC' }, + }); + return candidates[0] ?? null; + } + + // ─── Helpers ──────────────────────────────────────────────────────────── + + /** + * Deterministic allocation id: the protocol's own when it supplies one, + * otherwise a pure function of the creating event's chain identity. Either + * way, a replay of the same event reconstructs the same key — which is what + * makes the `allocationId` primary key a safe idempotency anchor. + */ + private deriveAllocationId(event: CanonicalEvent): string { + const supplied = readString(event.payload, 'allocationId'); + if (supplied) return supplied; + return `${event.chainId}:${event.txHash}:${event.logIndex}`; + } + + /** + * Deterministic withdrawal identity for a `RewardClaimed` event. Same + * derivation rule as {@link deriveAllocationId} and for the same reason: a + * replay must reconstruct the identical key, because that key is what makes + * the insert idempotent. + */ + private deriveWithdrawalId(event: CanonicalEvent): string { + const supplied = readString(event.payload, 'withdrawalId'); + if (supplied) return supplied; + return `${event.chainId}:${event.txHash}:${event.logIndex}`; + } + + private async recordAnomaly( + kind: IndexingAnomalyKind, + aggregateId: string, + event: CanonicalEvent, + detail: string, + ): Promise { + this.logger.warn(`${kind}: ${detail}`); + try { + await this.dataSource.getRepository(IndexingAnomaly).insert({ + sourceModule: PROJECTOR_NAME, + kind, + aggregateId, + eventTxHash: event.txHash, + eventLogIndex: event.logIndex, + detail, + }); + } catch (err) { + if (!this.isUniqueViolation(err)) throw err; + } + } + + private isUniqueViolation(err: unknown): boolean { + if (typeof err !== 'object' || err === null) return false; + const code = (err as { code?: string }).code; + return code === PG_UNIQUE_VIOLATION || code === 'SQLITE_CONSTRAINT'; + } +} diff --git a/src/v2/rewards/rewards-reconciliation.service.ts b/src/v2/rewards/rewards-reconciliation.service.ts new file mode 100644 index 00000000..f30a8340 --- /dev/null +++ b/src/v2/rewards/rewards-reconciliation.service.ts @@ -0,0 +1,323 @@ +import { Injectable, NotFoundException } from '@nestjs/common'; +import { InjectRepository } from '@nestjs/typeorm'; +import { Repository } from 'typeorm'; +import { ProjectRewardAllocation } from './entities/project-reward-allocation.entity'; +import { ProjectRewardPool } from './entities/project-reward-pool.entity'; +import { ProjectRewardClaim } from './entities/project-reward-claim.entity'; +import { + AllocationReconciliation, + PoolReconciliation, + reconcileAllocation, + reconcilePool, + sumAmounts, +} from './reward-reconciliation'; +import { AllocationKind, ALLOCATION_KINDS } from './reward-allocation-kind.enum'; + +export interface AllocationWithdrawalReconciliation { + allocationId: string; + /** `claimedAmount` — the running total maintained by the projector. */ + claimedCounter: string; + /** Sum of the withdrawal rows for this allocation. */ + claimedFromWithdrawals: string; + /** `claimedCounter - claimedFromWithdrawals`. Un-clamped and signed. */ + divergence: string; + /** True when the counter and the emitted claim events disagree. */ + divergent: boolean; + withdrawalCount: number; +} + +export interface RewardReconciliationReport { + chainId: number; + pools: PoolReconciliation[]; + allocations: AllocationReconciliation[]; + /** Counter-vs-event check for every allocation, in allocation-id order. */ + withdrawals: AllocationWithdrawalReconciliation[]; + summary: { + poolCount: number; + allocationCount: number; + /** Pools whose projected allocations do not sum to the emitted total. */ + divergentPoolCount: number; + /** Allocations whose tracked claims exceed the emitted allocation. */ + overClaimedAllocationCount: number; + /** + * Allocations whose `claimedAmount` counter disagrees with the sum of the + * emitted `RewardClaimed` events attributed to them. Normally zero; a + * non-zero value means the projection's counter and its event log have + * diverged. + */ + divergentWithdrawalCount: number; + /** Allocations with no attributed claims yet. */ + unclaimedCount: number; + /** Grand total of every projected allocation. Decimal string. */ + totalAllocated: string; + /** Grand total of every tracked claim. Decimal string. */ + totalClaimed: string; + /** Per-kind grand totals across every pool. */ + allocatedByKind: Record; + }; +} + +/** + * Read-side reconciliation of projected reward allocations against what the + * contract actually emitted (V2-BE-017). + * + * ## What this service does and does not do + * + * It **reports** two independent facts and refuses to reconcile them by + * fiat: + * + * 1. **Allocations vs source pools.** For each `RewardPoolSettled` pool, the + * sum of allocations projected from `RewardAllocated` events is compared + * against the pool amount the contract emitted. A difference is a + * `divergent` result. + * 2. **Claimed vs claimable.** For each allocation, the tracked claims are + * compared against the allocated amount. + * + * When either diverges, the correct action is to investigate the *indexing* + * path — a missed event, a quarantined log, an unmapped event name. It is + * never correct to "fix" the numbers here. This service exposes no write + * path at all: it has no method that mutates an allocation, a pool, or a + * status. It is a read model about the read model. + * + * ## Why the projector never lets a divergence in + * + * The projector rejects a `RewardClaimed` that would exceed its allocation + * (see `rewards-projector.service.ts`). So an `overClaimed` row here means the + * state arrived by some path other than that projector — exactly the situation + * a reconciliation pass exists to make loud rather than invisible. + * + * ## Determinism + * + * The report is a pure function of the projected rows: same rows in, same + * report out, no clock, no randomness, no network. Rows are read in a fixed + * order and every sum is exact `bigint` arithmetic (see + * `reward-reconciliation.ts`). + */ +@Injectable() +export class RewardsReconciliationService { + constructor( + @InjectRepository(ProjectRewardAllocation) + private readonly allocationRepo: Repository, + @InjectRepository(ProjectRewardPool) + private readonly poolRepo: Repository, + @InjectRepository(ProjectRewardClaim) + private readonly claimRepo: Repository, + ) {} + + /** + * Reconcile every pool and allocation for a chain. + */ + async reconcileChain(chainId: number): Promise { + const pools = await this.poolRepo.find({ + where: { chainId }, + order: { poolId: 'ASC' }, + }); + const allocations = await this.allocationRepo.find({ + where: { chainId }, + order: { allocationId: 'ASC' }, + }); + // Held in a local, not an instance field: this service is a singleton, so + // per-call scratch state on `this` would be a data race the moment two + // reconciliations overlap. + const withdrawalsByAllocation = await this.sumWithdrawalsByAllocation( + allocations.map((allocation) => allocation.allocationId), + ); + const byPool = this.groupAllocationsByPool(allocations); + const poolReconciliations = pools.map((pool) => + reconcilePool({ + poolId: pool.poolId, + claimId: pool.claimId, + asset: pool.asset, + poolAmount: pool.poolAmount, + allocations: byPool.get(pool.poolId) ?? [], + }), + ); + + const allocationReconciliations = allocations.map((allocation) => + reconcileAllocation({ + allocationId: allocation.allocationId, + kind: allocation.kind, + beneficiary: allocation.beneficiary, + allocatedAmount: allocation.allocatedAmount, + claimedAmount: allocation.claimedAmount, + }), + ); + + // The third check: the counter vs the emitted claim events. `claimedAmount` + // is a running total and therefore the one value here that a bug could + // silently corrupt; the withdrawal rows are the independent record it must + // agree with. + const withdrawals = allocations.map((allocation) => { + const rows = withdrawalsByAllocation.get(allocation.allocationId) ?? { + total: 0n, + count: 0, + }; + const divergence = BigInt(allocation.claimedAmount) - rows.total; + return { + allocationId: allocation.allocationId, + claimedCounter: allocation.claimedAmount, + claimedFromWithdrawals: rows.total.toString(), + divergence: divergence.toString(), + divergent: divergence !== 0n, + withdrawalCount: rows.count, + }; + }); + + return { + chainId, + pools: poolReconciliations, + allocations: allocationReconciliations, + withdrawals, + summary: this.summarise( + poolReconciliations, + allocationReconciliations, + withdrawals, + allocations, + ), + }; + } + + /** + * Reconcile a single pool. Throws rather than returning an empty report, so + * a caller asking about an unknown pool cannot mistake "no pool" for "a pool + * with zero allocations, therefore balanced". + */ + async reconcilePoolById( + chainId: number, + poolId: string, + ): Promise { + const pool = await this.poolRepo.findOne({ where: { chainId, poolId } }); + if (!pool) { + throw new NotFoundException( + `No reward pool ${poolId} projected on chain ${chainId}`, + ); + } + const allocations = await this.allocationRepo.find({ + where: { chainId, sourcePoolId: poolId }, + order: { allocationId: 'ASC' }, + }); + return reconcilePool({ + poolId: pool.poolId, + claimId: pool.claimId, + asset: pool.asset, + poolAmount: pool.poolAmount, + allocations, + }); + } + + /** Every allocation for a claim, keyed by beneficiary class. */ + async listForClaim( + chainId: number, + claimId: string, + ): Promise { + const allocations = await this.allocationRepo.find({ + where: { chainId, claimId }, + order: { allocationId: 'ASC' }, + }); + return allocations.map((allocation) => + reconcileAllocation({ + allocationId: allocation.allocationId, + kind: allocation.kind, + beneficiary: allocation.beneficiary, + allocatedAmount: allocation.allocatedAmount, + claimedAmount: allocation.claimedAmount, + }), + ); + } + + /** Every allocation for one beneficiary class across a chain. */ + async listByKind( + chainId: number, + kind: AllocationKind, + ): Promise { + return this.allocationRepo.find({ + where: { chainId, kind }, + order: { allocationId: 'ASC' }, + }); + } + + /** + * Exact `bigint` sum of the withdrawal rows per allocation. + * + * Read in a single ordered query rather than N per-allocation queries, so the + * report costs the same regardless of how many allocations exist. Amounts are + * summed as `bigint`; a malformed stored amount throws here rather than being + * silently dropped, because a silently dropped withdrawal would look exactly + * like a counter that drifted. + */ + private async sumWithdrawalsByAllocation( + allocationIds: string[], + ): Promise> { + const totals = new Map(); + if (allocationIds.length === 0) return totals; + + const rows = await this.claimRepo + .createQueryBuilder('w') + .where('w.allocationId IN (:...allocationIds)', { allocationIds }) + .orderBy('w.allocationId', 'ASC') + .addOrderBy('w.claimTxHash', 'ASC') + .addOrderBy('w.claimLogIndex', 'ASC') + .getMany(); + + for (const row of rows) { + const amount = BigInt(row.amount); + const existing = totals.get(row.allocationId); + if (existing) { + existing.total += amount; + existing.count += 1; + } else { + totals.set(row.allocationId, { total: amount, count: 1 }); + } + } + return totals; + } + + private groupAllocationsByPool( allocations: ProjectRewardAllocation[], + ): Map { + const grouped = new Map(); + for (const allocation of allocations) { + const bucket = grouped.get(allocation.sourcePoolId); + if (bucket) { + bucket.push(allocation); + } else { + grouped.set(allocation.sourcePoolId, [allocation]); + } + } + return grouped; + } + + private async summarise( + pools: PoolReconciliation[], + allocations: AllocationReconciliation[], + withdrawals: AllocationWithdrawalReconciliation[], + rows: ProjectRewardAllocation[], + ): RewardReconciliationReport['summary'] { + const kindTotals = new Map(); + for (const kind of ALLOCATION_KINDS) { + kindTotals.set(kind, 0n); + } + for (const row of rows) { + kindTotals.set( + row.kind, + (kindTotals.get(row.kind) ?? 0n) + BigInt(row.allocatedAmount), + ); + } + const allocatedByKind = {} as Record; + for (const kind of ALLOCATION_KINDS) { + allocatedByKind[kind] = (kindTotals.get(kind) ?? 0n).toString(); + } + + return { + poolCount: pools.length, + allocationCount: allocations.length, + divergentPoolCount: pools.filter((pool) => pool.divergent).length, + overClaimedAllocationCount: allocations.filter((a) => a.overClaimed) + .length, + divergentWithdrawalCount: withdrawals.filter((w) => w.divergent).length, + unclaimedCount: allocations.filter((a) => a.claimed === '0').length, + totalAllocated: sumAmounts(allocations.map((a) => a.allocated)), + totalClaimed: sumAmounts(allocations.map((a) => a.claimed)), + allocatedByKind, + }; + } +} diff --git a/src/v2/rewards/v2-rewards.module.ts b/src/v2/rewards/v2-rewards.module.ts new file mode 100644 index 00000000..2f988394 --- /dev/null +++ b/src/v2/rewards/v2-rewards.module.ts @@ -0,0 +1,35 @@ +import { Module } from '@nestjs/common'; +import { TypeOrmModule } from '@nestjs/typeorm'; +import { V2EventsModule } from '../events/v2-events.module'; +import { ProjectRewardAllocation } from './entities/project-reward-allocation.entity'; +import { ProjectRewardPool } from './entities/project-reward-pool.entity'; +import { ProjectRewardClaim } from './entities/project-reward-claim.entity'; +import { ProjectorCursor } from '../common/entities/projector-cursor.entity'; +import { IndexingAnomaly } from '../common/entities/indexing-anomaly.entity'; +import { RewardsProjectorService } from './rewards-projector.service'; +import { RewardsReconciliationService } from './rewards-reconciliation.service'; + +/** + * V2-BE-017 — reward allocation and claim projection. + * + * No controller is registered on purpose. This module is a projection plus a + * reconciliation read model, and exposing an endpoint that *creates* or + * *adjusts* an allocation would be precisely the backend-authored protocol + * truth this repository must never produce. A future read-only controller is a + * separate, additive change. + */ +@Module({ + imports: [ + TypeOrmModule.forFeature([ + ProjectRewardAllocation, + ProjectRewardPool, + ProjectRewardClaim, + ProjectorCursor, + IndexingAnomaly, + ]), + V2EventsModule, + ], + providers: [RewardsProjectorService, RewardsReconciliationService], + exports: [RewardsProjectorService, RewardsReconciliationService], +}) +export class V2RewardsModule {}