diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a741ff811..81dbe2e94 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -11,7 +11,7 @@ env: RUST_BACKTRACE: 1 RUST_LIB_BACKTRACE: 1 -# 7 parallel jobs. See doc/working/plan-test.md § "CI Job Grouping Guide" +# 10 parallel jobs. See doc/working/test.md § "Current CI Test Design" # for the assignment rule and how to add new test tasks. jobs: Lint: @@ -223,6 +223,70 @@ jobs: if: always() run: pixi run clean-env + IcebergE2E: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + with: + submodules: true + - name: Free disk space + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache + sudo apt-get clean + df -h / + - uses: prefix-dev/setup-pixi@v0.8.1 + with: + pixi-version: latest + - uses: Swatinem/rust-cache@v2 + - name: Run Iceberg acceptance + run: pixi run -e iceberg-e2e test-iceberg-e2e + - name: Upload Iceberg failure logs + if: failure() + uses: actions/upload-artifact@v4 + with: + name: runtime-iceberge2e-${{ github.run_attempt }} + path: | + .crowdb-runtime/ + target/iceberg-rck/open-api/build/reports/tests/ + target/iceberg-rck/open-api/build/test-results/ + if-no-files-found: ignore + retention-days: 7 + - name: Clean subprocesses + if: always() + run: pixi run clean-env + + IcebergSDK: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + with: + submodules: true + - name: Free disk space + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache + sudo apt-get clean + df -h / + - uses: prefix-dev/setup-pixi@v0.8.1 + with: + pixi-version: latest + - uses: Swatinem/rust-cache@v2 + - name: Run Iceberg acceptance + run: pixi run -e iceberg-e2e test-iceberg-sdk + - name: Upload Iceberg failure logs + if: failure() + uses: actions/upload-artifact@v4 + with: + name: runtime-icebergsdk-${{ github.run_attempt }} + path: | + .crowdb-runtime/ + target/iceberg-rck/open-api/build/reports/tests/ + target/iceberg-rck/open-api/build/test-results/ + if-no-files-found: ignore + retention-days: 7 + - name: Clean subprocesses + if: always() + run: pixi run clean-env + ConsoleTests: runs-on: ubuntu-latest steps: @@ -334,3 +398,39 @@ jobs: - name: Clean subprocesses if: always() run: pixi run clean-env + + DockerPreview: + runs-on: ubuntu-24.04 + permissions: + contents: read + steps: + - uses: actions/checkout@v4 + with: + submodules: true + - name: Free disk space + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache + sudo apt-get clean + df -h / + - uses: prefix-dev/setup-pixi@v0.8.1 + with: + pixi-version: latest + - name: Build and test single-node preview image + env: + CROWDB_PREVIEW_TEST_ARTIFACTS: ${{ runner.temp }}/crowdb-preview-artifacts + run: pixi run test-single-node-container + - name: Capture Docker diagnostics on failure + if: failure() + run: | + docker ps -a + docker images + docker info + df -h + - name: Upload preview failure logs + if: failure() + uses: actions/upload-artifact@v4 + with: + name: docker-preview-${{ github.run_attempt }} + path: ${{ runner.temp }}/crowdb-preview-artifacts + if-no-files-found: ignore + retention-days: 7 diff --git a/.github/workflows/release-container.yml b/.github/workflows/release-container.yml new file mode 100644 index 000000000..d047f2188 --- /dev/null +++ b/.github/workflows/release-container.yml @@ -0,0 +1,157 @@ +name: Publish crowdb-iceberge docker container + +on: + workflow_dispatch: + inputs: + tag: + description: Existing Git release tag to publish + required: true + type: string + +concurrency: + group: crowdb-iceberg-single-node-preview-release + cancel-in-progress: false + +jobs: + verify: + runs-on: ubuntu-24.04 + permissions: + contents: read + outputs: + version: ${{ steps.source.outputs.version }} + revision: ${{ steps.source.outputs.revision }} + steps: + - uses: actions/checkout@v4 + with: + ref: ${{ inputs.tag }} + fetch-depth: 0 + submodules: true + - uses: prefix-dev/setup-pixi@v0.8.1 + with: + pixi-version: latest + - name: Install host test prerequisites + run: sudo apt-get update && sudo apt-get install -y protobuf-compiler + - name: Verify release source + id: source + env: + RELEASE_TAG: ${{ inputs.tag }} + GH_TOKEN: ${{ github.token }} + run: | + pixi run bash -euc ' + [[ "$GITHUB_REPOSITORY" == buzzcrow/crowdb ]] + [[ "$RELEASE_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+(-[0-9A-Za-z.-]+)?$ ]] + [[ "$RELEASE_TAG" == "v$(cat VERSION)" ]] + revision=$(git rev-parse --verify "refs/tags/$RELEASE_TAG^{commit}") + [[ "$revision" == "$(git rev-parse HEAD)" ]] + [[ "$(gh release view "$RELEASE_TAG" --json isDraft --jq .isDraft)" == false ]] + printf "version=%s\nrevision=%s\n" "${RELEASE_TAG#v}" "$revision" >> "$GITHUB_OUTPUT" + ' + - name: Free disk space + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache + sudo apt-get clean + df -h / + - name: Build and test image without publication credentials + env: + CROWDB_PREVIEW_TEST_ARTIFACTS: ${{ runner.temp }}/crowdb-preview-artifacts + run: pixi run test-single-node-container + - name: Run S3 client acceptance + run: pixi run clean-env && pixi run -e s3-e2e test-boto3-e2e + - name: Run Iceberg client acceptance + run: pixi run clean-env && pixi run -e iceberg-e2e test-pyiceberg-e2e + - name: Run console acceptance + run: pixi run clean-env && pixi run test-console + - name: Require installed system browser + run: | + pixi run bash -euc ' + for browser in /snap/bin/chromium /usr/bin/chromium /usr/bin/chromium-browser /usr/bin/google-chrome /usr/bin/google-chrome-stable /usr/bin/microsoft-edge; do + [[ ! -x "$browser" ]] || exit 0 + done + echo "Release runner requires an installed system browser" >&2 + exit 1 + ' + - name: Run console UI acceptance + run: pixi run clean-env && pixi run test-console-ui + - name: Check Rust formatting and lint + run: pixi run rs-fmt-check && pixi run rs-lint + - name: Archive verified runtime files + run: pixi run tar -C target/container-runtime -czf target/container-runtime.tar.gz . + - uses: actions/upload-artifact@v4 + with: + name: verified-container-runtime + path: target/container-runtime.tar.gz + compression-level: 0 + retention-days: 7 + - name: Upload preview failure logs + if: failure() + uses: actions/upload-artifact@v4 + with: + name: release-preview-${{ github.run_attempt }} + path: ${{ runner.temp }}/crowdb-preview-artifacts + if-no-files-found: ignore + retention-days: 7 + + publish: + needs: verify + runs-on: ubuntu-24.04 + environment: DockerHub + permissions: + contents: read + id-token: write + steps: + - uses: actions/checkout@v4 + with: + ref: ${{ inputs.tag }} + fetch-depth: 0 + submodules: true + - uses: prefix-dev/setup-pixi@v0.8.1 + with: + pixi-version: latest + - name: Require publication credentials and unused immutable tags + env: + RELEASE_TAG: ${{ inputs.tag }} + REVISION: ${{ needs.verify.outputs.revision }} + DOCKERHUB_USERNAME: ${{ vars.DOCKERHUB_USERNAME }} + DOCKERHUB_TOKEN: ${{ secrets.DOCKERHUB_TOKEN }} + run: | + pixi run bash -euc ' + [[ -n "$DOCKERHUB_USERNAME" && -n "$DOCKERHUB_TOKEN" ]] + [[ "$(git rev-parse HEAD)" == "$REVISION" ]] + for tag in "$RELEASE_TAG" "git-$REVISION"; do + status=$(curl --silent --show-error --output /dev/null --write-out "%{http_code}" \ + "https://hub.docker.com/v2/namespaces/crowdb/repositories/crowdb-iceberg/tags/$tag") + [[ "$status" == 404 ]] || { echo "Immutable tag $tag is present or registry unavailable (HTTP $status)" >&2; exit 1; } + done + ' + - uses: docker/setup-buildx-action@v4 + - uses: actions/download-artifact@v4 + with: + name: verified-container-runtime + path: target + - name: Extract verified runtime files + run: pixi run bash -euc 'mkdir -p target/container-runtime && tar -C target/container-runtime -xzf target/container-runtime.tar.gz' + - uses: docker/login-action@v4 + with: + username: ${{ vars.DOCKERHUB_USERNAME }} + password: ${{ secrets.DOCKERHUB_TOKEN }} + - name: Build and publish signed-source image with attestations + id: build + uses: docker/build-push-action@v7 + with: + context: target/container-runtime + file: container/single-node-container/Dockerfile + platforms: linux/amd64 + push: true + provenance: mode=max + sbom: true + build-args: | + SOURCE_REVISION=${{ needs.verify.outputs.revision }} + PREVIEW_VERSION=${{ needs.verify.outputs.version }} + tags: | + docker.io/crowdb/crowdb-iceberg:${{ inputs.tag }} + docker.io/crowdb/crowdb-iceberg:git-${{ needs.verify.outputs.revision }} + - uses: sigstore/cosign-installer@v4.1.2 + - name: Sign published digest + env: + DIGEST: ${{ steps.build.outputs.digest }} + run: pixi run cosign sign --yes "docker.io/crowdb/crowdb-iceberg@$DIGEST" diff --git a/.gitignore b/.gitignore index 7db8a88fb..5a062c071 100644 --- a/.gitignore +++ b/.gitignore @@ -44,3 +44,6 @@ lib/crowdb-rpc/.cache/ app/crowdb-diskio/build/ app/crowdb-diskio/build*/ app/crowdb-diskio/.cache/ + +# Python tooling bytecode +__pycache__/ diff --git a/CHANGELOG.md b/CHANGELOG.md index 1e4375329..3ff11d724 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,36 +1,39 @@ + + + # Changelog -All notable changes to CROWDB will be documented in this file. +CROWDB is preparing its first development release, `0.1.0-dev`. Publication +is pending; this is not a production release or a compatibility promise. + +CROWDB does not yet maintain compatibility for persisted data, WAL, metadata, +or other on-disk formats. A newer checkout may be unable to read data created by +an older checkout. Use disposable data only. + +Changelog history begins with the first public Docker preview. Development work +before that baseline remains available in Git history and project requirements +but is intentionally not reconstructed as released change history. -The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), -and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +The changelog will follow [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) +from that baseline. Version tags alone do not imply Semantic Versioning +compatibility until the project explicitly adopts and documents a compatibility +policy. ## [Unreleased] -### Added -- Copyright headers on all source files -- AGENTS.md with project overview and dispatch table -- CONTRIBUTING.md, PR/Issue templates -- Demo recording plan (`doc/working/plan-demo.md`) - -### Changed -- Restructured agent workflows: conventions merged into `/coding` workflow -- Slimmed coding.md, doc.md, review.md -- README: added badges, folded Getting Started into `
` - -## [0.1.0] - 2026-07-13 - -### Added -- Multi-Paxos consensus with per-key slot pipelining and out-of-order apply -- WAL with multi-disk segments, batched durable flush, replay, and GC -- crowdb-tree storage engine: B+tree with delta chains, io_uring async I/O, epoch-safe lock-free reads, buffer pool -- `KVEngine` trait with in-memory and crowdb-tree backends -- crowdb-rpc services: Paxos (Prepare/Promise/Accept/Accepted), KV, Snapshot -- Leader election with term/ballot fencing and leader lease -- Reconfiguration: member add/remove, leader transfer, membership epoch fence -- `crowdb-kv-server` binary with HTTP management API -- `crowdb-kv-client` library with topology cache, retry, idempotency -- `crowdb-console`: web UI (Axum + React) and CLI for cluster lifecycle management -- Comprehensive design documentation (`doc/`) -- CI with GitHub Actions (fmt, clippy, test, Playwright E2E) -- Pre-commit hooks (cargo fmt, clippy, clang-format, clang-tidy) +### 0.1.0-dev preparation + +- Single-node Linux amd64 container with native Iceberg REST catalog and FileIO, + backed by CROWDB metadata, chunk storage and disk services. +- Persistent bootstrap, generated client credentials, health checks, bounded + service recovery and restart validation. +- S3 object access through an optional published endpoint. +- Host builds and runtime-only container packaging, with a manual Docker Hub + publication workflow for version and commit tags, signatures, SBOM and provenance. + +The intended image is `crowdb/crowdb-iceberg:v0.1.0-dev`; no published digest is +recorded yet. The GUI is not ready for this container. Multi-node deployment, +production hardening and data-format upgrades are outside this release. + +See the [container guide](doc/user-manual/docker-single-node-user-guide.md) for +supported startup, persistence, credentials and recovery behavior. diff --git a/CODE_OF_CONDUCT.md b/CODE_OF_CONDUCT.md index 2101d7d90..f68c57706 100644 --- a/CODE_OF_CONDUCT.md +++ b/CODE_OF_CONDUCT.md @@ -1,39 +1,78 @@ -# Contributor Covenant Code of Conduct + + -## Our Pledge +# CROWDB Code of Conduct -We pledge to make participation in our community a harassment-free experience for everyone, regardless of age, body size, visible or invisible disability, ethnicity, sex characteristics, gender identity and expression, level of experience, education, socio-economic status, nationality, personal appearance, race, religion, or sexual identity and orientation. +## Our pledge -## Our Standards +We pledge to make participation in CROWDB a harassment-free experience for +everyone, regardless of age, body size, visible or invisible disability, +ethnicity, sex characteristics, gender identity and expression, level of +experience, education, socioeconomic status, nationality, personal appearance, +race, caste, color, religion, or sexual identity and orientation. -Examples of behavior that contributes to a positive environment: +We will act and interact in ways that contribute to an open, welcoming, +diverse, inclusive, and healthy community. -- Demonstrating empathy and kindness toward other people -- Being respectful of differing opinions, viewpoints, and experiences -- Giving and gracefully accepting constructive feedback -- Accepting responsibility and apologizing to those affected by our mistakes -- Focusing on what is best for the overall community +## Expected behavior -Examples of unacceptable behavior: +- Be respectful of different backgrounds, viewpoints, and levels of experience. +- Give technical feedback about the work, supported by evidence where possible. +- Ask questions and correct mistakes without belittling people. +- Accept responsibility, apologize when appropriate, and repair harm. +- Respect privacy, security reports, embargoes, and requests for confidentiality. +- Prioritize the health of the project and community over winning an argument. -- The use of sexualized language or imagery, and sexual attention or advances -- Trolling, insulting or derogatory comments, and personal or political attacks -- Public or private harassment -- Publishing others' private information without explicit permission -- Other conduct which could reasonably be considered inappropriate in a professional setting +## Unacceptable behavior -## Enforcement Responsibilities - -Community leaders are responsible for clarifying and enforcing our standards and will take appropriate and fair corrective action in response to any behavior they deem inappropriate, threatening, offensive, or harmful. +- Harassment, intimidation, stalking, threats, or sustained disruption. +- Sexualized language, imagery, attention, or advances. +- Insults, derogatory comments, trolling, or personal and political attacks. +- Publishing private information without explicit permission. +- Pressuring anyone to disclose identity, credentials, employer, or private + communications. +- Retaliation against a person who raises a concern or participates in an + investigation. +- Conduct that would reasonably be considered inappropriate in a professional + community. ## Scope -This Code of Conduct applies within all community spaces, and also applies when an individual is officially representing the community in public spaces. +This code applies in project repositories, issue trackers, reviews, discussions, +chat, events, and other CROWDB community spaces. It also applies when someone is +officially representing the project in public. + +## Reporting + +Report abusive, harassing, or otherwise unacceptable behavior privately to +**crow.db@outlook.com**. Do not include sensitive personal information in a +public issue. + +Reports will be reviewed promptly, impartially, and as confidentially as +possible. People handling a report must disclose conflicts of interest and +recuse themselves when necessary. The project will protect the privacy and +safety of reporters and affected community members to the extent possible. + +Security vulnerabilities follow [SECURITY.md](SECURITY.md), not the conduct +reporting process. ## Enforcement -Instances of abusive, harassing, or otherwise unacceptable behavior may be reported to **crow.db@outlook.com**. All complaints will be reviewed and investigated promptly and fairly. +Project maintainers may remove, edit, or reject comments, commits, code, issues, +and other contributions that violate this code. Responses will be proportionate +to the behavior, its impact, and any pattern of prior conduct. Actions may +include: + +1. A private correction and explanation of the impact. +2. A formal warning with conditions for continued participation. +3. A temporary restriction from project interaction or representation. +4. A permanent ban from project spaces and representation. + +Maintainers will not publicly identify a reporter or disclose private report +details without permission, except when required to protect people or comply +with law. ## Attribution -This Code of Conduct is adapted from the [Contributor Covenant](https://www.contributor-covenant.org/), version 2.1. +This code is adapted from the [Contributor Covenant](https://www.contributor-covenant.org/), +version 2.1. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 6471147dc..06131b708 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -3,74 +3,139 @@ # Contributing to CROWDB -Thanks for your interest in contributing! This guide covers setup, conventions, and the PR process. +Thank you for contributing to CROWDB. -## Development Environment +## Development status -CROWDB uses [Pixi](https://pixi.sh) to pin the C++ toolchain, Rust compiler, and all native dependencies in a single lockfile. +CROWDB is under active development at version `0.1.0-dev`. It has not reached +alpha, is not recommended for production, and must be tested with disposable +data. Compatibility is not yet maintained for persisted data, WAL, metadata, or +other on-disk formats. A change may deliberately replace an unreleased format +without migration support when its requirement says so. -```bash -# Install pixi -curl -fsSL https://pixi.sh/install.sh | sh +The root [VERSION](VERSION) file is the project version source of truth. Cargo, +Pixi, the web package, and lockfiles must match it. Do not bump the version in an +ordinary contribution unless the pull request is explicitly release work. + +## Community and security + +Follow [CODE_OF_CONDUCT.md](CODE_OF_CONDUCT.md). Report vulnerabilities through +[SECURITY.md](SECURITY.md), not a public issue. + +## Development environment + +CROWDB uses [Pixi](https://pixi.sh) to pin Rust, C++, Node.js, and native build +dependencies. Run builds, tests, linters, and project executables through Pixi. -# Build everything (crowdb-tree C++ + Rust workspace + web UI) +```bash +# Build C++, the Rust workspace, and the web UI. pixi run build -# Run all tests +# Run host component, UI and Iceberg SDK suites. pixi run test-suite -# Lint -pixi run rs-fmt # Rust format -pixi run rs-lint # Rust clippy -pixi run tree-fmt # C++ format -pixi run tree-lint # C++ lint +# Check version metadata. +pixi run check-version + +# Check Rust formatting and lint. +pixi run rs-fmt-check +pixi run rs-lint + +# Format and lint changed C++ code. +pixi run tree-fmt +pixi run tree-lint + +# Check TypeScript and browser behavior when the UI changes. +pixi run ts-lint +pixi run test-console-ui ``` -See `pixi.toml` for the full list of tasks. +Playwright uses an installed system browser; do not install a repository-local +browser. See [pixi.toml](pixi.toml) for focused component tasks and +[tools/README.md](tools/README.md) for their scripts. Native Iceberg and official +SDK suites use the `iceberg-e2e` environment; see +[the test inventory](doc/working/test.md) for CI coverage and timing. -## Code Conventions +Container acceptance is a separate Linux amd64 gate requiring Docker: + +```sh +pixi run test-single-node-container +``` + +## Before writing code + +- Search existing requirements, designs, tests, and neighboring components. +- For architectural or externally visible behavior, agree on the requirement or + design before implementation. +- Preserve the existing authority and recovery model; do not create a local + fallback that can diverge from Group 0 or another durable authority. +- Do not add a lock to a hot path without discussing contention, ordering, + progress, and complexity trade-offs. +- Never commit credentials, tokens, private keys, production data, or generated + runtime directories. +- Add dependencies through the owning package manager and avoid newly published + versions until they have had time for ecosystem review. + +Start documentation work at `doc/doc_index.md`. Permanent architecture belongs +under `doc/design/`, user behavior in `doc/user-manual/user-guide.md`, future +contracts in `doc/backlog/`, and temporary execution plans in `doc/working/`. + +## Code and tests ### Rust -- `unsafe_code = deny` (except `crowdb-tree-ffi`). Clippy `pedantic = warn`. -- `Px` prefix for Paxos types (e.g. `PxGroupId`, `PxReplicaService`). -- Integration tests only — under each crate's `tests/`. No inline `#[cfg(test)] mod tests`. -- Shared test helpers: `tests/testkit/.rs`. -- Logging via `tracing` with structured fields, not inline in messages. -- No doc references in code comments — keep docs in `doc/`. +- Workspace crates deny unsafe code by default. Keep necessary unsafe code + confined to the existing FFI and low-level boundaries. +- Put integration tests in each crate's `tests/` directory. +- Put shared integration-test helpers in `tests/common/` and name helper types + with a `Test` prefix. +- Use structured `tracing` fields and established domain identifiers. +- Follow the existing `foo.rs` plus `foo/` module layout; do not add `mod.rs`. + +### C++ + +- Follow the repository `.clang-format` and `.clang-tidy` configuration. +- Keep public subsystem headers under the matching `include//` tree and + private headers with their implementation. +- Add GoogleTest coverage under the owning component's `tests/` directory. -### C++ (crowdb-tree) +### Web UI -- Follow `.clang-format` and `.clang-tidy` configs. -- GoogleTest for tests under `lib/crowdb-tree/tests/`. +- Follow existing React and TypeScript component patterns. +- Add focused unit tests and update Playwright E2E coverage for visible behavior. +- Verify the real backend path when UI behavior depends on service state. -### Design Docs +A bug fix should normally add a failing regression test first, then fix the root +cause. Do not weaken assertions, add retries, disable durability, or bypass +security boundaries to make a test pass. -- Start at `doc/doc_index.md` — match your task to a row, then open only that doc. -- If you add/rename/rescope a doc, update `doc_index.md` in the same commit. -- See `doc/design/kv/design-crowdb-kv.md` for architecture context before making non-trivial changes. +## Pull requests -## Pull Request Process +1. Create a focused branch from `main`. +2. Keep the change aligned with one requirement or one coherent maintenance + purpose. +3. Add or update tests and documentation with the implementation. +4. Run the relevant Pixi gates; run the full suite for cross-component changes. +5. Review generated files and the complete diff for secrets and unrelated edits. +6. Open a pull request explaining the problem, design choice, verification, and + known limitations. -1. Fork the repo and create a branch from `main`. -2. Write tests for your changes. All existing tests must pass. -3. Run `pixi run rs-fmt && pixi run rs-lint` before pushing. -4. Keep commits focused — one logical change per commit. -5. Reference the upstream design doc in your commit body (e.g. `design-slot.md §3`). -6. Open a PR with a clear description of what and why. +Use concise, single-line commit subjects. Keep unrelated refactors in separate +commits or pull requests. Do not force-push shared branches or bypass hooks and +release gates. -## Project Structure +## Changelog and releases -| Crate | What it is | -| --- | --- | -| `crowdb-kv` | Core library: Multi-Paxos consensus, WAL, storage engine, RPC | -| `crowdb-kv-server` | Server binary: crowdb-rpc + HTTP management API | -| `crowdb-kv-client` | Client library: topology cache, retry, idempotency | -| `crowdb-tree` | C++ storage engine (B+tree, delta chains, io_uring, buffer pool) | -| `crowdb-console` | Operations console: web UI (Axum + React) and CLI | +CROWDB begins public change history with its first Docker preview. Before that +baseline, Git history and requirement documents are the development record; do +not fabricate historical releases. Release preparation creates the first +versioned changelog entry. After that baseline, user-visible changes belong +under `Unreleased` and move to a dated section during release. -See `AGENTS.md` for a dispatch table on which docs to read for each type of task. +Only maintainers publish releases. A version or image tag is not a production or +compatibility promise unless the release notes explicitly make that promise. ## License -By contributing, you agree that your contributions will be licensed under the Apache License 2.0. +By contributing, you agree that your contribution is licensed under the Apache +License 2.0. diff --git a/Cargo.lock b/Cargo.lock index 45e176022..bd65d2d3d 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -331,6 +331,8 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a1dce859f0832a7d088c4f1119888ab94ef4b5d6795d1ce05afb7fe159d79f98" dependencies = [ "find-msvc-tools", + "jobserver", + "libc", "shlex", ] @@ -493,6 +495,21 @@ dependencies = [ "libc", ] +[[package]] +name = "crc" +version = "3.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5eb8a2a1cd12ab0d987a5d5e825195d372001a4094a0376319d5a0ad71c1ba0d" +dependencies = [ + "crc-catalog", +] + +[[package]] +name = "crc-catalog" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "217698eaf96b4a3f0bc4f3662aaa55bdf913cd54d7204591faa790070c6d0853" + [[package]] name = "crc32fast" version = "1.5.0" @@ -590,9 +607,41 @@ version = "0.8.21" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d0a5c400df2834b80a4c3327b3aad3a4c4cd4de0629063962b03235697506a28" +[[package]] +name = "crowdb-access-iceberg" +version = "0.1.0-dev" +dependencies = [ + "arc-swap", + "async-trait", + "base64", + "bytes", + "chrono", + "crc32fast", + "crowdb-access-iceberg", + "crowdb-chunk-client", + "crowdb-chunk-kv-client", + "crowdb-common", + "crowdb-protocol", + "data-encoding", + "flatbuffers", + "flate2", + "hmac", + "lz4_flex", + "serde", + "serde_json", + "sha2", + "snap", + "subtle", + "thiserror 2.0.18", + "tokio", + "tracing", + "uuid", + "zstd", +] + [[package]] name = "crowdb-access-s3" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "aes-gcm", "arc-swap", @@ -623,26 +672,39 @@ dependencies = [ [[package]] name = "crowdb-access-server" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ + "arc-swap", "async-trait", + "base64", "chrono", + "crc", + "crowdb-access-iceberg", "crowdb-access-s3", + "crowdb-access-server", "crowdb-chunk-client", "crowdb-chunk-kv-client", "crowdb-chunkdb-client", "crowdb-common", + "crowdb-diskdb-client", "crowdb-diskio-client", "crowdb-kv-client", "crowdb-protocol", "crowdb-rpc-ffi", "crowdb-test-harness", "futures", + "hmac", "http-body-util", "hyper", "hyper-util", + "md-5", "percent-encoding", + "quick-xml", + "reqwest", + "serde", "serde_json", + "sha1", + "sha2", "thiserror 2.0.18", "tokio", "tracing", @@ -651,7 +713,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-client" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "async-trait", @@ -675,7 +737,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-kv" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "async-trait", @@ -694,7 +756,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-kv-client" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "async-trait", @@ -713,7 +775,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-kv-server" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "async-trait", @@ -739,7 +801,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-stream" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "async-trait", @@ -761,7 +823,7 @@ dependencies = [ [[package]] name = "crowdb-chunkdb" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "axum", @@ -797,7 +859,7 @@ dependencies = [ [[package]] name = "crowdb-chunkdb-client" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "bytes", @@ -812,7 +874,7 @@ dependencies = [ [[package]] name = "crowdb-cli" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "axum", "chrono", @@ -843,7 +905,7 @@ dependencies = [ [[package]] name = "crowdb-common" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "crowdb-test-harness", "flate2", @@ -857,11 +919,12 @@ dependencies = [ "tracing", "tracing-appender", "tracing-subscriber", + "xxhash-rust", ] [[package]] name = "crowdb-console-shared" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "async-trait", "axum", @@ -885,7 +948,7 @@ dependencies = [ [[package]] name = "crowdb-diskdb" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "axum", @@ -921,7 +984,7 @@ dependencies = [ [[package]] name = "crowdb-diskdb-client" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "bytes", @@ -940,7 +1003,7 @@ dependencies = [ [[package]] name = "crowdb-diskio-client" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "bytes", @@ -960,7 +1023,7 @@ dependencies = [ [[package]] name = "crowdb-kv" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "bytes", @@ -993,7 +1056,7 @@ dependencies = [ [[package]] name = "crowdb-kv-client" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "axum", @@ -1017,7 +1080,7 @@ dependencies = [ [[package]] name = "crowdb-kv-server" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "axum", @@ -1040,9 +1103,32 @@ dependencies = [ "utoipa", ] +[[package]] +name = "crowdb-monitor" +version = "0.1.0-dev" +dependencies = [ + "clap", + "crowdb-diskio-client", + "crowdb-kv-client", + "crowdb-protocol", + "crowdb-rpc-ffi", + "crowdb-test-harness", + "flatbuffers", + "rand 0.8.6", + "reqwest", + "rustix", + "serde", + "serde_json", + "sha2", + "thiserror 2.0.18", + "tokio", + "toml", + "uuid", +] + [[package]] name = "crowdb-protocol" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "bincode", "bytes", @@ -1060,7 +1146,7 @@ dependencies = [ [[package]] name = "crowdb-rpc-ffi" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "arc-swap", "bytes", @@ -1075,7 +1161,7 @@ dependencies = [ [[package]] name = "crowdb-test-harness" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "crowdb-chunkdb-client", "crowdb-diskdb-client", @@ -1092,7 +1178,7 @@ dependencies = [ [[package]] name = "crowdb-tree-ffi" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "bytes", "cc", @@ -1104,7 +1190,7 @@ dependencies = [ [[package]] name = "crowdb-web" -version = "0.1.0" +version = "0.1.0-dev" dependencies = [ "async-trait", "axum", @@ -1114,6 +1200,7 @@ dependencies = [ "crowdb-diskdb-client", "crowdb-kv", "crowdb-kv-client", + "crowdb-monitor", "crowdb-protocol", "crowdb-rpc-ffi", "crowdb-test-harness", @@ -1123,10 +1210,12 @@ dependencies = [ "reqwest", "serde", "serde_json", + "subtle", "tokio", "tower", "tower-http", "tracing", + "uuid", ] [[package]] @@ -2084,6 +2173,15 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" +[[package]] +name = "jobserver" +version = "0.1.32" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "48d1dbcbbeb6a7fec7e059840aa538bd62aaccf972c7346c4d9d2059312853d0" +dependencies = [ + "libc", +] + [[package]] name = "js-sys" version = "0.3.98" @@ -2197,6 +2295,15 @@ version = "0.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "112b39cec0b298b6c1999fee3e31427f74f676e4cb9879ed1a121b43661a4154" +[[package]] +name = "lz4_flex" +version = "0.11.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a" +dependencies = [ + "twox-hash", +] + [[package]] name = "matchers" version = "0.2.0" @@ -2212,6 +2319,16 @@ version = "0.7.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0e7465ac9959cc2b1404e8e2367b43684a6d13790fe23056cc8c6c5a6b7bcb94" +[[package]] +name = "md-5" +version = "0.10.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d89e7ee0cfbedfc4da3340218492196241d89eefb6dab27de5df917a6d2e78cf" +dependencies = [ + "cfg-if", + "digest", +] + [[package]] name = "md5" version = "0.7.0" @@ -3465,6 +3582,12 @@ version = "1.15.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "67b1b7a3b5fe4f1376887184045fcf45c69e92af734b7aaddc05fb777b6fbd03" +[[package]] +name = "snap" +version = "1.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" + [[package]] name = "socket2" version = "0.6.3" @@ -3995,6 +4118,12 @@ version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" +[[package]] +name = "twox-hash" +version = "2.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8464ec13c3691491391d9fce00f6416c9a48e46972f72d7865688be2080192c9" + [[package]] name = "typenum" version = "1.20.0" @@ -4090,6 +4219,7 @@ checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239" dependencies = [ "getrandom 0.4.2", "js-sys", + "serde_core", "wasm-bindgen", ] @@ -4832,3 +4962,31 @@ name = "zmij" version = "1.0.21" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa" + +[[package]] +name = "zstd" +version = "0.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a" +dependencies = [ + "zstd-safe", +] + +[[package]] +name = "zstd-safe" +version = "7.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882" +dependencies = [ + "zstd-sys", +] + +[[package]] +name = "zstd-sys" +version = "2.1.0+zstd.1.5.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0" +dependencies = [ + "cc", + "pkg-config", +] diff --git a/Cargo.toml b/Cargo.toml index 56f6c1e77..0040296b1 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -23,7 +23,9 @@ members = [ "lib/crowdb-test-harness", "app/crowdb-chunk-kv-server", "app/crowdb-access-server", + "container/crowdb-monitor", "lib/crowdb-access-s3", + "lib/crowdb-access-iceberg", ] exclude = ["third-party/hyper"] # : crowdb-tree/ffi moved from `exclude` into `members` now that @@ -37,7 +39,7 @@ exclude = ["third-party/hyper"] # `unsafe_code = "deny"`. [workspace.package] -version = "0.1.0" +version = "0.1.0-dev" edition = "2021" rust-version = "1.75" license = "Apache-2.0" diff --git a/README.md b/README.md index 7b38cb775..79d5ec707 100644 --- a/README.md +++ b/README.md @@ -10,12 +10,13 @@ CROWDB is a distributed storage platform for objects, tables, and AI datasets. It owns the data path from S3, Iceberg, and native Dataset access through distributed metadata and chunk storage to disk—and eventually GPU memory. -S3, Iceberg, and Dataset are first-class access models, not wrappers stacked on -top of one another. +Version `0.1.0-dev` is the first development release being prepared for public +evaluation. Use disposable data; production use and on-disk upgrade compatibility +are not supported. Dataset and direct GPU delivery remain planned work. - Use **S3** for familiar object access. - Use **Iceberg** for native catalogs, tables, snapshots, and immutable files. -- Use **Dataset** for samples, shards, tensors, batches, and direct data access. +- **Dataset** is planned for samples, shards, tensors, batches, and direct data access. ## Three Layers, One Data Path @@ -29,7 +30,7 @@ top of one another. | LAYER 3 — ACCESS | | | | S3 Iceberg Dataset | -| [implemented] [in progress] [design] | +| [implemented] [implemented] [design] | | HTTP objects HTTP tables HTTP + native client | | | | Access Server serves HTTP. Dataset native access can bypass it. | @@ -40,9 +41,9 @@ top of one another. | LAYER 2 — CHUNK | | | | Distributed structures: Chunk Stream Chunk-KV | -| | | | +| | | | | Data path: chunk client -> Chunk I/O -> ChunkDB -> DiskIO -> DiskDB | -| | | +| | | | Accelerated path: DiskIO buffer -- RDMA / GDS -------> GPU memory | +------------------------------------+-------------------------------------+ | @@ -65,7 +66,7 @@ another namespace, lifecycle, RPC, copy, and recovery model. When that boundary becomes the bottleneck, the layers above it can only work around it. CROWDB exists to own the complete data path. S3 objects, Iceberg tables, and AI -datasets are native access models over the same distributed storage core. They +datasets are intended as native access models over the same distributed storage core. They share durability, placement, protection, and reclamation without pretending that one model is merely a convention inside another. @@ -91,16 +92,16 @@ layers that remain useful independently. append. Chunk-KV uses range partitions that split and rebalance online while reads and writes continue. - **Native access models:** S3, Iceberg, and Dataset share the core without - being wrappers around one another. Dataset can also route directly to the - data topology and is designed toward RDMA and direct GPU delivery. + being wrappers around one another. The planned Dataset model targets direct + topology access, RDMA and GPU delivery. ## Where It Stands -| Access model | Status | What it means | -| ------------ | ----------- | ----------------------------------------------------- | -| S3 | Implemented | Core HTTP object operations and bounded streaming | -| Iceberg | In progress | Native core format v1, v2, and v3 storage semantics | -| Dataset | Design | HTTP, topology-aware native client, and GPU delivery | +| Access model | Status | What it means | +| ------------ | ----------- | ---------------------------------------------------- | +| S3 | Implemented | Core HTTP object operations and bounded streaming | +| Iceberg | Implemented | Native catalog, FileIO and core v1/v2/v3 semantics | +| Dataset | Design | HTTP, topology-aware native client, and GPU delivery | The KV, tree, DiskDB, ChunkDB, chunk I/O, Chunk Stream, Chunk-KV, RPC, operations console, and core S3 foundation have working implementations. See @@ -108,33 +109,19 @@ the [backlog](doc/backlog/backlog.md) for current delivery scope. ## Quick Start -CROWDB uses [Pixi](https://pixi.sh) to pin its Rust and C++ toolchains and -native dependencies. +The first Linux amd64 image is being prepared for manual publication. Once +`v0.1.0-dev` is published, start the Iceberg catalog and storage with Docker: -### S3 cluster example - -Build the binaries, then start a local S3 cluster with the CLI: - -```bash -curl -fsSL https://pixi.sh/install.sh | sh -pixi run build - -./target/release/crowdb-cli s3 cluster start --root /tmp/s3-cluster -./target/release/crowdb-cli s3 cluster status --root /tmp/s3-cluster +```sh +docker run -d --name crowdb-iceberg \ + -p 127.0.0.1:80:80 \ + crowdb/crowdb-iceberg:v0.1.0-dev ``` -The command starts the storage services, S3 endpoint, and Web management -server. Open the printed Web URL, normally -[http://127.0.0.1:14000/](http://127.0.0.1:14000/). - -Create a bucket and round-trip an object: - -```bash -./target/release/crowdb-cli s3 bucket put --root /tmp/s3-cluster bucket1 -./target/release/crowdb-cli s3 object put --root /tmp/s3-cluster bucket1 hello.txt \ - --text "hello from CROWDB" -./target/release/crowdb-cli s3 object get --root /tmp/s3-cluster bucket1 hello.txt -``` +Follow the [single-node Docker guide](doc/user-manual/docker-single-node-user-guide.md) +for startup checks, client credentials, persistent volumes and recovery. +For source builds and development with Pixi, see +[CONTRIBUTING.md](CONTRIBUTING.md). ## Explore @@ -144,6 +131,8 @@ Create a bucket and round-trip an object: subsystem design. - [User guide](doc/user-manual/user-guide.md) — setup, console, CLI, and supported operations. +- [Single-node Docker guide](doc/user-manual/docker-single-node-user-guide.md) + — preview image, volume, credentials, clients, and recovery. - [Backlog](doc/backlog/backlog.md) — what is implemented, in progress, and planned. diff --git a/SECURITY.md b/SECURITY.md index 924b362b7..0fe289ccc 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -1,3 +1,6 @@ + + + # Security Policy ## Reporting a Vulnerability @@ -5,13 +8,18 @@ If you discover a security vulnerability in CROWDB, please report it responsibly: 1. **Do not** open a public GitHub issue. -2. Email **crow.db@outlook.com** with a description of the vulnerability and reproduction steps. -3. You will receive an acknowledgment within 48 hours. +2. Email **crow.db@outlook.com** with the affected version or commit, a description + of the vulnerability, its impact, and reproduction steps. +3. Omit live credentials and private user data from the report. ## Scope -CROWDB is currently a pre-production project. Security fixes will be prioritized but may not have defined SLAs. +CROWDB `0.1.0-dev` is a development version for evaluation with disposable data. +There is no production support commitment, supported stable release series, or +guaranteed response time. Security reports are reviewed by the maintainers. ## Disclosure -Once a fix is released, we will publish a GitHub Security Advisory crediting the reporter (unless they prefer to remain anonymous). +Please coordinate public disclosure with the maintainers while a report is +investigated and a fix is prepared. Reporter credit should follow the reporter's +preference, including anonymity. diff --git a/VERSION b/VERSION new file mode 100644 index 000000000..0d4d12494 --- /dev/null +++ b/VERSION @@ -0,0 +1 @@ +0.1.0-dev diff --git a/app/crowdb-access-server/Cargo.toml b/app/crowdb-access-server/Cargo.toml index a045e4638..9d38a1e8a 100644 --- a/app/crowdb-access-server/Cargo.toml +++ b/app/crowdb-access-server/Cargo.toml @@ -12,52 +12,91 @@ workspace = true [features] default = ["s3"] +test-util = [] s3-e2e = ["s3"] +iceberg-e2e = [] s3 = [ - "dep:async-trait", - "dep:chrono", - "dep:crowdb-access-s3", - "dep:crowdb-chunk-client", - "dep:crowdb-chunk-kv-client", "dep:crowdb-common", - "dep:crowdb-kv-client", "dep:futures", - "dep:http-body-util", - "dep:hyper", - "dep:hyper-util", - "dep:percent-encoding", - "dep:thiserror", ] [dependencies] -async-trait = { version = "0.1", optional = true } -chrono = { version = "0.4", default-features = false, features = ["std"], optional = true } -crowdb-access-s3 = { path = "../../lib/crowdb-access-s3", optional = true } -crowdb-chunk-client = { path = "../../lib/crowdb-chunk-client", optional = true } -crowdb-chunk-kv-client = { path = "../../lib/crowdb-chunk-kv-client", optional = true } +async-trait = "0.1" +chrono = { version = "0.4", default-features = false, features = ["std"] } +crowdb-access-iceberg = { path = "../../lib/crowdb-access-iceberg" } +crowdb-access-s3 = { path = "../../lib/crowdb-access-s3" } +crowdb-chunk-client = { path = "../../lib/crowdb-chunk-client" } +crowdb-chunk-kv-client = { path = "../../lib/crowdb-chunk-kv-client" } crowdb-common = { path = "../../lib/crowdb-common/rust", optional = true } -crowdb-kv-client = { path = "../../lib/crowdb-kv-client", optional = true } +crowdb-kv-client = { path = "../../lib/crowdb-kv-client" } +crowdb-protocol = { path = "../../lib/crowdb-protocol" } futures = { version = "0.3", optional = true } -http-body-util = { version = "0.1", optional = true } -hyper = { workspace = true, features = ["http1", "server"], optional = true } -hyper-util = { version = "0.1", features = ["tokio"], optional = true } -percent-encoding = { version = "2", optional = true } +http-body-util = "0.1" +hyper = { workspace = true, features = ["http1", "server"] } +hyper-util = { version = "0.1", features = ["tokio"] } +percent-encoding = "2" +quick-xml = "0.38" +base64 = "0.22" +md-5 = "0.10" +serde_json = "1" +serde = { version = "1", features = ["derive"] } +sha2 = "0.10" +sha1 = "0.10" +crc = "3.3" tokio = { workspace = true, features = ["io-util", "macros", "net", "rt-multi-thread", "signal", "sync"] } tracing = { workspace = true } tracing-subscriber = { workspace = true, features = ["env-filter", "fmt"] } -thiserror = { workspace = true, optional = true } +thiserror = { workspace = true } [dev-dependencies] +crowdb-access-server = { path = ".", default-features = false, features = ["test-util"] } +hmac = "0.12" +arc-swap = "1.9" +async-trait = "0.1" crowdb-chunkdb-client = { path = "../../lib/crowdb-chunkdb-client" } +crowdb-diskdb-client = { path = "../../lib/crowdb-diskdb-client" } crowdb-diskio-client = { path = "../../lib/crowdb-diskio-client", features = ["test-util"] } -crowdb-protocol = { path = "../../lib/crowdb-protocol" } crowdb-rpc-ffi = { path = "../../lib/crowdb-rpc/ffi" } crowdb-test-harness = { path = "../../lib/crowdb-test-harness", features = ["chunk-kv", "chunkdb", "diskdb", "diskio"] } serde_json = "1" -tokio = { workspace = true, features = ["io-util", "macros", "net", "rt-multi-thread", "sync"] } +reqwest = { version = "0.12", default-features = false, features = ["rustls-tls"] } +tokio = { workspace = true, features = ["io-util", "macros", "net", "rt-multi-thread", "sync", "test-util"] } [[test]] name = "s3_full_stack_test" path = "tests/s3_full_stack_test.rs" harness = false required-features = ["s3-e2e"] + +[[bin]] +name = "crowdb-iceberg" +path = "src/iceberg_main.rs" + +[[test]] +name = "iceberg_full_stack_test" +path = "tests/iceberg_full_stack_test.rs" +required-features = ["iceberg-e2e"] + +[[test]] +name = "iceberg_file_storage_test" +path = "tests/iceberg_file_storage_test.rs" +required-features = ["iceberg-e2e"] + +[[test]] +name = "iceberg_gc_control_test" +path = "tests/iceberg_gc_control_test.rs" +required-features = ["iceberg-e2e"] + +[[test]] +name = "iceberg_gc_budget_test" +path = "tests/iceberg_gc_budget_test.rs" + +[[test]] +name = "iceberg_gc_capacity_test" +path = "tests/iceberg_gc_capacity_test.rs" +required-features = ["iceberg-e2e"] + +[[test]] +name = "iceberg_file_http_test" +path = "tests/iceberg_file_http_test.rs" +required-features = ["iceberg-e2e"] diff --git a/app/crowdb-access-server/src/credentials.rs b/app/crowdb-access-server/src/credentials.rs index 854669d32..8465dd0cf 100644 --- a/app/crowdb-access-server/src/credentials.rs +++ b/app/crowdb-access-server/src/credentials.rs @@ -30,6 +30,10 @@ pub enum CredentialAuthorityError { Record(#[from] SecretError), #[error("S3 credential key is malformed")] InvalidKey, + #[error("S3 user has multiple credential records")] + DuplicateUser, + #[error("S3 user credential is disabled or malformed")] + InvalidUserCredential, } impl CredentialAuthority { @@ -64,6 +68,75 @@ impl CredentialAuthority { Err(CredentialAuthorityError::Collision) } + /// Reuses an existing credential for a single-writer bootstrap, including + /// after an issuance response is lost. Never creates a second credential + /// when the user's durable record is already present. + /// + /// # Errors + /// + /// Fails closed on duplicate or invalid user records and storage errors. + pub async fn ensure_user(&self, user: &[u8]) -> Result { + match self.lookup_user(user).await? { + Some(token) => Ok(token), + None => self.issue_user(user).await, + } + } + + /// Reads one user's credential without creating durable state. + /// + /// # Errors + /// + /// Fails closed on duplicate or invalid user records and storage errors. + pub async fn lookup_user( + &self, + user: &[u8], + ) -> Result, CredentialAuthorityError> { + if user.is_empty() { + return Err(CredentialAuthorityError::EmptyUser); + } + let outcome = self + .control + .scan( + 0, + 0, + CREDENTIAL_PREFIX, + b"", + b"", + u32::MAX, + ReadMode::Linearizable, + None, + false, + None, + ) + .await?; + let mut found = None; + for (key, value) in outcome.items { + let access_key = key + .strip_prefix(CREDENTIAL_PREFIX) + .and_then(|value| std::str::from_utf8(value).ok()) + .filter(|value| !value.is_empty()) + .ok_or(CredentialAuthorityError::InvalidKey)?; + let record = DurableCredentialRecord::decode(&value)?; + if record.user != user { + continue; + } + if found.is_some() { + return Err(CredentialAuthorityError::DuplicateUser); + } + let credential = self.cipher.decrypt(user, access_key, &record.encrypted)?; + if !credential.enabled { + return Err(CredentialAuthorityError::InvalidUserCredential); + } + let secret_key = String::from_utf8(credential.secret_key.clone()) + .map_err(|_| CredentialAuthorityError::InvalidUserCredential)?; + found = Some(IssuedUserToken { + access_key_id: access_key.to_owned(), + secret_key, + }); + } + Ok(found) + } + /// Loads and decrypts the complete credential snapshot from group 0. /// /// # Errors diff --git a/app/crowdb-access-server/src/iceberg.rs b/app/crowdb-access-server/src/iceberg.rs new file mode 100644 index 000000000..f6b55c9da --- /dev/null +++ b/app/crowdb-access-server/src/iceberg.rs @@ -0,0 +1,48 @@ +//! Independent Iceberg listener and catalog-management runtime. + +mod body; +mod connection; +mod file_admission; +mod file_auth; +mod file_body; +mod file_complete; +mod file_encoding; +mod file_http; +mod file_recovery; +mod file_request; +mod file_response; +mod file_selection; +mod file_upload; +mod gc_control; +mod gc_runtime; +mod http; +mod metrics; +mod namespace_read; +mod namespace_request; +mod namespace_write; +mod recovery; +mod routes; +mod runtime; +mod table_credentials; +mod table_limits; +mod table_read; +mod table_recovery; +mod table_write; + +pub use file_admission::{FileAdmissionError, FileServiceLimits, FileTransferAdmission}; +pub use file_auth::authenticate_file_request; +pub use file_body::{FileBodyError, FileReadBody, FileResponseBudget}; +pub use file_complete::FileCompleteBody; +pub use file_encoding::{FileEncodingError, FileUploadBody}; +pub use file_request::{FileRequest, FileRequestError, MultipartRequest}; +pub use file_response::{FileResponseError, FileS3ErrorCode, MultipartResponses}; +pub use file_selection::{CompletePart, CompleteRequestError, CompleteResolveError, CompleteSelection}; +pub use file_upload::{FileUploadBudget, FileUploadConstraints, FileUploadError}; +pub use http::{serve, IcebergHttpService}; +pub use metrics::{IcebergMetricsSnapshot, MetricCounts, ICEBERG_OUTCOME_NAMES, ICEBERG_ROUTE_NAMES}; +pub use runtime::{run, IcebergRuntimeConfig}; + +#[cfg(feature = "test-util")] +pub use connection::active_io_for_tests; +#[cfg(feature = "test-util")] +pub use gc_runtime::budget::{BudgetedGcBlocks, BudgetedGcStore, GcIoBudget}; diff --git a/app/crowdb-access-server/src/iceberg/body.rs b/app/crowdb-access-server/src/iceberg/body.rs new file mode 100644 index 000000000..5c0c8d292 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/body.rs @@ -0,0 +1,138 @@ +use std::pin::Pin; +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; +use std::task::{Context, Poll}; + +use hyper::body::{Body, Bytes, Frame, SizeHint}; + +use super::file_body::FileReadBody; +use super::file_complete::FileCompleteBody; +use super::metrics::RequestObservation; + +pub(super) struct SpoolPermit(Arc); + +impl SpoolPermit { + pub(super) fn acquire(active: &Arc) -> Option { + active + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |count| { + (count < 4).then_some(count + 1) + }) + .ok()?; + Some(Self(active.clone())) + } +} + +impl Drop for SpoolPermit { + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::AcqRel); + } +} + +pub(super) struct IcebergBody { + bytes: Bytes, + permit: Option, + file: Option, + complete: Option, + observation: Option>, +} + +impl IcebergBody { + pub(super) fn with_observation(mut self, observation: Arc) -> Self { + self.observation = Some(observation); + self + } + + pub(super) fn with_spool_permit(mut self, permit: SpoolPermit) -> Self { + self.permit = Some(permit); + self + } + + pub(super) fn new(bytes: Vec) -> Self { + Self { + bytes: Bytes::from(bytes), + permit: None, + file: None, + complete: None, + observation: None, + } + } + pub(super) fn with_permit(bytes: Vec, permit: SpoolPermit) -> Self { + Self { + bytes: Bytes::from(bytes), + permit: Some(permit), + file: None, + complete: None, + observation: None, + } + } + + pub(super) fn file(body: FileReadBody) -> Self { + Self { + bytes: Bytes::new(), + permit: None, + file: Some(body), + complete: None, + observation: None, + } + } + + pub(super) fn complete(body: FileCompleteBody) -> Self { + Self { + bytes: Bytes::new(), + permit: None, + file: None, + complete: Some(body), + observation: None, + } + } +} + +impl Body for IcebergBody { + type Data = Bytes; + type Error = Box; + + fn poll_frame( + self: Pin<&mut Self>, + context: &mut Context<'_>, + ) -> Poll, Self::Error>>> { + let body = self.get_mut(); + let result = if let Some(complete) = &mut body.complete { + Pin::new(complete) + .poll_frame(context) + .map(|frame| frame.map(|result| result.map_err(Into::into))) + } else if let Some(file) = &mut body.file { + Pin::new(file) + .poll_frame(context) + .map(|frame| frame.map(|result| result.map_err(Into::into))) + } else if body.bytes.is_empty() { + Poll::Ready(None) + } else { + let length = body.bytes.len().min(16 * 1024); + Poll::Ready(Some(Ok(Frame::data(body.bytes.split_to(length))))) + }; + if let Poll::Ready(Some(Ok(frame))) = &result { + if let Some(bytes) = frame.data_ref() { + if let Some(observation) = &body.observation { + observation.response_bytes(bytes.len()); + } + } + } + result + } + + fn is_end_stream(&self) -> bool { + self.bytes.is_empty() + && self.file.as_ref().map_or(true, Body::is_end_stream) + && self.complete.as_ref().map_or(true, Body::is_end_stream) + } + fn size_hint(&self) -> SizeHint { + if let Some(complete) = &self.complete { + return complete.size_hint(); + } + self.file + .as_ref() + .map_or_else(|| SizeHint::with_exact(self.bytes.len() as u64), Body::size_hint) + } +} diff --git a/app/crowdb-access-server/src/iceberg/connection.rs b/app/crowdb-access-server/src/iceberg/connection.rs new file mode 100644 index 000000000..5e74d066e --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/connection.rs @@ -0,0 +1,117 @@ +use std::io; +use std::pin::Pin; +use std::sync::{ + atomic::{AtomicBool, AtomicU64, Ordering}, + Arc, +}; +use std::task::{Context, Poll}; +use std::time::Duration; + +use tokio::io::{AsyncRead, AsyncWrite, ReadBuf}; +use tokio::time::Instant; + +pub(super) struct ConnectionActivity { + start: Instant, + latest_ms: AtomicU64, + request_started: AtomicBool, +} + +impl ConnectionActivity { + pub(super) fn new() -> Arc { + Arc::new(Self { + start: Instant::now(), + latest_ms: AtomicU64::new(0), + request_started: AtomicBool::new(false), + }) + } + + fn record(&self) { + let elapsed = u64::try_from(self.start.elapsed().as_millis()).unwrap_or(u64::MAX); + self.latest_ms.fetch_max(elapsed, Ordering::Relaxed); + } + + pub(super) fn dispatch_deadline(&self, request_timeout: Duration) -> Instant { + self.start + request_timeout - (request_timeout / 10).min(Duration::from_millis(100)) + } + + pub(super) fn mark_request_started(&self) { + self.request_started.store(true, Ordering::Release); + } + + pub(super) async fn header_expired(&self, deadline: Instant) { + tokio::time::sleep_until(deadline).await; + if self.request_started.load(Ordering::Acquire) { + std::future::pending::<()>().await; + } + } + + pub(super) async fn expired(&self, idle: Duration) { + loop { + let latest = self.latest_ms.load(Ordering::Relaxed); + let idle_deadline = self.start + Duration::from_millis(latest) + idle; + tokio::time::sleep_until(idle_deadline).await; + if self.latest_ms.load(Ordering::Relaxed) == latest { + return; + } + } + } +} + +pub(super) struct ActiveIo { + stream: Stream, + activity: Arc, +} + +impl ActiveIo { + pub(super) fn new(stream: Stream, activity: Arc) -> Self { + Self { stream, activity } + } +} + +impl AsyncRead for ActiveIo { + fn poll_read( + self: Pin<&mut Self>, + context: &mut Context<'_>, + buffer: &mut ReadBuf<'_>, + ) -> Poll> { + let this = self.get_mut(); + let before = buffer.filled().len(); + let result = Pin::new(&mut this.stream).poll_read(context, buffer); + if matches!(result, Poll::Ready(Ok(()))) && buffer.filled().len() > before { + this.activity.record(); + } + result + } +} + +impl AsyncWrite for ActiveIo { + fn poll_write(self: Pin<&mut Self>, context: &mut Context<'_>, bytes: &[u8]) -> Poll> { + let this = self.get_mut(); + let result = Pin::new(&mut this.stream).poll_write(context, bytes); + if matches!(result, Poll::Ready(Ok(length)) if length > 0) { + this.activity.record(); + } + result + } + + fn poll_flush(self: Pin<&mut Self>, context: &mut Context<'_>) -> Poll> { + Pin::new(&mut self.get_mut().stream).poll_flush(context) + } + + fn poll_shutdown(self: Pin<&mut Self>, context: &mut Context<'_>) -> Poll> { + Pin::new(&mut self.get_mut().stream).poll_shutdown(context) + } +} + +#[cfg(feature = "test-util")] +pub fn active_io_for_tests( + stream: Stream, + idle: Duration, +) -> ( + impl AsyncRead + AsyncWrite + Unpin, + impl std::future::Future, +) { + let activity = ConnectionActivity::new(); + let tracked = ActiveIo::new(stream, activity.clone()); + (tracked, async move { activity.expired(idle).await }) +} diff --git a/app/crowdb-access-server/src/iceberg/file_admission.rs b/app/crowdb-access-server/src/iceberg/file_admission.rs new file mode 100644 index 000000000..ae9e48ca1 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_admission.rs @@ -0,0 +1,302 @@ +use std::sync::Arc; + +use crowdb_access_iceberg::catalog::CatalogContext; +use crowdb_access_iceberg::file::{ + ByteRange, FileBlockStore, FileGrant, FileGrantError, FileIdentity, FileLocation, FileOperation, + FileRecord, FileTree, MultipartAdmissionRecord, MultipartLimits, MultipartPhase, MultipartSession, +}; +use hyper::body::{Body, Bytes}; + +use super::file_body::{FileBodyError, FileReadBody, FileResponseBudget}; +use super::file_request::{FileRequest, MultipartRequest}; +use super::file_upload::{FileUploadBudget, FileUploadConstraints, FileUploadError}; + +pub(super) const MULTIPART_COPY_BYTES: usize = 1024 * 1024 + / crowdb_access_iceberg::file::NATIVE_FILE_BLOCK_BYTES + * crowdb_access_iceberg::file::NATIVE_FILE_BLOCK_BYTES; + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct FileServiceLimits { + pub max_request_bytes: u64, + pub max_file_bytes: u64, + pub max_part_bytes: u64, + pub max_staged_bytes: u64, +} + +impl FileServiceLimits { + fn validate(self) -> Result<(), FileAdmissionError> { + if self.max_request_bytes == 0 + || self.max_file_bytes == 0 + || self.max_part_bytes == 0 + || self.max_staged_bytes == 0 + || self.max_request_bytes > u64::MAX / 8 + || self.max_file_bytes > u64::MAX / 8 + || self.max_part_bytes > u64::MAX / 8 + || self.max_staged_bytes > u64::MAX / 8 + { + return Err(FileAdmissionError::Bounds); + } + Ok(()) + } +} + +#[derive(Debug, thiserror::Error)] +pub enum FileAdmissionError { + #[error(transparent)] + Grant(#[from] FileGrantError), + #[error("native file request is outside the authorized session")] + Scope, + #[error("native file request exceeds service or session limits")] + Bounds, + #[error("native file admission state is invalid or exhausted")] + State, + #[error(transparent)] + Upload(#[from] FileUploadError), + #[error(transparent)] + Read(#[from] FileBodyError), +} + +pub struct FileTransferAdmission { + context: CatalogContext, + principal: [u8; 32], + location: FileLocation, + operation: FileOperation, + request_bytes: u64, + file_bytes: u64, + staged_bytes: u64, +} + +impl FileTransferAdmission { + #[must_use] + pub const fn context(&self) -> CatalogContext { + self.context + } + + #[must_use] + pub const fn principal(&self) -> [u8; 32] { + self.principal + } + + /// Returns limits for a newly created durable multipart session. + /// # Errors + /// Rejects non-create requests or an exhausted byte intersection. + pub fn multipart_limits(&self) -> Result { + if self.operation != FileOperation::CreateMultipart { + return Err(FileAdmissionError::Scope); + } + let limits = MultipartLimits { + max_parts: 10_000, + max_part_bytes: self.request_bytes.min(self.file_bytes), + max_file_bytes: self.file_bytes, + max_staged_bytes: self.staged_bytes, + ttl_ms: 24 * 60 * 60 * 1000, + }; + limits.validate().map_err(|_| FileAdmissionError::Bounds)?; + Ok(limits) + } + + /// Intersects verified credentials with service and durable session bounds. + /// # Errors + /// Rejects wrong table, principal, operation, upload, expiry or missing credit. + pub fn authorize( + grant: &FileGrant, + request: &FileRequest, + service: FileServiceLimits, + session: Option<&MultipartSession>, + now_ms: u64, + ) -> Result { + service.validate()?; + if now_ms < grant.issued_ms || now_ms >= grant.expires_ms { + return Err(FileAdmissionError::Grant(FileGrantError::Expired)); + } + if !consistent(request) { + return Err(FileAdmissionError::Scope); + } + grant.authorize(request.operation, &request.location, 0, 0)?; + let mut request_bytes = grant.max_request_bytes.min(service.max_request_bytes); + let mut file_bytes = grant.max_file_bytes.min(service.max_file_bytes); + let mut staged_bytes = service.max_staged_bytes; + match &request.multipart { + None | Some(MultipartRequest::Create) if session.is_some() => { + return Err(FileAdmissionError::Scope); + } + None => {} + Some(MultipartRequest::Create) => { + request_bytes = request_bytes.min(service.max_part_bytes); + } + Some(multipart) => { + let session = session.ok_or(FileAdmissionError::Scope)?; + session.validate().map_err(|_| FileAdmissionError::State)?; + let upload_id = match multipart { + MultipartRequest::Upload { upload_id, .. } + | MultipartRequest::List { upload_id, .. } + | MultipartRequest::Complete { upload_id } + | MultipartRequest::Abort { upload_id } => upload_id, + MultipartRequest::Create => return Err(FileAdmissionError::State), + }; + if session.context != grant.context + || session.principal != grant.principal + || session.location != request.location + || upload_id != &session.upload.to_string() + { + return Err(FileAdmissionError::Scope); + } + if now_ms < session.created_ms || now_ms >= session.expires_ms { + return Err(FileAdmissionError::State); + } + if session.credit.map_or(true, |credit| credit.released) + && !(request.operation == FileOperation::CompleteMultipart + && session.phase == MultipartPhase::Published) + { + return Err(FileAdmissionError::State); + } + file_bytes = file_bytes.min(session.limits.max_file_bytes); + staged_bytes = staged_bytes.min(session.limits.max_staged_bytes); + if let MultipartRequest::Upload { part_number, .. } = multipart { + if session.phase != MultipartPhase::Open || *part_number > session.limits.max_parts { + return Err(FileAdmissionError::State); + } + request_bytes = request_bytes + .min(service.max_part_bytes) + .min(session.limits.max_part_bytes); + } + } + } + Ok(Self { + context: grant.context, + principal: grant.principal, + location: request.location.clone(), + operation: request.operation, + request_bytes, + file_bytes, + staged_bytes, + }) + } + + /// Preflights a new multipart reservation; the caller must still run durable + /// `MultipartAdmission::reserve` before exposing the upload ID. + /// # Errors + /// Rejects incompatible requested limits or exhausted admission snapshots. + pub fn check_create( + &self, + session: &MultipartSession, + policy: &MultipartAdmissionRecord, + ) -> Result<(), FileAdmissionError> { + if self.operation != FileOperation::CreateMultipart + || session.validate().is_err() + || policy.validate().is_err() + || session.phase != MultipartPhase::Open + || session.revision != 1 + || session.part_count != 0 + || session.pending.is_some() + || session.credit.is_some() + || session.context != self.context + || session.principal != self.principal + || session.location != self.location + || policy.context != self.context + || policy.pending.is_some() + { + return Err(FileAdmissionError::State); + } + if session.limits.max_part_bytes > self.request_bytes.min(self.file_bytes) + || session.limits.max_file_bytes > self.file_bytes + || session.limits.max_staged_bytes > self.staged_bytes + || policy.sessions >= policy.limits.max_sessions + || policy + .reserved_bytes + .checked_add(session.limits.max_staged_bytes) + .map_or(true, |total| total > policy.limits.max_reserved_bytes) + { + return Err(FileAdmissionError::Bounds); + } + Ok(()) + } + + /// Checks actual transferred bytes against all intersected limits. + /// # Errors + /// Rejects a stream exceeding either the request or complete-file ceiling. + pub fn check_bytes(&self, request_bytes: u64, file_bytes: u64) -> Result<(), FileAdmissionError> { + if request_bytes > self.request_bytes || file_bytes > self.file_bytes { + return Err(FileAdmissionError::Bounds); + } + Ok(()) + } + + pub(super) const fn request_byte_limit(&self) -> u64 { + self.request_bytes + } + + /// Receives a bounded immutable PUT or multipart part without publishing it. + /// # Errors + /// Rejects declared/actual size, digest, owner or storage failures. + pub async fn receive + Unpin>( + &self, + budget: &FileUploadBudget, + body: Input, + store: Arc, + owner: FileIdentity, + content_length: Option, + sha256: Option<[u8; 32]>, + ) -> Result { + if !matches!(self.operation, FileOperation::Put | FileOperation::UploadPart) + || owner.table != self.location.table() + { + return Err(FileAdmissionError::Scope); + } + if let Some(length) = content_length { + self.check_bytes(length, length)?; + } + let tree = budget + .receive( + body, + store, + owner, + FileUploadConstraints { + max_bytes: self.request_bytes.min(self.file_bytes), + content_length, + sha256, + }, + ) + .await?; + self.check_bytes(tree.length, tree.length)?; + Ok(tree) + } + + /// Opens an authorized range only after checking the complete file and response bytes. + /// # Errors + /// Rejects another location, excessive range/file bytes or reader admission failure. + pub fn read_body( + &self, + budget: &FileResponseBudget, + store: Arc, + record: FileRecord, + range: Option, + ) -> Result { + if self.operation != FileOperation::Get || record.location != self.location { + return Err(FileAdmissionError::Scope); + } + let bytes = range.map_or(record.length, |range| range.end.saturating_sub(range.start)); + self.check_bytes(bytes, record.length)?; + Ok(budget.body(store, record, range)?) + } +} + +fn consistent(request: &FileRequest) -> bool { + matches!( + (request.operation, request.multipart.as_ref()), + ( + FileOperation::Head | FileOperation::Get | FileOperation::Put, + None + ) | (FileOperation::CreateMultipart, Some(MultipartRequest::Create)) + | (FileOperation::UploadPart, Some(MultipartRequest::Upload { .. })) + | (FileOperation::ListParts, Some(MultipartRequest::List { .. })) + | ( + FileOperation::CompleteMultipart, + Some(MultipartRequest::Complete { .. }) + ) + | ( + FileOperation::AbortMultipart, + Some(MultipartRequest::Abort { .. }) + ) + ) +} diff --git a/app/crowdb-access-server/src/iceberg/file_auth.rs b/app/crowdb-access-server/src/iceberg/file_auth.rs new file mode 100644 index 000000000..694508ff5 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_auth.rs @@ -0,0 +1,135 @@ +use crowdb_access_iceberg::catalog::CatalogContext; +use crowdb_access_iceberg::file::{FileCredentials, FileGrant, FileGrantIssuer}; +use crowdb_access_s3::auth::StreamingPayloadVerifier; +use crowdb_access_s3::auth::{AuthError, Credential, CredentialProvider, RawAuthRequest, SigV4Verifier}; + +pub(super) fn authenticate_file_transfer( + issuer: &FileGrantIssuer, + context: CatalogContext, + request: RawAuthRequest<'_>, + region: &str, + now_ms: u64, +) -> Result<(FileGrant, Option), AuthError> { + if !request + .headers + .get("x-amz-content-sha256") + .is_some_and(|value| value.as_bytes().starts_with(b"STREAMING-")) + { + return authenticate_file_request(issuer, context, request, region, now_ms) + .map(|grant| (grant, None)); + } + validate_bounds(request)?; + let credentials = issuer + .verify_token(&session_token(request)?, context, now_ms) + .map_err(|_| AuthError::Rejected)?; + let verifier = SigV4Verifier::new(FileCredentialProvider(&credentials), region.to_owned(), 900) + .verify_streaming(request, now_ms / 1000)?; + Ok((credentials.grant().clone(), Some(verifier))) +} + +/// Authenticates native file credentials without consulting general S3 authority. +/// Callers must freshly validate the Ready context and authorize the returned grant +/// against the routed operation, location and actual streamed byte counts. +/// # Errors +/// Rejects oversized or ambiguous authentication, stale tokens and bad signatures. +pub fn authenticate_file_request( + issuer: &FileGrantIssuer, + context: CatalogContext, + request: RawAuthRequest<'_>, + region: &str, + now_ms: u64, +) -> Result { + validate_bounds(request)?; + let token = session_token(request)?; + let credentials = issuer + .verify_token(&token, context, now_ms) + .map_err(|_| AuthError::Rejected)?; + let provider = FileCredentialProvider(&credentials); + SigV4Verifier::new(provider, region.to_owned(), 900).verify(request, now_ms / 1000)?; + Ok(credentials.grant().clone()) +} + +struct FileCredentialProvider<'a>(&'a FileCredentials); + +impl CredentialProvider for FileCredentialProvider<'_> { + fn lookup(&self, access_key: &str) -> Option { + (access_key == self.0.access_key_id()).then(|| Credential { + secret_key: self.0.secret_access_key().as_bytes().to_vec(), + session_token: Some(self.0.session_token().to_owned()), + enabled: true, + }) + } +} + +fn validate_bounds(request: RawAuthRequest<'_>) -> Result<(), AuthError> { + let uri_bytes = request + .uri + .path_and_query() + .map_or(0, |value| value.as_str().len()) + + request.uri.authority().map_or(0, |value| value.as_str().len()) + + request.uri.scheme_str().map_or(0, str::len); + if uri_bytes > 8192 || request.headers.len() > 64 { + return Err(AuthError::Rejected); + } + let total = request + .headers + .iter() + .try_fold(0_usize, |size, (name, value)| { + size.checked_add(name.as_str().len())?.checked_add(value.len()) + }) + .ok_or(AuthError::Rejected)?; + if total > 16 * 1024 { + return Err(AuthError::Rejected); + } + for name in [ + "authorization", + "host", + "x-amz-date", + "x-amz-content-sha256", + "x-amz-security-token", + ] { + if request.headers.get_all(name).iter().count() > 1 { + return Err(AuthError::Rejected); + } + } + Ok(()) +} + +fn session_token(request: RawAuthRequest<'_>) -> Result { + let header_signed = request.headers.contains_key("authorization"); + let mut token = None; + let mut names = std::collections::BTreeSet::new(); + for part in request.uri.query().unwrap_or_default().split('&') { + let (name, value) = part.split_once('=').unwrap_or((part, "")); + let decoded = percent_encoding::percent_decode_str(name) + .decode_utf8() + .map_err(|_| AuthError::Rejected)?; + if !decoded.starts_with("X-Amz-") { + continue; + } + if header_signed || decoded != name || !names.insert(name) { + return Err(AuthError::Rejected); + } + if name == "X-Amz-Security-Token" { + token = Some( + percent_encoding::percent_decode_str(value) + .decode_utf8() + .map_err(|_| AuthError::Rejected)? + .into_owned(), + ); + } + } + if header_signed { + request + .headers + .get("x-amz-security-token") + .ok_or(AuthError::Rejected)? + .to_str() + .map(str::to_owned) + .map_err(|_| AuthError::Rejected) + } else if request.headers.contains_key("x-amz-security-token") { + Err(AuthError::Rejected) + } else { + token.ok_or(AuthError::Rejected) + } +} diff --git a/app/crowdb-access-server/src/iceberg/file_body.rs b/app/crowdb-access-server/src/iceberg/file_body.rs new file mode 100644 index 000000000..d5e16329c --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_body.rs @@ -0,0 +1,151 @@ +use std::future::Future; +use std::pin::Pin; +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; +use std::task::{Context, Poll}; + +use crowdb_access_iceberg::file::{ByteRange, FileBlockStore, FileIoError, FileReader, FileRecord}; +use hyper::body::{Body, Bytes, Frame, SizeHint}; + +#[derive(Debug, thiserror::Error)] +pub enum FileBodyError { + #[error(transparent)] + Storage(#[from] FileIoError), + #[error("file response capacity exhausted")] + Busy, +} + +pub struct FileResponseBudget { + active: Arc, + limit: usize, +} + +impl FileResponseBudget { + /// # Errors + /// Rejects empty or unbounded response concurrency limits. + pub fn new(limit: usize) -> Result { + if limit == 0 || limit > 64 { + return Err(FileIoError::Bounds); + } + Ok(Self { + active: Arc::new(AtomicUsize::new(0)), + limit, + }) + } + + #[must_use] + pub fn active(&self) -> usize { + self.active.load(Ordering::Acquire) + } + + /// Admits one pull response without fetching any file blocks. + /// # Errors + /// Rejects exhausted response capacity or invalid file records/ranges. + pub fn body( + &self, + store: Arc, + record: FileRecord, + range: Option, + ) -> Result { + self.active + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |active| { + (active < self.limit).then_some(active + 1) + }) + .map_err(|_| FileBodyError::Busy)?; + let permit = Permit(self.active.clone()); + let length = range.map_or(record.length, |range| range.end.saturating_sub(range.start)); + let reader = FileReader::new(store, record, range, 16 * 1024)?; + Ok(FileReadBody { + reader: (length > 0).then_some(reader), + pending: None, + remaining: length, + permit: (length > 0).then_some(permit), + }) + } +} + +struct Permit(Arc); +impl Drop for Permit { + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::AcqRel); + } +} + +type ReadFuture = Pin>, FileIoError>)> + Send>>; + +pub struct FileReadBody { + reader: Option, + pending: Option, + remaining: u64, + permit: Option, +} + +impl FileReadBody { + fn finish(&mut self) { + self.reader = None; + self.pending = None; + self.remaining = 0; + self.permit = None; + } +} + +impl Body for FileReadBody { + type Data = Bytes; + type Error = FileIoError; + + fn poll_frame( + self: Pin<&mut Self>, + context: &mut Context<'_>, + ) -> Poll, FileIoError>>> { + let body = self.get_mut(); + if body.remaining == 0 { + return Poll::Ready(None); + } + if body.pending.is_none() { + let Some(mut reader) = body.reader.take() else { + body.finish(); + return Poll::Ready(Some(Err(FileIoError::Finished))); + }; + body.pending = Some(Box::pin(async move { + let result = reader.next().await; + (reader, result) + })); + } + let (reader, result) = match body + .pending + .as_mut() + .expect("pending read is initialized") + .as_mut() + .poll(context) + { + Poll::Pending => return Poll::Pending, + Poll::Ready(value) => value, + }; + body.pending = None; + match result { + Ok(Some(bytes)) if !bytes.is_empty() && bytes.len() as u64 <= body.remaining => { + body.remaining -= bytes.len() as u64; + if body.remaining == 0 { + body.finish(); + } else { + body.reader = Some(reader); + } + Poll::Ready(Some(Ok(Frame::data(Bytes::from(bytes))))) + } + result => { + body.finish(); + Poll::Ready(Some(Err(result.err().unwrap_or(FileIoError::Bounds)))) + } + } + } + + fn is_end_stream(&self) -> bool { + self.remaining == 0 + } + + fn size_hint(&self) -> SizeHint { + SizeHint::with_exact(self.remaining) + } +} diff --git a/app/crowdb-access-server/src/iceberg/file_complete.rs b/app/crowdb-access-server/src/iceberg/file_complete.rs new file mode 100644 index 000000000..7892dc56e --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_complete.rs @@ -0,0 +1,110 @@ +use std::convert::Infallible; +use std::future::Future; +use std::pin::Pin; +use std::task::{Context, Poll}; +use std::time::Duration; + +use crowdb_access_iceberg::key::OperationId; +use hyper::body::{Body, Bytes, Frame, SizeHint}; +use tokio::time::{Instant, Sleep}; + +use super::file_response::{FileResponseError, FileS3ErrorCode, MultipartResponses}; + +const XML_PREFIX: &[u8] = b""; +type Completion = Pin, FileS3ErrorCode>> + Send>>; + +pub struct FileCompleteBody { + completion: Option, + heartbeat: Pin>, + deadline: Pin>, + interval: Duration, + resource: String, + started: bool, +} + +impl FileCompleteBody { + /// # Errors + /// Rejects unbounded resource names and invalid heartbeat or work limits. + pub fn new( + completion: impl Future, FileS3ErrorCode>> + Send + 'static, + resource: &str, + interval: Duration, + timeout: Duration, + ) -> Result { + if resource.len() > 2048 + || interval.is_zero() + || interval > Duration::from_secs(30) + || timeout <= interval + || timeout > Duration::from_secs(300) + { + return Err(FileResponseError::Invalid); + } + Ok(Self { + completion: Some(Box::pin(completion)), + heartbeat: Box::pin(tokio::time::sleep(interval)), + deadline: Box::pin(tokio::time::sleep(timeout)), + interval, + resource: resource.to_owned(), + started: false, + }) + } + + fn finish(&mut self, result: Result, FileS3ErrorCode>) -> Bytes { + self.completion = None; + let bytes = result.unwrap_or_else(|code| { + MultipartResponses::error(code, &self.resource, &OperationId::random().to_string()) + .expect("validated resource and fixed-size request ID") + .into_body() + }); + let bytes = Bytes::from(bytes); + if bytes.starts_with(XML_PREFIX) { + bytes.slice(XML_PREFIX.len()..) + } else { + bytes + } + } +} + +impl Body for FileCompleteBody { + type Data = Bytes; + type Error = Infallible; + + fn poll_frame( + self: Pin<&mut Self>, + context: &mut Context<'_>, + ) -> Poll, Self::Error>>> { + let body = self.get_mut(); + if body.completion.is_none() { + return Poll::Ready(None); + } + if !body.started { + body.started = true; + return Poll::Ready(Some(Ok(Frame::data(Bytes::from_static(XML_PREFIX))))); + } + let result = if body.deadline.as_mut().poll(context).is_ready() { + Poll::Ready(Err(FileS3ErrorCode::SlowDown)) + } else { + body.completion.as_mut().unwrap().as_mut().poll(context) + }; + if let Poll::Ready(result) = result { + return Poll::Ready(Some(Ok(Frame::data(body.finish(result))))); + } + if body.heartbeat.as_mut().poll(context).is_ready() { + body.heartbeat.as_mut().reset(Instant::now() + body.interval); + return Poll::Ready(Some(Ok(Frame::data(Bytes::from_static(b"\n"))))); + } + Poll::Pending + } + + fn is_end_stream(&self) -> bool { + self.completion.is_none() + } + + fn size_hint(&self) -> SizeHint { + if self.is_end_stream() { + SizeHint::with_exact(0) + } else { + SizeHint::default() + } + } +} diff --git a/app/crowdb-access-server/src/iceberg/file_encoding.rs b/app/crowdb-access-server/src/iceberg/file_encoding.rs new file mode 100644 index 000000000..3b44a86a4 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_encoding.rs @@ -0,0 +1,224 @@ +use std::pin::Pin; +use std::task::{Context, Poll}; + +use crowdb_access_s3::auth::StreamingPayloadVerifier; +use hyper::body::{Body, Bytes, Frame}; +use hyper::HeaderMap; + +mod checksum; +mod chunks; +mod content_md5; + +#[derive(Clone, Copy, Debug, thiserror::Error)] +pub enum FileEncodingError { + #[error("invalid upload framing")] + Framing, + #[error("upload length exceeds its declared bounds")] + Length, + #[error("upload chunk signature is invalid")] + Signature, + #[error("upload checksum is invalid")] + Checksum, + #[error("upload transport failed")] + Transport, +} + +pub struct FileUploadBody { + input: Input, + buffered: Bytes, + chunks: Option, + checksum: Option, + content_md5: Option, + length: Option, + wire_length: Option, + wire_bytes: u64, + max_wire_bytes: u64, + done: bool, + failure: Option, +} + +impl FileUploadBody { + /// The streaming verifier must come from authenticating these exact headers. + /// Returned data is staging input; only successful EOF authorizes publication. + /// # Errors + /// Rejects ambiguous framing, unsupported checksums and excessive encoded lengths. + pub fn new( + input: Input, + headers: &HeaderMap, + verifier: Option, + max_wire_bytes: u64, + ) -> Result { + let wire_length = length_header(headers, "content-length")?; + if wire_length.is_some_and(|length| length > max_wire_bytes) { + return Err(FileEncodingError::Length); + } + let (length, chunks, checksum) = if let Some(verifier) = verifier { + if header(headers, "content-encoding")? != Some("aws-chunked") { + return Err(FileEncodingError::Framing); + } + let length = + length_header(headers, "x-amz-decoded-content-length")?.ok_or(FileEncodingError::Framing)?; + if length > max_wire_bytes { + return Err(FileEncodingError::Length); + } + if !verifier.has_trailer() && headers.contains_key("x-amz-trailer") { + return Err(FileEncodingError::Framing); + } + let checksum = checksum::Checksum::from_headers(headers, verifier.has_trailer())?; + ( + Some(length), + Some(chunks::Chunks::new(verifier, checksum, length)), + None, + ) + } else { + if headers.contains_key("x-amz-decoded-content-length") + || headers.contains_key("x-amz-trailer") + || header(headers, "content-encoding")? + .is_some_and(|value| value.split(',').any(|encoding| encoding.trim() == "aws-chunked")) + { + return Err(FileEncodingError::Framing); + } + ( + wire_length, + None, + checksum::Checksum::from_headers(headers, false)?, + ) + }; + Ok(Self { + input, + buffered: Bytes::new(), + chunks, + checksum, + content_md5: content_md5::ContentMd5::from_headers(headers)?, + length, + wire_length, + wire_bytes: 0, + max_wire_bytes, + done: false, + failure: None, + }) + } + + #[must_use] + pub const fn decoded_length(&self) -> Option { + self.length + } + + pub(super) const fn failure(&self) -> Option { + self.failure + } + + fn finish(&self) -> Result<(), FileEncodingError> { + if self.wire_length.is_some_and(|length| length != self.wire_bytes) { + return Err(FileEncodingError::Length); + } + if let Some(chunks) = &self.chunks { + chunks.finish()?; + } + if let Some(checksum) = &self.checksum { + checksum.verify()?; + } + if let Some(checksum) = &self.content_md5 { + checksum.verify()?; + } + Ok(()) + } +} + +impl + Unpin> FileUploadBody { + fn poll_data(&mut self, context: &mut Context<'_>) -> Poll, FileEncodingError>> { + for _ in 0..64 { + if !self.buffered.is_empty() { + if let Some(chunks) = &mut self.chunks { + if let Some(bytes) = chunks.next(&mut self.buffered)? { + return Poll::Ready(Ok(Some(bytes))); + } + } else { + let bytes = self.buffered.split_to(self.buffered.len().min(64 * 1024)); + if let Some(checksum) = &mut self.checksum { + checksum.update(&bytes); + } + return Poll::Ready(Ok(Some(bytes))); + } + } + match std::task::ready!(Pin::new(&mut self.input).poll_frame(context)) { + Some(Ok(frame)) => { + self.buffered = frame.into_data().map_err(|_| FileEncodingError::Framing)?; + super::metrics::record_request_bytes(self.buffered.len()); + self.wire_bytes = self + .wire_bytes + .checked_add(self.buffered.len() as u64) + .ok_or(FileEncodingError::Length)?; + if self.wire_bytes > self.max_wire_bytes { + return Poll::Ready(Err(FileEncodingError::Length)); + } + } + Some(Err(_)) => return Poll::Ready(Err(FileEncodingError::Transport)), + None => { + self.finish()?; + return Poll::Ready(Ok(None)); + } + } + } + context.waker().wake_by_ref(); + Poll::Pending + } +} + +impl + Unpin> Body for FileUploadBody { + type Data = Bytes; + type Error = FileEncodingError; + + fn poll_frame( + self: Pin<&mut Self>, + context: &mut Context<'_>, + ) -> Poll, Self::Error>>> { + let body = self.get_mut(); + if body.done { + return Poll::Ready(None); + } + match std::task::ready!(body.poll_data(context)) { + Ok(Some(bytes)) => { + if let Some(checksum) = &mut body.content_md5 { + checksum.update(&bytes); + } + Poll::Ready(Some(Ok(Frame::data(bytes)))) + } + Ok(None) => { + body.done = true; + Poll::Ready(None) + } + Err(error) => { + body.done = true; + body.failure = Some(error); + body.buffered = Bytes::new(); + Poll::Ready(Some(Err(error))) + } + } + } + + fn is_end_stream(&self) -> bool { + self.done + } +} + +fn header<'a>(headers: &'a HeaderMap, name: &str) -> Result, FileEncodingError> { + if headers.get_all(name).iter().count() > 1 { + return Err(FileEncodingError::Framing); + } + headers + .get(name) + .map(|value| value.to_str().map_err(|_| FileEncodingError::Framing)) + .transpose() +} + +fn length_header(headers: &HeaderMap, name: &str) -> Result, FileEncodingError> { + header(headers, name)? + .map(|value| { + if value.is_empty() || !value.bytes().all(|byte| byte.is_ascii_digit()) { + return Err(FileEncodingError::Framing); + } + value.parse().map_err(|_| FileEncodingError::Length) + }) + .transpose() +} diff --git a/app/crowdb-access-server/src/iceberg/file_encoding/checksum.rs b/app/crowdb-access-server/src/iceberg/file_encoding/checksum.rs new file mode 100644 index 000000000..3dec60253 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_encoding/checksum.rs @@ -0,0 +1,112 @@ +use base64::{engine::general_purpose::STANDARD, Engine}; +use hyper::HeaderMap; +use sha1::Sha1; +use sha2::{Digest, Sha256}; + +use super::FileEncodingError; + +const CRC32: crc::Crc = crc::Crc::::new(&crc::CRC_32_ISO_HDLC); +const CRC32C: crc::Crc = crc::Crc::::new(&crc::CRC_32_ISCSI); +const CRC64: crc::Crc = crc::Crc::::new(&crc::CRC_64_NVME); +const NAMES: [&str; 5] = ["crc32", "crc32c", "crc64nvme", "sha1", "sha256"]; + +#[derive(Clone)] +enum DigestState { + Crc32(crc::Digest<'static, u32>), + Crc64(crc::Digest<'static, u64>), + Sha1(Sha1), + Sha256(Sha256), +} + +pub(super) struct Checksum { + name: String, + state: DigestState, + expected: Option, +} + +impl Checksum { + pub(super) fn from_headers( + headers: &HeaderMap, + trailer: bool, + ) -> Result, FileEncodingError> { + for name in headers.keys() { + if let Some(algorithm) = name.as_str().strip_prefix("x-amz-checksum-") { + if !NAMES.contains(&algorithm) && algorithm != "type" { + return Err(FileEncodingError::Framing); + } + } + } + let mut selected = None; + for name in NAMES { + let header = format!("x-amz-checksum-{name}"); + if let Some(value) = super::header(headers, &header)? { + if selected.is_some() || trailer { + return Err(FileEncodingError::Framing); + } + selected = Some(Self::new(&header, Some(value.to_owned()))?); + } + } + if trailer { + selected = Some(Self::new( + super::header(headers, "x-amz-trailer")?.ok_or(FileEncodingError::Framing)?, + None, + )?); + } + if let Some(algorithm) = super::header(headers, "x-amz-sdk-checksum-algorithm")? { + if selected.as_ref().map_or(true, |selected| { + selected.name != format!("x-amz-checksum-{}", algorithm.to_ascii_lowercase()) + }) { + return Err(FileEncodingError::Framing); + } + } + Ok(selected) + } + + fn new(name: &str, expected: Option) -> Result { + let state = match name { + "x-amz-checksum-crc32" => DigestState::Crc32(CRC32.digest()), + "x-amz-checksum-crc32c" => DigestState::Crc32(CRC32C.digest()), + "x-amz-checksum-crc64nvme" => DigestState::Crc64(CRC64.digest()), + "x-amz-checksum-sha1" => DigestState::Sha1(Sha1::new()), + "x-amz-checksum-sha256" => DigestState::Sha256(Sha256::new()), + _ => return Err(FileEncodingError::Framing), + }; + Ok(Self { + name: name.to_owned(), + state, + expected, + }) + } + + pub(super) fn update(&mut self, bytes: &[u8]) { + match &mut self.state { + DigestState::Crc32(digest) => digest.update(bytes), + DigestState::Crc64(digest) => digest.update(bytes), + DigestState::Sha1(digest) => digest.update(bytes), + DigestState::Sha256(digest) => digest.update(bytes), + } + } + + pub(super) fn trailer(&mut self, line: &str) -> Result<(), FileEncodingError> { + let (name, value) = line.split_once(':').ok_or(FileEncodingError::Framing)?; + if name != self.name || self.expected.is_some() { + return Err(FileEncodingError::Framing); + } + self.expected = Some(value.to_owned()); + self.verify() + } + + pub(super) fn verify(&self) -> Result<(), FileEncodingError> { + let value = match self.state.clone() { + DigestState::Crc32(digest) => STANDARD.encode(digest.finalize().to_be_bytes()), + DigestState::Crc64(digest) => STANDARD.encode(digest.finalize().to_be_bytes()), + DigestState::Sha1(digest) => STANDARD.encode(digest.finalize()), + DigestState::Sha256(digest) => STANDARD.encode(digest.finalize()), + }; + if self.expected.as_deref() == Some(value.as_str()) { + Ok(()) + } else { + Err(FileEncodingError::Checksum) + } + } +} diff --git a/app/crowdb-access-server/src/iceberg/file_encoding/chunks.rs b/app/crowdb-access-server/src/iceberg/file_encoding/chunks.rs new file mode 100644 index 000000000..9ea22be47 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_encoding/chunks.rs @@ -0,0 +1,165 @@ +use crowdb_access_s3::auth::StreamingPayloadVerifier; +use hyper::body::Bytes; +use sha2::{Digest, Sha256}; + +use super::{checksum::Checksum, FileEncodingError}; + +enum State { + Header, + Data(u64), + Separator, + Checksum, + Signature, + End, + Done, +} + +pub(super) struct Chunks { + verifier: StreamingPayloadVerifier, + state: State, + line: Vec, + signature: Option, + hash: Sha256, + checksum: Option, + canonical_trailer: String, + remaining: u64, +} + +impl Chunks { + pub(super) fn new(verifier: StreamingPayloadVerifier, checksum: Option, length: u64) -> Self { + Self { + verifier, + state: State::Header, + line: Vec::new(), + signature: None, + hash: Sha256::new(), + checksum, + canonical_trailer: String::new(), + remaining: length, + } + } + + pub(super) fn next(&mut self, input: &mut Bytes) -> Result, FileEncodingError> { + while !input.is_empty() { + if let State::Data(remaining) = self.state { + let length = input + .len() + .min(64 * 1024) + .min(usize::try_from(remaining).unwrap_or(usize::MAX)); + let bytes = input.split_to(length); + self.hash.update(&bytes); + if let Some(checksum) = &mut self.checksum { + checksum.update(&bytes); + } + self.remaining -= length as u64; + let remaining = remaining - length as u64; + self.state = if remaining == 0 { + self.verify_chunk()?; + State::Separator + } else { + State::Data(remaining) + }; + return Ok(Some(bytes)); + } + if matches!(self.state, State::Done) { + return Err(FileEncodingError::Framing); + } + let byte = input.split_to(1)[0]; + self.line.push(byte); + if self.line.len() > 1024 { + return Err(FileEncodingError::Framing); + } + if byte == b'\n' { + let line = std::mem::take(&mut self.line); + let line = line.strip_suffix(b"\r\n").ok_or(FileEncodingError::Framing)?; + let line = std::str::from_utf8(line).map_err(|_| FileEncodingError::Framing)?; + self.line(line)?; + } + } + Ok(None) + } + + fn line(&mut self, line: &str) -> Result<(), FileEncodingError> { + match self.state { + State::Header => self.start_chunk(line)?, + State::Separator if line.is_empty() => self.state = State::Header, + State::Checksum => { + self.checksum + .as_mut() + .ok_or(FileEncodingError::Framing)? + .trailer(line)?; + self.canonical_trailer = format!("{line}\n"); + self.state = if self.verifier.is_signed() { + State::Signature + } else { + State::End + }; + } + State::Signature => { + let signature = line + .strip_prefix("x-amz-trailer-signature:") + .ok_or(FileEncodingError::Framing)?; + self.verifier + .verify_trailer(&self.canonical_trailer, Some(signature)) + .map_err(|_| FileEncodingError::Signature)?; + self.state = State::End; + } + State::End if line.is_empty() => self.state = State::Done, + _ => return Err(FileEncodingError::Framing), + } + Ok(()) + } + + fn start_chunk(&mut self, line: &str) -> Result<(), FileEncodingError> { + let (length, signature) = if self.verifier.is_signed() { + let (length, signature) = line + .split_once(";chunk-signature=") + .ok_or(FileEncodingError::Framing)?; + if signature.len() != 64 || !signature.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(FileEncodingError::Framing); + } + (length, Some(signature.to_owned())) + } else { + (line, None) + }; + if length.is_empty() || length.len() > 16 || !length.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(FileEncodingError::Framing); + } + let length = u64::from_str_radix(length, 16).map_err(|_| FileEncodingError::Framing)?; + if length > self.remaining { + return Err(FileEncodingError::Length); + } + self.signature = signature; + if length == 0 { + if self.remaining != 0 { + return Err(FileEncodingError::Length); + } + self.verify_chunk()?; + self.state = if self.verifier.has_trailer() { + State::Checksum + } else { + State::End + }; + } else { + self.state = State::Data(length); + } + Ok(()) + } + + fn verify_chunk(&mut self) -> Result<(), FileEncodingError> { + let digest = std::mem::take(&mut self.hash).finalize().into(); + self.verifier + .verify_chunk(digest, self.signature.as_deref()) + .map_err(|_| FileEncodingError::Signature) + } + + pub(super) fn finish(&self) -> Result<(), FileEncodingError> { + if !matches!(self.state, State::Done) || !self.line.is_empty() || self.remaining != 0 { + return Err(FileEncodingError::Framing); + } + if let Some(checksum) = &self.checksum { + checksum.verify()?; + } + Ok(()) + } +} diff --git a/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs b/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs new file mode 100644 index 000000000..ec8465249 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs @@ -0,0 +1,38 @@ +use base64::{engine::general_purpose::STANDARD, Engine}; +use hyper::HeaderMap; +use md5::{Digest, Md5}; + +use super::FileEncodingError; + +pub(super) struct ContentMd5 { + expected: [u8; 16], + digest: Md5, +} + +impl ContentMd5 { + pub(super) fn from_headers(headers: &HeaderMap) -> Result, FileEncodingError> { + let Some(value) = super::header(headers, "content-md5")? else { + return Ok(None); + }; + let expected = STANDARD + .decode(value) + .map_err(|_| FileEncodingError::Framing)? + .try_into() + .map_err(|_| FileEncodingError::Framing)?; + Ok(Some(Self { + expected, + digest: Md5::new(), + })) + } + + pub(super) fn update(&mut self, bytes: &[u8]) { + self.digest.update(bytes); + } + + pub(super) fn verify(&self) -> Result<(), FileEncodingError> { + if <[u8; 16]>::from(self.digest.clone().finalize()) != self.expected { + return Err(FileEncodingError::Checksum); + } + Ok(()) + } +} diff --git a/app/crowdb-access-server/src/iceberg/file_http.rs b/app/crowdb-access-server/src/iceberg/file_http.rs new file mode 100644 index 000000000..f1de58464 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_http.rs @@ -0,0 +1,365 @@ +use std::fmt::Write; +use std::sync::Arc; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use crowdb_access_iceberg::catalog::{CatalogError, CatalogLifecycle, CatalogRepository, RootState}; +use crowdb_access_iceberg::file::{ + resolve_range, FileBlockStore, FileGrantError, FileGrantIssuer, FileOperation, FileRecord, + FileRepository, FileSealer, MultipartAdmission, MultipartLister, MultipartPartStore, MultipartRepository, + RangeError, +}; +use crowdb_access_iceberg::key::OperationId; +use crowdb_access_s3::auth::{RawAuthRequest, StreamingPayloadVerifier}; +use hyper::body::Incoming; +use hyper::http::header::{ACCEPT_RANGES, CONTENT_LENGTH, CONTENT_RANGE, ETAG, RANGE}; +use hyper::{Method, Request, Response, StatusCode}; + +use super::body::IcebergBody; +use super::file_admission::{FileAdmissionError, FileServiceLimits, FileTransferAdmission}; +use super::file_auth::authenticate_file_transfer; +use super::file_body::FileResponseBudget; +use super::file_encoding::FileUploadBody; +use super::file_request::{FileRequest, FileRequestError}; +use super::file_response::{FileS3ErrorCode, MultipartResponses}; +use super::file_upload::FileUploadBudget; + +mod multipart; + +pub(super) struct FileHttp { + pins: crowdb_access_iceberg::gc::ReaderPins, + repository: FileRepository, + multipart: MultipartRepository, + admission: MultipartAdmission, + lister: MultipartLister, + blocks: Arc, + issuer: FileGrantIssuer, + responses: FileResponseBudget, + uploads: FileUploadBudget, + region: String, + limits: FileServiceLimits, +} + +impl FileHttp { + pub(super) fn new( + store: Arc, + blocks: Arc, + secret: [u8; 32], + region: String, + ) -> Result { + if region.is_empty() + || region.len() > 64 + || !region + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || byte == b'-') + { + return Err(FileGrantError::Invalid); + } + Ok(Self { + pins: crowdb_access_iceberg::gc::ReaderPins::new(store.clone()), + repository: FileRepository::new(store.clone()), + multipart: MultipartRepository::new(store.clone()), + admission: MultipartAdmission::new(store.clone()), + lister: MultipartLister::new(store), + blocks, + issuer: FileGrantIssuer::new(secret, 15 * 60 * 1000)?, + responses: FileResponseBudget::new(64).map_err(|_| FileGrantError::Invalid)?, + uploads: FileUploadBudget::new(64).map_err(|_| FileGrantError::Invalid)?, + region, + limits: FileServiceLimits { + max_request_bytes: 1024 * 1024 * 1024, + max_file_bytes: 1024 * 1024 * 1024 * 1024, + max_part_bytes: 1024 * 1024 * 1024, + max_staged_bytes: 1024 * 1024 * 1024 * 1024, + }, + }) + } + + pub(super) async fn dispatch( + self: &Arc, + catalog: &CatalogRepository, + request: Request, + request_timeout: Duration, + ) -> Response { + let path = request.uri().path().to_owned(); + match Box::pin(self.execute(catalog, request, request_timeout)).await { + Ok(response) => response, + Err(code) => { + tracing::debug!(?code, %path, "native file request rejected"); + s3_error(code, &path) + } + } + } + + async fn execute( + self: &Arc, + catalog: &CatalogRepository, + request: Request, + request_timeout: Duration, + ) -> Result, FileS3ErrorCode> { + let file_request = FileRequest::parse(request.method(), request.uri()).map_err(request_error)?; + let (root, authority) = catalog.status().await.map_err(catalog_error)?; + if root.state != RootState::Ready + || authority.lifecycle != CatalogLifecycle::Ready + || request_timeout.is_zero() + || request_timeout > Duration::from_millis(authority.admission_bounds.request_ms) + { + return Err(FileS3ErrorCode::SlowDown); + } + let now_ms = now_ms()?; + let (grant, streaming) = authenticate_file_transfer( + &self.issuer, + root.context, + RawAuthRequest::from_parts(request.method(), request.uri(), request.headers()), + &self.region, + now_ms, + ) + .map_err(|_| FileS3ErrorCode::AccessDenied)?; + grant + .authorize(file_request.operation, &file_request.location, 0, 0) + .map_err(|_| FileS3ErrorCode::AccessDenied)?; + let expires_ms = self + .pins + .request_expiry(root.context, now_ms) + .await + .map_err(catalog_error)?; + if matches!(file_request.operation, FileOperation::Head | FileOperation::Get) { + self.pins + .protect_file_reads( + root.context, + file_request.location.table().table, + "file-request", + expires_ms, + now_ms, + ) + .await + .map_err(catalog_error)?; + } else { + self.pins + .protect_files( + root.context, + file_request.location.table().table, + "file-request", + expires_ms, + now_ms, + ) + .await + .map_err(catalog_error)?; + } + let session = self.load_session(root.context, &file_request).await?; + let admission = + FileTransferAdmission::authorize(&grant, &file_request, self.limits, session.as_ref(), now_ms) + .map_err(admission_error)?; + match file_request.operation { + FileOperation::Head | FileOperation::Get => { + self.read(&file_request, &request, root.context, &admission).await + } + FileOperation::Put => { + self.put(&file_request, request, root.context, &admission, streaming) + .await + } + _ => { + Box::pin(self.multipart_request( + &file_request, + request, + session, + &admission, + now_ms, + streaming, + )) + .await + } + } + } + + async fn put( + &self, + file_request: &FileRequest, + request: Request, + context: crowdb_access_iceberg::catalog::CatalogContext, + admission: &FileTransferAdmission, + streaming: Option, + ) -> Result, FileS3ErrorCode> { + let digest = if streaming.is_some() { + None + } else { + multipart::signed_digest(request.headers().get("x-amz-content-sha256"))? + }; + let (parts, body) = request.into_parts(); + let mut body = FileUploadBody::new(body, &parts.headers, streaming, admission.request_byte_limit()) + .map_err(multipart::encoding_error)?; + let length = body.decoded_length(); + let owner = crowdb_access_iceberg::file::FileIdentity { + table: file_request.location.table(), + file: crowdb_access_iceberg::key::FileId::random(), + }; + let tree = admission + .receive( + &self.uploads, + &mut body, + self.blocks.clone(), + owner, + length, + digest, + ) + .await + .map_err(|error| { + body.failure() + .map_or_else(|| admission_error(error), multipart::encoding_error) + })?; + let sealed = FileSealer::new(self.blocks.clone(), self.limits.max_file_bytes) + .map_err(|_| FileS3ErrorCode::InternalError)? + .seal_uploaded(owner, file_request.location.clone(), tree) + .await + .map_err(multipart::seal_error)?; + let published = self + .repository + .publish(context, &sealed) + .await + .map_err(catalog_error)?; + let mut response = Response::new(IcebergBody::new(Vec::new())); + set_header(&mut response, ETAG, &etag(&published))?; + Ok(response) + } + + async fn read( + &self, + file_request: &FileRequest, + request: &Request, + context: crowdb_access_iceberg::catalog::CatalogContext, + admission: &FileTransferAdmission, + ) -> Result, FileS3ErrorCode> { + let record = self + .repository + .load(context, &file_request.location) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::NoSuchKey)?; + let range = match request + .headers() + .get_all(RANGE) + .iter() + .collect::>() + .as_slice() + { + [] => None, + [value] => Some(value.to_str().map_err(|_| FileS3ErrorCode::InvalidRange)?), + _ => return Err(FileS3ErrorCode::InvalidRange), + }; + let range = resolve_range(range, record.length).map_err(range_error)?; + let bytes = range.map_or(record.length, |range| range.end - range.start); + let body = if request.method() == Method::HEAD { + admission.check_bytes(0, record.length).map_err(admission_error)?; + IcebergBody::new(Vec::new()) + } else { + IcebergBody::file( + admission + .read_body(&self.responses, self.blocks.clone(), record.clone(), range) + .map_err(admission_error)?, + ) + }; + let mut response = Response::new(body); + *response.status_mut() = if range.is_some() { + StatusCode::PARTIAL_CONTENT + } else { + StatusCode::OK + }; + set_header(&mut response, CONTENT_LENGTH, &bytes.to_string())?; + set_header(&mut response, ETAG, &etag(&record))?; + response + .headers_mut() + .insert(ACCEPT_RANGES, hyper::header::HeaderValue::from_static("bytes")); + if let Some(range) = range { + set_header( + &mut response, + CONTENT_RANGE, + &format!("bytes {}-{}/{}", range.start, range.end - 1, record.length), + )?; + } + Ok(response) + } +} + +fn set_header( + response: &mut Response, + name: hyper::header::HeaderName, + value: &str, +) -> Result<(), FileS3ErrorCode> { + response.headers_mut().insert( + name, + hyper::header::HeaderValue::from_str(value).map_err(|_| FileS3ErrorCode::InternalError)?, + ); + Ok(()) +} + +fn etag(record: &FileRecord) -> String { + let mut value = String::from("\""); + for byte in record.digest { + write!(value, "{byte:02x}").expect("string writes do not fail"); + } + value.push('"'); + value +} + +fn now_ms() -> Result { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .ok() + .and_then(|elapsed| u64::try_from(elapsed.as_millis()).ok()) + .ok_or(FileS3ErrorCode::InternalError) +} + +fn request_error(error: FileRequestError) -> FileS3ErrorCode { + match error { + FileRequestError::Invalid | FileRequestError::Unsupported => FileS3ErrorCode::InvalidRequest, + } +} + +#[allow(clippy::needless_pass_by_value)] +fn catalog_error(error: CatalogError) -> FileS3ErrorCode { + match error { + CatalogError::Forbidden => FileS3ErrorCode::AccessDenied, + CatalogError::Conflict => FileS3ErrorCode::Conflict, + CatalogError::Busy | CatalogError::Uninitialized => FileS3ErrorCode::SlowDown, + CatalogError::Invalid(_) | CatalogError::Store(_) => FileS3ErrorCode::InternalError, + } +} + +#[allow(clippy::needless_pass_by_value)] +fn admission_error(error: FileAdmissionError) -> FileS3ErrorCode { + match error { + FileAdmissionError::Grant(FileGrantError::Bounds) + | FileAdmissionError::Bounds + | FileAdmissionError::Upload(super::file_upload::FileUploadError::Bounds) => { + FileS3ErrorCode::EntityTooLarge + } + FileAdmissionError::Scope | FileAdmissionError::Grant(_) => FileS3ErrorCode::AccessDenied, + FileAdmissionError::State | FileAdmissionError::Read(super::file_body::FileBodyError::Busy) => { + FileS3ErrorCode::SlowDown + } + FileAdmissionError::Upload(super::file_upload::FileUploadError::Digest) => FileS3ErrorCode::BadDigest, + FileAdmissionError::Upload( + super::file_upload::FileUploadError::Length | super::file_upload::FileUploadError::Trailers, + ) => FileS3ErrorCode::InvalidRequest, + FileAdmissionError::Upload(super::file_upload::FileUploadError::Busy) => FileS3ErrorCode::SlowDown, + FileAdmissionError::Read(_) | FileAdmissionError::Upload(_) => FileS3ErrorCode::InternalError, + } +} + +fn range_error(error: RangeError) -> FileS3ErrorCode { + match error { + RangeError::Invalid | RangeError::Multiple | RangeError::Unsatisfiable => { + FileS3ErrorCode::InvalidRange + } + } +} + +pub(super) fn unavailable(path: &str) -> Response { + s3_error(FileS3ErrorCode::SlowDown, path) +} + +fn s3_error(code: FileS3ErrorCode, path: &str) -> Response { + let resource = if path.len() <= 2048 { path } else { "/" }; + MultipartResponses::error(code, resource, &OperationId::random().to_string()) + .expect("bounded S3 error fields") + .map(IcebergBody::new) +} diff --git a/app/crowdb-access-server/src/iceberg/file_http/multipart.rs b/app/crowdb-access-server/src/iceberg/file_http/multipart.rs new file mode 100644 index 000000000..962003acf --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_http/multipart.rs @@ -0,0 +1,512 @@ +use std::sync::Arc; + +use crowdb_access_iceberg::catalog::{CatalogContext, CatalogError}; +use crowdb_access_iceberg::file::{ + FileIdentity, FileOperation, FileSealError, FileSealer, MultipartAdmissionLimits, MultipartPart, + MultipartPhase, MultipartSession, MultipartWorkError, +}; +use crowdb_access_iceberg::key::{FileId, OperationId}; +use crowdb_access_s3::auth::StreamingPayloadVerifier; +use http_body_util::BodyExt; +use hyper::body::Incoming; +use hyper::http::header::HeaderValue; +use hyper::{Request, Response}; +use sha2::{Digest, Sha256}; + +use super::{admission_error, catalog_error, FileHttp, FileS3ErrorCode, FileTransferAdmission}; +use crate::iceberg::body::IcebergBody; +use crate::iceberg::file_request::{FileRequest, MultipartRequest}; +use crate::iceberg::file_response::MultipartResponses; +use crate::iceberg::{FileEncodingError, FileUploadBody}; + +impl FileHttp { + pub(super) async fn load_session( + &self, + context: CatalogContext, + request: &FileRequest, + ) -> Result, FileS3ErrorCode> { + let upload_id = match &request.multipart { + None | Some(MultipartRequest::Create) => return Ok(None), + Some( + MultipartRequest::Upload { upload_id, .. } + | MultipartRequest::List { upload_id, .. } + | MultipartRequest::Complete { upload_id } + | MultipartRequest::Abort { upload_id }, + ) => upload_id, + }; + let upload = upload_id + .parse::() + .map_err(|_| FileS3ErrorCode::NoSuchUpload)?; + self.multipart + .load(context, upload) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::NoSuchUpload) + .map(Some) + } + + pub(super) async fn multipart_request( + self: &Arc, + file: &FileRequest, + request: Request, + session: Option, + admission: &FileTransferAdmission, + now_ms: u64, + streaming: Option, + ) -> Result, FileS3ErrorCode> { + match (&file.multipart, file.operation) { + (Some(MultipartRequest::Create), FileOperation::CreateMultipart) => { + self.create(file, admission, now_ms).await + } + (Some(MultipartRequest::Upload { part_number, .. }), FileOperation::UploadPart) => { + self.upload( + session.ok_or(FileS3ErrorCode::NoSuchUpload)?, + *part_number, + request, + admission, + now_ms, + streaming, + ) + .await + } + ( + Some(MultipartRequest::List { + marker, max_parts, .. + }), + FileOperation::ListParts, + ) => { + let session = session.ok_or(FileS3ErrorCode::NoSuchUpload)?; + let page = self + .lister + .list(&session, *marker, *max_parts, now_ms) + .await + .map_err(catalog_error)?; + MultipartResponses::list_parts(&session, &page, *marker, *max_parts) + .map(|response| response.map(IcebergBody::new)) + .map_err(|_| FileS3ErrorCode::InternalError) + } + (Some(MultipartRequest::Abort { .. }), FileOperation::AbortMultipart) => { + self.abort(session.ok_or(FileS3ErrorCode::NoSuchUpload)?).await + } + (Some(MultipartRequest::Complete { .. }), FileOperation::CompleteMultipart) => { + self.complete(session.ok_or(FileS3ErrorCode::NoSuchUpload)?, request, now_ms) + .await + } + _ => Err(FileS3ErrorCode::InvalidRequest), + } + } + + async fn create( + &self, + request: &FileRequest, + admission: &FileTransferAdmission, + now_ms: u64, + ) -> Result, FileS3ErrorCode> { + let limits = admission.multipart_limits().map_err(admission_error)?; + let session = MultipartSession { + context: admission.context(), + upload: OperationId::random(), + owner: FileIdentity { + table: request.location.table(), + file: FileId::random(), + }, + location: request.location.clone(), + principal: admission.principal(), + revision: 1, + created_ms: now_ms, + expires_ms: now_ms + .checked_add(limits.ttl_ms) + .ok_or(FileS3ErrorCode::InternalError)?, + limits, + phase: MultipartPhase::Open, + part_count: 0, + staged_bytes: 0, + completion: None, + published: None, + pending: None, + credit: None, + }; + let policy = self + .admission + .initialize( + session.context, + MultipartAdmissionLimits { + max_sessions: 1024, + max_reserved_bytes: 64 * 1024 * 1024 * 1024 * 1024, + }, + ) + .await + .map_err(catalog_error)?; + admission + .check_create(&session, &policy) + .map_err(admission_error)?; + if !self + .admission + .reserve(&policy, &session, now_ms) + .await + .map_err(catalog_error)? + { + return Err(FileS3ErrorCode::SlowDown); + } + let durable = self + .multipart + .load(session.context, session.upload) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::InternalError)?; + MultipartResponses::create(&durable) + .map(|response| response.map(IcebergBody::new)) + .map_err(|_| FileS3ErrorCode::InternalError) + } + + async fn upload( + &self, + session: MultipartSession, + part_number: u16, + request: Request, + admission: &FileTransferAdmission, + now_ms: u64, + streaming: Option, + ) -> Result, FileS3ErrorCode> { + let digest = if streaming.is_some() { + None + } else { + signed_digest(request.headers().get("x-amz-content-sha256"))? + }; + let (parts, body) = request.into_parts(); + let mut body = FileUploadBody::new(body, &parts.headers, streaming, admission.request_byte_limit()) + .map_err(encoding_error)?; + let length = body.decoded_length(); + let owner = FileIdentity { + table: session.owner.table, + file: FileId::random(), + }; + let tree = admission + .receive( + &self.uploads, + &mut body, + self.blocks.clone(), + owner, + length, + digest, + ) + .await + .map_err(|error| { + body.failure() + .map_or_else(|| admission_error(error), encoding_error) + })?; + let before = self + .multipart + .part(&session, part_number) + .await + .map_err(catalog_error)?; + let part = MultipartPart { + upload: session.upload, + number: part_number, + revision: before.map_or(1, |part| part.revision.checked_add(1).unwrap_or(0)), + modified_ms: now_ms, + owner, + tree, + }; + if !self + .multipart + .reserve_part(&session, &part, now_ms) + .await + .map_err(catalog_error)? + { + return Err(FileS3ErrorCode::SlowDown); + } + let pending = self + .multipart + .load(session.context, session.upload) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::InternalError)?; + if !self + .multipart + .settle_part(&pending) + .await + .map_err(catalog_error)? + { + return Err(FileS3ErrorCode::SlowDown); + } + let settled = self + .multipart + .load(session.context, session.upload) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::InternalError)?; + let part = self + .multipart + .part(&settled, part_number) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::InternalError)?; + MultipartResponses::upload_part(&part) + .map(|response| response.map(IcebergBody::new)) + .map_err(|_| FileS3ErrorCode::InternalError) + } + + async fn abort(&self, session: MultipartSession) -> Result, FileS3ErrorCode> { + if !self.multipart.abort(&session).await.map_err(catalog_error)? { + return Err(FileS3ErrorCode::SlowDown); + } + let terminal = self + .multipart + .load(session.context, session.upload) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::InternalError)?; + let policy = self + .admission + .load(session.context) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::InternalError)?; + if !self + .admission + .release(&policy, &terminal) + .await + .map_err(catalog_error)? + { + return Err(FileS3ErrorCode::SlowDown); + } + Ok(MultipartResponses::abort().map(IcebergBody::new)) + } + + async fn complete( + self: &Arc, + mut session: MultipartSession, + request: Request, + now_ms: u64, + ) -> Result, FileS3ErrorCode> { + let host = request + .headers() + .get(hyper::header::HOST) + .and_then(|value| value.to_str().ok()) + .filter(|host| !host.is_empty() && host.len() <= 256 && !host.contains('/')) + .ok_or(FileS3ErrorCode::InvalidRequest)? + .to_owned(); + let url = format!("http://{host}{}", request.uri().path()); + let signed = signed_digest(request.headers().get("x-amz-content-sha256"))?; + let bytes = read_complete_body(request.into_body()).await?; + if signed.is_some_and(|digest| digest != <[u8; 32]>::from(Sha256::digest(&bytes))) { + return Err(FileS3ErrorCode::BadDigest); + } + let requested = + crate::iceberg::CompleteSelection::parse(&bytes).map_err(|_| FileS3ErrorCode::InvalidPart)?; + let selection = requested + .resolve(&self.multipart, &session) + .await + .map_err(|error| match error { + crate::iceberg::CompleteResolveError::InvalidPart => FileS3ErrorCode::InvalidPart, + crate::iceberg::CompleteResolveError::EntityTooSmall => FileS3ErrorCode::EntityTooSmall, + crate::iceberg::CompleteResolveError::Catalog(error) => catalog_error(error), + })?; + if session.phase == MultipartPhase::Open { + if !self + .multipart + .freeze_completion(&session, &selection, now_ms) + .await + .map_err(catalog_error)? + { + return Err(FileS3ErrorCode::SlowDown); + } + session = self.current(&session).await?; + } + let expected: [u8; 32] = Sha256::digest(selection.encode()).into(); + let service = Arc::clone(self); + let resource = session.location.object_key(); + let body = crate::iceberg::FileCompleteBody::new( + async move { + service + .drive_complete(session, expected, now_ms, &url) + .await + .map(Response::into_body) + }, + &resource, + std::time::Duration::from_secs(10), + std::time::Duration::from_secs(300), + ) + .map_err(|_| FileS3ErrorCode::InternalError)?; + let mut response = Response::new(IcebergBody::complete(body)); + response.headers_mut().insert( + hyper::header::CONTENT_TYPE, + HeaderValue::from_static("application/xml"), + ); + Ok(response) + } + + async fn drive_complete( + &self, + mut session: MultipartSession, + expected: [u8; 32], + now_ms: u64, + url: &str, + ) -> Result>, FileS3ErrorCode> { + if session + .completion + .as_ref() + .map_or(true, |completion| completion.selection.digest != expected) + { + return Err(FileS3ErrorCode::InvalidPart); + } + loop { + match session.phase { + MultipartPhase::Completing => { + let completion = session + .completion + .as_ref() + .ok_or(FileS3ErrorCode::InternalError)?; + if completion.progress.next_part < completion.selected_parts { + self.multipart + .advance_completion( + &session, + self.blocks.clone(), + crate::iceberg::file_admission::MULTIPART_COPY_BYTES, + crowdb_access_iceberg::file::NATIVE_FILE_BLOCK_BYTES, + ) + .await + .map_err(work_error)?; + } else { + let tree = self + .multipart + .assembled_tree( + &session, + self.blocks.clone(), + crowdb_access_iceberg::file::NATIVE_FILE_BLOCK_BYTES, + ) + .await + .map_err(work_error)?; + let sealed = FileSealer::new(self.blocks.clone(), self.limits.max_file_bytes) + .map_err(seal_error)? + .seal_uploaded(session.owner, session.location.clone(), tree.clone()) + .await + .map_err(seal_error)?; + self.multipart + .prepare_publication(&session, &tree, &sealed, now_ms) + .await + .map_err(catalog_error)?; + } + } + MultipartPhase::Publishing | MultipartPhase::Published => { + let Some(record) = self.multipart.publish(&session).await.map_err(catalog_error)? else { + session = self.current(&session).await?; + continue; + }; + session = self.current(&session).await?; + self.release_terminal(&session).await; + return MultipartResponses::complete(&session, &record, url) + .map_err(|_| FileS3ErrorCode::InternalError); + } + _ => return Err(FileS3ErrorCode::Conflict), + } + session = self.current(&session).await?; + } + } + + async fn current(&self, session: &MultipartSession) -> Result { + self.multipart + .load(session.context, session.upload) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::NoSuchUpload) + } + + async fn release_terminal(&self, session: &MultipartSession) { + if session.credit.is_some_and(|credit| credit.released) { + return; + } + let result = async { + let policy = self + .admission + .load(session.context) + .await? + .ok_or(CatalogError::Uninitialized)?; + self.admission.release(&policy, session).await + } + .await; + match result { + Ok(true) => {} + Ok(false) => { + tracing::debug!(upload = %session.upload, "terminal credit release deferred to recovery"); + } + Err(error) => { + tracing::warn!(upload = %session.upload, %error, "terminal credit release deferred to recovery"); + } + } + } +} + +async fn read_complete_body(mut body: Incoming) -> Result, FileS3ErrorCode> { + let mut bytes = Vec::new(); + while let Some(frame) = body.frame().await { + let data = frame + .map_err(|_| FileS3ErrorCode::InvalidRequest)? + .into_data() + .map_err(|_| FileS3ErrorCode::InvalidRequest)?; + crate::iceberg::metrics::record_request_bytes(data.len()); + if bytes + .len() + .checked_add(data.len()) + .map_or(true, |length| length > 2 * 1024 * 1024) + { + return Err(FileS3ErrorCode::EntityTooLarge); + } + bytes.extend_from_slice(&data); + } + Ok(bytes) +} + +fn work_error(error: MultipartWorkError) -> FileS3ErrorCode { + match error { + MultipartWorkError::Catalog(error) => catalog_error(error), + MultipartWorkError::File(_) | MultipartWorkError::Invalid(_) => FileS3ErrorCode::InternalError, + } +} + +#[allow(clippy::needless_pass_by_value)] +pub(super) fn seal_error(error: FileSealError) -> FileS3ErrorCode { + match error { + FileSealError::Bounds => FileS3ErrorCode::EntityTooLarge, + FileSealError::Storage(_) => FileS3ErrorCode::InternalError, + FileSealError::Invalid(_) + | FileSealError::Json(_) + | FileSealError::Avro(_) + | FileSealError::Format(_) + | FileSealError::Puffin(_) => FileS3ErrorCode::InvalidRequest, + } +} + +pub(super) fn encoding_error(error: FileEncodingError) -> FileS3ErrorCode { + match error { + FileEncodingError::Length => FileS3ErrorCode::EntityTooLarge, + FileEncodingError::Checksum => FileS3ErrorCode::BadDigest, + FileEncodingError::Signature => FileS3ErrorCode::AccessDenied, + FileEncodingError::Framing | FileEncodingError::Transport => FileS3ErrorCode::InvalidRequest, + } +} + +pub(super) fn signed_digest(value: Option<&HeaderValue>) -> Result, FileS3ErrorCode> { + let Some(value) = value else { + return Ok(None); + }; + let value = value.to_str().map_err(|_| FileS3ErrorCode::InvalidRequest)?; + if value == "UNSIGNED-PAYLOAD" { + return Ok(None); + } + if value.len() != 64 { + return Err(FileS3ErrorCode::InvalidRequest); + } + let mut digest = [0; 32]; + for (target, pair) in digest.iter_mut().zip(value.as_bytes().chunks_exact(2)) { + let hex = |byte| match byte { + b'0'..=b'9' => Some(byte - b'0'), + b'a'..=b'f' => Some(byte - b'a' + 10), + _ => None, + }; + *target = (hex(pair[0]).ok_or(FileS3ErrorCode::InvalidRequest)? << 4) + | hex(pair[1]).ok_or(FileS3ErrorCode::InvalidRequest)?; + } + Ok(Some(digest)) +} diff --git a/app/crowdb-access-server/src/iceberg/file_recovery.rs b/app/crowdb-access-server/src/iceberg/file_recovery.rs new file mode 100644 index 000000000..fcbde06c0 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_recovery.rs @@ -0,0 +1,97 @@ +use std::sync::Arc; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use crowdb_access_iceberg::catalog::{CatalogError, CatalogRepository, RootState, RoutedCatalogStore}; +use crowdb_access_iceberg::file::{FileBlockStore, MultipartRecovery, NATIVE_FILE_BLOCK_BYTES}; + +pub(super) async fn run( + catalog: Arc, + store: Arc, + blocks: Arc, +) { + let mut interval = tokio::time::interval(Duration::from_secs(1)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + let mut context = None; + let mut continuation = None; + let mut observation = None; + loop { + interval.tick().await; + let status = tokio::time::timeout(Duration::from_secs(1), catalog.status()).await; + let Ok(Ok((root, authority))) = status else { + continuation = None; + observation = None; + tracing::warn!( + ?status, + "multipart catalog status unavailable; deferring recovery" + ); + continue; + }; + if root.state != RootState::Ready || context != Some(root.context) { + context = Some(root.context); + continuation = None; + observation = None; + } + if root.state != RootState::Ready { + continue; + } + let budget = + Duration::from_millis(authority.admission_bounds.request_ms).min(Duration::from_secs(60)); + let recovery = MultipartRecovery::new( + store.clone(), + blocks.clone(), + super::file_admission::MULTIPART_COPY_BYTES, + NATIVE_FILE_BLOCK_BYTES, + ) + .and_then(|recovery| recovery.with_session_timeout(budget)); + let Ok(recovery) = recovery else { + tracing::error!("multipart recovery bounds invalid; deferring page until catalog is corrected"); + continue; + }; + let Some(observed) = observation.take() else { + match tokio::time::timeout(budget, recovery.observe_page(root.context, continuation.clone())) + .await + { + Ok(Ok(observed)) => observation = Some(observed), + result => { + continuation = None; + tracing::warn!(?result, "multipart observation failed; restarting sweep"); + } + } + continue; + }; + let Some(now_ms) = SystemTime::now() + .duration_since(UNIX_EPOCH) + .ok() + .and_then(|elapsed| u64::try_from(elapsed.as_millis()).ok()) + else { + tracing::error!("multipart recovery clock invalid; deferring expiry processing"); + continue; + }; + let page_budget = budget.saturating_mul(5).saturating_add(Duration::from_secs(2)); + let result = + tokio::time::timeout(page_budget, recovery.recover_observed_page(observed, now_ms)).await; + match result { + Ok(Ok(page)) => { + continuation = page.continuation; + for (upload, error) in page.failures { + tracing::error!(%upload, %error, "multipart recovery failed; retaining evidence for a later sweep"); + } + tracing::debug!( + progressed = page.progressed, + deferred = page.deferred, + retained = page.retained, + awaiting_seal = page.awaiting_seal.len(), + "multipart recovery page processed" + ); + } + Ok(Err(CatalogError::Busy | CatalogError::Conflict)) => { + continuation = None; + } + Ok(Err(error)) => { + continuation = None; + tracing::error!(%error, "multipart recovery scan failed; restarting sweep"); + } + Err(_) => tracing::warn!("multipart page scan budget exhausted; retrying cursor"), + } + } +} diff --git a/app/crowdb-access-server/src/iceberg/file_request.rs b/app/crowdb-access-server/src/iceberg/file_request.rs new file mode 100644 index 000000000..a6264da5d --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_request.rs @@ -0,0 +1,177 @@ +use std::collections::BTreeMap; + +use crowdb_access_iceberg::file::{FileLocation, FileOperation}; +use hyper::{Method, Uri}; + +#[derive(Clone, Debug, Eq, PartialEq)] +pub enum MultipartRequest { + Create, + Upload { + upload_id: String, + part_number: u16, + }, + List { + upload_id: String, + marker: u16, + max_parts: u16, + }, + Complete { + upload_id: String, + }, + Abort { + upload_id: String, + }, +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct FileRequest { + pub location: FileLocation, + pub operation: FileOperation, + pub multipart: Option, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] +pub enum FileRequestError { + #[error("malformed native file request")] + Invalid, + #[error("unsupported native file operation")] + Unsupported, +} + +impl FileRequest { + /// Parses the native path-style object surface without general S3 routing. + /// Authentication must use the original URI before these fields are decoded. + /// # Errors + /// Rejects escaped paths, ambiguous query values and unsupported operations. + pub fn parse(method: &Method, uri: &Uri) -> Result { + let target = uri.path_and_query().ok_or(FileRequestError::Invalid)?.as_str(); + if target.len() > 8192 { + return Err(FileRequestError::Invalid); + } + let path = decode(uri.path())?; + let path = path.strip_prefix('/').ok_or(FileRequestError::Invalid)?; + let location = format!("s3://{path}") + .parse() + .map_err(|_| FileRequestError::Invalid)?; + let mut query = query(uri.query())?; + let multipart = multipart(method, &mut query)?; + if !query.is_empty() { + return Err(FileRequestError::Unsupported); + } + let operation = match &multipart { + Some(MultipartRequest::Create) => FileOperation::CreateMultipart, + Some(MultipartRequest::Upload { .. }) => FileOperation::UploadPart, + Some(MultipartRequest::List { .. }) => FileOperation::ListParts, + Some(MultipartRequest::Complete { .. }) => FileOperation::CompleteMultipart, + Some(MultipartRequest::Abort { .. }) => FileOperation::AbortMultipart, + None if method == Method::GET => FileOperation::Get, + None if method == Method::HEAD => FileOperation::Head, + None if method == Method::PUT => FileOperation::Put, + None => return Err(FileRequestError::Unsupported), + }; + Ok(Self { + location, + operation, + multipart, + }) + } +} + +fn multipart( + method: &Method, + query: &mut BTreeMap, +) -> Result, FileRequestError> { + if let Some(value) = query.remove("uploads") { + if method != Method::POST || !value.is_empty() || !query.is_empty() { + return Err(FileRequestError::Invalid); + } + return Ok(Some(MultipartRequest::Create)); + } + let Some(upload_id) = query.remove("uploadId") else { + return Ok(None); + }; + if upload_id.is_empty() || upload_id.len() > 256 || !upload_id.bytes().all(|byte| byte.is_ascii_graphic()) + { + return Err(FileRequestError::Invalid); + } + let request = if method == Method::PUT { + let part_number = number(query.remove("partNumber"), None, 1, 10_000)?; + MultipartRequest::Upload { + upload_id, + part_number, + } + } else if method == Method::GET { + let marker = number(query.remove("part-number-marker"), Some(0), 0, 10_000)?; + let max_parts = number(query.remove("max-parts"), Some(1000), 1, 1000)?; + MultipartRequest::List { + upload_id, + marker, + max_parts, + } + } else if method == Method::POST { + MultipartRequest::Complete { upload_id } + } else if method == Method::DELETE { + MultipartRequest::Abort { upload_id } + } else { + return Err(FileRequestError::Unsupported); + }; + Ok(Some(request)) +} + +fn number(value: Option, default: Option, min: u16, max: u16) -> Result { + let value = match value { + Some(value) if !value.is_empty() && value.bytes().all(|byte| byte.is_ascii_digit()) => { + value.parse::().map_err(|_| FileRequestError::Invalid)? + } + Some(_) => return Err(FileRequestError::Invalid), + None => default.ok_or(FileRequestError::Invalid)?, + }; + if !(min..=max).contains(&value) { + return Err(FileRequestError::Invalid); + } + Ok(value) +} + +fn query(query: Option<&str>) -> Result, FileRequestError> { + let mut fields = BTreeMap::new(); + let Some(query) = query else { + return Ok(fields); + }; + for field in query.split('&') { + let (name, value) = field.split_once('=').unwrap_or((field, "")); + let name = decode(name)?; + let value = decode(value)?; + if name.is_empty() || fields.len() == 16 || fields.insert(name, value).is_some() { + return Err(FileRequestError::Invalid); + } + } + for name in [ + "X-Amz-Algorithm", + "X-Amz-Credential", + "X-Amz-Date", + "X-Amz-Expires", + "X-Amz-Security-Token", + "X-Amz-SignedHeaders", + "X-Amz-Signature", + ] { + fields.remove(name); + } + Ok(fields) +} + +fn decode(value: &str) -> Result { + let bytes = value.as_bytes(); + for (offset, byte) in bytes.iter().enumerate() { + if *byte == b'%' + && !bytes + .get(offset + 1..offset + 3) + .is_some_and(|pair| pair.iter().all(u8::is_ascii_hexdigit)) + { + return Err(FileRequestError::Invalid); + } + } + percent_encoding::percent_decode_str(value) + .decode_utf8() + .map(std::borrow::Cow::into_owned) + .map_err(|_| FileRequestError::Invalid) +} diff --git a/app/crowdb-access-server/src/iceberg/file_response.rs b/app/crowdb-access-server/src/iceberg/file_response.rs new file mode 100644 index 000000000..9fbf3bed3 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_response.rs @@ -0,0 +1,291 @@ +use std::fmt::Write; + +use chrono::{DateTime, SecondsFormat, Utc}; +use crowdb_access_iceberg::file::{ + FileRecord, MultipartPart, MultipartPartPage, MultipartPhase, MultipartSession, +}; +use hyper::http::header::{HeaderValue, CONTENT_TYPE, ETAG}; +use hyper::{Response, StatusCode}; + +const XML_TYPE: &str = "application/xml"; +const XML_PREFIX: &str = ""; +const XML_NAMESPACE: &str = "http://s3.amazonaws.com/doc/2006-03-01/"; + +#[derive(Debug, thiserror::Error)] +pub enum FileResponseError { + #[error("multipart response state is invalid")] + Invalid, + #[error("multipart part timestamp is out of range")] + Timestamp, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum FileS3ErrorCode { + AccessDenied, + NoSuchUpload, + InvalidPart, + EntityTooLarge, + InvalidRequest, + InternalError, + NoSuchKey, + InvalidRange, + SlowDown, + Conflict, + BadDigest, + EntityTooSmall, +} + +impl FileS3ErrorCode { + const fn status(self) -> StatusCode { + match self { + Self::AccessDenied => StatusCode::FORBIDDEN, + Self::NoSuchUpload | Self::NoSuchKey => StatusCode::NOT_FOUND, + Self::InvalidPart | Self::InvalidRequest | Self::BadDigest | Self::EntityTooSmall => { + StatusCode::BAD_REQUEST + } + Self::EntityTooLarge => StatusCode::PAYLOAD_TOO_LARGE, + Self::InternalError => StatusCode::INTERNAL_SERVER_ERROR, + Self::InvalidRange => StatusCode::RANGE_NOT_SATISFIABLE, + Self::SlowDown => StatusCode::SERVICE_UNAVAILABLE, + Self::Conflict => StatusCode::CONFLICT, + } + } + + const fn message(self) -> &'static str { + match self { + Self::AccessDenied => "Access Denied", + Self::NoSuchUpload => "The specified upload does not exist", + Self::InvalidPart => "One or more parts are invalid", + Self::EntityTooLarge => "The request exceeds the allowed size", + Self::InvalidRequest => "The request is invalid", + Self::InternalError => "The service could not complete the request", + Self::NoSuchKey => "The specified key does not exist", + Self::InvalidRange => "The requested range cannot be satisfied", + Self::SlowDown => "The service is temporarily unavailable", + Self::Conflict => "The immutable object already exists with different content", + Self::BadDigest => "The supplied digest does not match the uploaded content", + Self::EntityTooSmall => "A nonfinal upload part is smaller than 5 MiB", + } + } + + const fn name(self) -> &'static str { + match self { + Self::AccessDenied => "AccessDenied", + Self::NoSuchUpload => "NoSuchUpload", + Self::InvalidPart => "InvalidPart", + Self::EntityTooLarge => "EntityTooLarge", + Self::InvalidRequest => "InvalidRequest", + Self::InternalError => "InternalError", + Self::NoSuchKey => "NoSuchKey", + Self::InvalidRange => "InvalidRange", + Self::SlowDown => "SlowDown", + Self::Conflict => "OperationAborted", + Self::BadDigest => "BadDigest", + Self::EntityTooSmall => "EntityTooSmall", + } + } +} + +pub struct MultipartResponses; + +impl MultipartResponses { + /// # Errors + /// Rejects invalid durable session state before emitting a success response. + pub fn create(session: &MultipartSession) -> Result>, FileResponseError> { + session.validate().map_err(|_| FileResponseError::Invalid)?; + if session.phase != MultipartPhase::Open { + return Err(FileResponseError::Invalid); + } + let mut body = start("InitiateMultipartUploadResult"); + location_fields(&mut body, session); + element(&mut body, "UploadId", &session.upload.to_string()); + end(&mut body, "InitiateMultipartUploadResult"); + Ok(xml(StatusCode::OK, body)) + } + + /// # Errors + /// Rejects a part that lacks durable identity or an upload timestamp. + pub fn upload_part(part: &MultipartPart) -> Result>, FileResponseError> { + part.validate().map_err(|_| FileResponseError::Invalid)?; + let mut response = Response::new(Vec::new()); + response.headers_mut().insert( + ETAG, + HeaderValue::from_str(&etag(part.tree.digest)).map_err(|_| FileResponseError::Invalid)?, + ); + Ok(response) + } + + /// # Errors + /// Rejects incoherent requested markers, pages, part bindings or timestamps. + pub fn list_parts( + session: &MultipartSession, + page: &MultipartPartPage, + marker: u16, + max_parts: u16, + ) -> Result>, FileResponseError> { + session.validate().map_err(|_| FileResponseError::Invalid)?; + if marker > 10_000 || max_parts == 0 || max_parts > 1000 || page.parts.len() > max_parts as usize { + return Err(FileResponseError::Invalid); + } + let mut body = start("ListPartsResult"); + location_fields(&mut body, session); + element(&mut body, "UploadId", &session.upload.to_string()); + element(&mut body, "PartNumberMarker", &marker.to_string()); + if let Some(next) = page.next_marker { + if page.parts.last().map(|part| part.number) != Some(next) { + return Err(FileResponseError::Invalid); + } + element(&mut body, "NextPartNumberMarker", &next.to_string()); + } + element(&mut body, "MaxParts", &max_parts.to_string()); + element( + &mut body, + "IsTruncated", + if page.next_marker.is_some() { + "true" + } else { + "false" + }, + ); + let mut previous = marker; + for part in &page.parts { + part.validate_for(session) + .map_err(|_| FileResponseError::Invalid)?; + if part.number <= previous + || part.modified_ms < session.created_ms + || part.modified_ms >= session.expires_ms + { + return Err(FileResponseError::Invalid); + } + previous = part.number; + body.push_str(""); + element(&mut body, "PartNumber", &part.number.to_string()); + let timestamp = DateTime::::from_timestamp_millis( + i64::try_from(part.modified_ms).map_err(|_| FileResponseError::Timestamp)?, + ) + .ok_or(FileResponseError::Timestamp)?; + element( + &mut body, + "LastModified", + ×tamp.to_rfc3339_opts(SecondsFormat::Millis, true), + ); + element(&mut body, "ETag", &etag(part.tree.digest)); + element(&mut body, "Size", &part.tree.length.to_string()); + body.push_str(""); + } + end(&mut body, "ListPartsResult"); + Ok(xml(StatusCode::OK, body)) + } + + /// # Errors + /// Rejects completion without a matching, durable published file record. + pub fn complete( + session: &MultipartSession, + record: &FileRecord, + response_url: &str, + ) -> Result>, FileResponseError> { + session.validate().map_err(|_| FileResponseError::Invalid)?; + record.validate().map_err(|_| FileResponseError::Invalid)?; + if session.phase != MultipartPhase::Published + || session.published != Some(record.file) + || session.location != record.location + || response_url.len() > 2048 + || !(response_url.starts_with("https://") || response_url.starts_with("http://")) + || session + .completion + .as_ref() + .and_then(|completion| completion.candidate.as_ref()) + .map_or(true, |candidate| { + candidate.length != record.length || candidate.digest != record.digest + }) + { + return Err(FileResponseError::Invalid); + } + let mut body = start("CompleteMultipartUploadResult"); + element(&mut body, "Location", response_url); + location_fields(&mut body, session); + element(&mut body, "ETag", &etag(record.digest)); + end(&mut body, "CompleteMultipartUploadResult"); + Ok(xml(StatusCode::OK, body)) + } + + #[must_use] + pub fn abort() -> Response> { + let mut response = Response::new(Vec::new()); + *response.status_mut() = StatusCode::NO_CONTENT; + response + } + + /// Formats a bounded S3 error body using a stable code and request identity. + /// # Errors + /// Rejects unbounded or invalid resource and request identifiers. + pub fn error( + code: FileS3ErrorCode, + resource: &str, + request_id: &str, + ) -> Result>, FileResponseError> { + if resource.len() > 2048 || request_id.len() > 128 || request_id.is_empty() { + return Err(FileResponseError::Invalid); + } + let mut body = format!("{XML_PREFIX}"); + element(&mut body, "Code", code.name()); + element(&mut body, "Message", code.message()); + element(&mut body, "Resource", resource); + element(&mut body, "RequestId", request_id); + end(&mut body, "Error"); + Ok(xml(code.status(), body)) + } +} + +fn xml(status: StatusCode, body: String) -> Response> { + let mut response = Response::new(body.into_bytes()); + *response.status_mut() = status; + response + .headers_mut() + .insert(CONTENT_TYPE, HeaderValue::from_static(XML_TYPE)); + response +} + +fn etag(digest: [u8; 32]) -> String { + let mut tag = String::with_capacity(66); + tag.push('"'); + for byte in digest { + write!(tag, "{byte:02x}").expect("string writes do not fail"); + } + tag.push('"'); + tag +} + +fn start(root: &str) -> String { + format!("{XML_PREFIX}<{root} xmlns=\"{XML_NAMESPACE}\">") +} + +fn end(body: &mut String, root: &str) { + body.push_str("'); +} + +fn location_fields(body: &mut String, session: &MultipartSession) { + element(body, "Bucket", &session.location.table().bucket()); + element(body, "Key", &session.location.object_key()); +} + +fn element(body: &mut String, name: &str, value: &str) { + body.push('<'); + body.push_str(name); + body.push('>'); + for character in value.chars() { + match character { + '&' => body.push_str("&"), + '<' => body.push_str("<"), + '>' => body.push_str(">"), + '"' => body.push_str("""), + '\'' => body.push_str("'"), + _ => body.push(character), + } + } + body.push_str("'); +} diff --git a/app/crowdb-access-server/src/iceberg/file_selection.rs b/app/crowdb-access-server/src/iceberg/file_selection.rs new file mode 100644 index 000000000..88a1dc5b2 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_selection.rs @@ -0,0 +1,214 @@ +use crowdb_access_iceberg::catalog::CatalogError; +use crowdb_access_iceberg::file::{MultipartRepository, MultipartSelection, MultipartSession, SelectedPart}; +use quick_xml::events::Event; +use quick_xml::Reader; + +const MAX_COMPLETE_XML_BYTES: usize = 2 * 1024 * 1024; +const MAX_COMPLETE_PARTS: usize = 10_000; +const S3_NAMESPACE: &[u8] = b"http://s3.amazonaws.com/doc/2006-03-01/"; + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct CompletePart { + pub number: u16, + pub digest: [u8; 32], +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct CompleteSelection { + parts: Vec, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] +#[error("invalid multipart completion XML")] +pub struct CompleteRequestError; + +#[derive(Debug, thiserror::Error)] +pub enum CompleteResolveError { + #[error("multipart selection does not match durable parts")] + InvalidPart, + #[error("a nonfinal multipart part is smaller than 5 MiB")] + EntityTooSmall, + #[error(transparent)] + Catalog(#[from] CatalogError), +} + +impl CompleteSelection { + /// Parses a bounded S3 `CompleteMultipartUpload` body. The caller must match + /// each selected digest to the current durable part revision before freezing. + /// # Errors + /// Rejects malformed XML, extra fields and unordered or duplicate parts. + pub fn parse(bytes: &[u8]) -> Result { + if bytes.is_empty() || bytes.len() > MAX_COMPLETE_XML_BYTES { + return Err(CompleteRequestError); + } + let mut reader = Reader::from_reader(bytes); + let mut state = State::Start; + let mut parts = Vec::new(); + let mut number = None; + let mut digest = None; + let mut etag = Vec::new(); + loop { + match reader.read_event().map_err(|_| CompleteRequestError)? { + Event::Decl(_) if state == State::Start => {} + Event::Start(event) if valid_attributes(state, &event)? => { + state = match (state, event.name().as_ref()) { + (State::Start, b"CompleteMultipartUpload") => State::Root, + (State::Root, b"Part") if parts.len() < MAX_COMPLETE_PARTS => State::Part, + (State::Part, b"PartNumber") if number.is_none() => State::Number, + (State::Part, b"ETag") if digest.is_none() => State::Etag, + _ => return Err(CompleteRequestError), + }; + } + Event::Text(event) => match state { + State::Number if number.is_none() => { + let value: &[u8] = event.as_ref(); + if value.is_empty() || !value.iter().all(u8::is_ascii_digit) { + return Err(CompleteRequestError); + } + number = Some( + std::str::from_utf8(value) + .map_err(|_| CompleteRequestError)? + .parse::() + .map_err(|_| CompleteRequestError)?, + ); + } + State::Etag => append_etag(&mut etag, &event)?, + State::Start | State::Root | State::Part | State::Done + if event.iter().all(u8::is_ascii_whitespace) => {} + _ => return Err(CompleteRequestError), + }, + Event::GeneralRef(event) if state == State::Etag => { + if event.len() > 16 { + return Err(CompleteRequestError); + } + let name = std::str::from_utf8(&event).map_err(|_| CompleteRequestError)?; + let encoded = format!("&{name};"); + let decoded = quick_xml::escape::unescape(&encoded).map_err(|_| CompleteRequestError)?; + append_etag(&mut etag, decoded.as_bytes())?; + } + Event::End(event) => { + state = match (state, event.name().as_ref()) { + (State::Number, b"PartNumber") if number.is_some() => State::Part, + (State::Etag, b"ETag") => { + digest = Some(parse_etag(&etag)?); + etag.clear(); + State::Part + } + (State::Part, b"Part") => { + let number = number.take().ok_or(CompleteRequestError)?; + let digest = digest.take().ok_or(CompleteRequestError)?; + if number == 0 + || number > 10_000 + || parts + .last() + .is_some_and(|part: &CompletePart| part.number >= number) + { + return Err(CompleteRequestError); + } + parts.push(CompletePart { number, digest }); + State::Root + } + (State::Root, b"CompleteMultipartUpload") if !parts.is_empty() => State::Done, + _ => return Err(CompleteRequestError), + }; + } + Event::Eof if state == State::Done => return Ok(Self { parts }), + _ => return Err(CompleteRequestError), + } + } + } + + #[must_use] + pub fn parts(&self) -> &[CompletePart] { + &self.parts + } + + /// Resolves the selected parts against one current durable session snapshot. + /// # Errors + /// Rejects missing, replaced or differently hashed parts and storage failures. + pub async fn resolve( + &self, + repository: &MultipartRepository, + session: &MultipartSession, + ) -> Result { + if self.parts.len() > usize::from(session.limits.max_parts) { + return Err(CompleteResolveError::InvalidPart); + } + let mut selected = Vec::with_capacity(self.parts.len()); + for (index, requested) in self.parts.iter().enumerate() { + let part = repository + .part(session, requested.number) + .await? + .ok_or(CompleteResolveError::InvalidPart)?; + if part.tree.digest != requested.digest { + return Err(CompleteResolveError::InvalidPart); + } + if index + 1 < self.parts.len() && part.tree.length < 5 * 1024 * 1024 { + return Err(CompleteResolveError::EntityTooSmall); + } + selected.push(SelectedPart { + number: part.number, + revision: part.revision, + digest: part.tree.digest, + }); + } + MultipartSelection::new(selected).map_err(|_| CompleteResolveError::InvalidPart) + } +} + +#[derive(Clone, Copy, Eq, PartialEq)] +enum State { + Start, + Root, + Part, + Number, + Etag, + Done, +} + +fn parse_etag(bytes: &[u8]) -> Result<[u8; 32], CompleteRequestError> { + let hex = bytes + .strip_prefix(b"\"") + .and_then(|bytes| bytes.strip_suffix(b"\"")) + .ok_or(CompleteRequestError)?; + if hex.len() != 64 { + return Err(CompleteRequestError); + } + let mut digest = [0; 32]; + for (target, pair) in digest.iter_mut().zip(hex.chunks_exact(2)) { + *target = (hex_digit(pair[0])? << 4) | hex_digit(pair[1])?; + } + Ok(digest) +} + +fn append_etag(etag: &mut Vec, bytes: &[u8]) -> Result<(), CompleteRequestError> { + if etag.len().saturating_add(bytes.len()) > 66 { + return Err(CompleteRequestError); + } + etag.extend_from_slice(bytes); + Ok(()) +} + +fn hex_digit(byte: u8) -> Result { + match byte { + b'0'..=b'9' => Ok(byte - b'0'), + b'a'..=b'f' => Ok(byte - b'a' + 10), + _ => Err(CompleteRequestError), + } +} + +fn valid_attributes( + state: State, + event: &quick_xml::events::BytesStart<'_>, +) -> Result { + let mut attributes = event.attributes(); + let Some(attribute) = attributes.next() else { + return Ok(true); + }; + let attribute = attribute.map_err(|_| CompleteRequestError)?; + Ok(state == State::Start + && event.name().as_ref() == b"CompleteMultipartUpload" + && attribute.key.as_ref() == b"xmlns" + && attribute.value.as_ref() == S3_NAMESPACE + && attributes.next().is_none()) +} diff --git a/app/crowdb-access-server/src/iceberg/file_upload.rs b/app/crowdb-access-server/src/iceberg/file_upload.rs new file mode 100644 index 000000000..6a7992505 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_upload.rs @@ -0,0 +1,134 @@ +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; + +use crowdb_access_iceberg::file::{ + FileBlockStore, FileIdentity, FileIoError, FileTree, FileTreeWriter, NATIVE_FILE_BLOCK_BYTES, +}; +use http_body_util::BodyExt; +use hyper::body::{Body, Bytes}; + +#[derive(Clone, Copy, Debug)] +pub struct FileUploadConstraints { + pub max_bytes: u64, + pub content_length: Option, + pub sha256: Option<[u8; 32]>, +} + +impl FileUploadConstraints { + fn validate(self) -> Result<(), FileUploadError> { + if self.max_bytes == 0 + || self.max_bytes > u64::MAX / 8 + || self.content_length.is_some_and(|length| length > self.max_bytes) + { + return Err(FileUploadError::Bounds); + } + Ok(()) + } +} + +#[derive(Debug, thiserror::Error)] +pub enum FileUploadError { + #[error(transparent)] + Storage(#[from] FileIoError), + #[error("file upload capacity exhausted")] + Busy, + #[error("file upload byte bounds exceeded")] + Bounds, + #[error("file upload body read failed")] + Body, + #[error("file upload length differs from declared content length")] + Length, + #[error("file upload digest differs from signed payload digest")] + Digest, + #[error("file upload trailers are not supported")] + Trailers, +} + +pub struct FileUploadBudget { + active: Arc, + limit: usize, +} + +impl FileUploadBudget { + /// # Errors + /// Rejects empty or unbounded concurrent upload limits. + pub fn new(limit: usize) -> Result { + if limit == 0 || limit > 64 { + return Err(FileUploadError::Bounds); + } + Ok(Self { + active: Arc::new(AtomicUsize::new(0)), + limit, + }) + } + + #[must_use] + pub fn active(&self) -> usize { + self.active.load(Ordering::Acquire) + } + + /// Stages bytes only; the caller must authorize intersected limits and seal before publication. + /// Writes each received frame in bounded slices and never polls ahead of a pending storage write. + /// # Errors + /// Rejects exhausted admission, byte bounds, body failures and length/digest mismatches. + /// Cancellation or failure retains orphan blocks without publishing any authority. + pub async fn receive + Unpin>( + &self, + mut body: Input, + store: Arc, + owner: FileIdentity, + constraints: FileUploadConstraints, + ) -> Result { + constraints.validate()?; + self.active + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |active| { + (active < self.limit).then_some(active + 1) + }) + .map_err(|_| FileUploadError::Busy)?; + let _permit = Permit(self.active.clone()); + let mut writer = FileTreeWriter::new(store, owner, NATIVE_FILE_BLOCK_BYTES)?; + while let Some(frame) = body.frame().await { + let bytes = frame + .map_err(|_| FileUploadError::Body)? + .into_data() + .map_err(|_| FileUploadError::Trailers)?; + let length = writer + .length() + .checked_add(bytes.len() as u64) + .ok_or(FileUploadError::Bounds)?; + if length > constraints.max_bytes { + return Err(FileUploadError::Bounds); + } + if constraints + .content_length + .is_some_and(|declared| length > declared) + { + return Err(FileUploadError::Length); + } + for chunk in bytes.chunks(64 * 1024) { + writer.push(chunk).await?; + } + } + if constraints + .content_length + .is_some_and(|declared| declared != writer.length()) + { + return Err(FileUploadError::Length); + } + let tree = writer.finish().await?; + if constraints.sha256.is_some_and(|digest| digest != tree.digest) { + return Err(FileUploadError::Digest); + } + Ok(tree) + } +} + +struct Permit(Arc); + +impl Drop for Permit { + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::AcqRel); + } +} diff --git a/app/crowdb-access-server/src/iceberg/gc_control.rs b/app/crowdb-access-server/src/iceberg/gc_control.rs new file mode 100644 index 000000000..3b8aeac3a --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/gc_control.rs @@ -0,0 +1,242 @@ +use std::sync::Arc; + +use crowdb_access_iceberg::{ + catalog::{ + CatalogContext, CatalogRepository, CatalogStore, ManagementPrivilege, RootState, RoutedCatalogStore, + }, + gc::{GcLimits, GcPin, GcRepository, GcTask, ReaderPins}, + key::{CatalogId, OperationId, TableId}, + operation::{ManagementAction, ManagementPhase}, + record::StorageRecord, + table::{head_key, TableHead, TableLifecycle}, + wire::BearerAuthenticator, +}; + +use super::gc_runtime::GcRuntimeConfig; + +type BoxError = Box; + +pub(super) async fn manage( + catalog: &CatalogRepository, + store: Arc, + authentication: &BearerAuthenticator, + arguments: &[String], +) -> Result<(), BoxError> { + let token = std::env::var("CROWDB_ICEBERG_TOKEN")?; + let principal = authentication + .authenticate(&format!("Bearer {token}")) + .ok_or("invalid management bearer token")?; + if principal.management == ManagementPrivilege::None { + return Err("management privilege is required".into()); + } + let config = GcRuntimeConfig::from_env()?; + let limits = config.limits; + let repository = GcRepository::new(store.clone()); + let pins = ReaderPins::new(store.clone()); + match arguments.iter().map(String::as_str).collect::>().as_slice() { + ["limits"] => { + println!("{}", serde_json::json!({ + "enabled": config.enabled, + "interval_ms": config.interval_ms, + "page_items": limits.page_items, + "page_bytes": limits.page_bytes, + "step_bytes": limits.step_bytes, + "step_ms": limits.step_ms, + "concurrency": limits.concurrency, + "minimum_retention_ms": limits.minimum_retention_ms, + "retry_base_ms": limits.retry_base_ms, + "retry_max_ms": limits.retry_max_ms, + "corruption_attempts": limits.corruption_attempts, + "kv_bytes": config.kv_bytes, + "kv_requests": config.kv_requests, + "chunk_bytes": config.chunk_bytes, + "chunk_requests": config.chunk_requests, + })); + } + ["start-table", identity, table] => { + start_table(catalog, store.as_ref(), &repository, identity, table, limits).await?; + } + ["start-retired", identity, catalog_id, epoch] => { + if principal.management != ManagementPrivilege::Clear { + return Err("clear privilege is required for retired catalogs".into()); + } + start_retired(catalog, &repository, identity, catalog_id, epoch, limits).await?; + } + ["inspect" | "pause" | "resume" | "retry", catalog_id, identity] => { + let catalog_id: CatalogId = catalog_id.parse()?; + let identity: OperationId = identity.parse()?; + let task = repository.task(catalog_id, identity).await?.ok_or("GC task is missing")?; + let task = match arguments[0].as_str() { + "pause" => repository.pause(&task, true).await?, + "resume" => repository.pause(&task, false).await?, + "retry" => repository.retry(&task).await?, + _ => task, + }; + show(&task); + } + ["pin", identity, table] => { + pin_table(catalog, store.as_ref(), &pins, principal.name, identity, table).await?; + } + ["unpin", catalog_id, table, identity] => { + unpin_table(&pins, principal.name, catalog_id, table, identity).await?; + } + _ => return Err("usage: crowdb-iceberg gc limits | start-table UUID TABLE_ID | start-retired UUID CATALOG_ID EPOCH | inspect|pause|resume|retry CATALOG_ID TASK_ID | pin UUID TABLE_ID | unpin CATALOG_ID TABLE_ID PIN_ID".into()), + } + Ok(()) +} + +async fn start_table( + catalog: &CatalogRepository, + store: &RoutedCatalogStore, + repository: &GcRepository, + identity: &str, + table: &str, + limits: GcLimits, +) -> Result<(), BoxError> { + let (root, _) = catalog.status().await?; + if root.state != RootState::Ready { + return Err("catalog is not ready".into()); + } + let table: TableId = table.parse()?; + let identity: OperationId = identity.parse()?; + let head = load_head(store, root.context.catalog, table).await?; + if head.lifecycle != TableLifecycle::Tombstone { + return Err("live-table GC is disabled; only tombstoned tables can be reclaimed".into()); + } + if let Some(existing) = repository.task(root.context.catalog, identity).await? { + if existing.context != root.context || existing.head.as_ref() != Some(&head) { + return Err("GC task identity is already bound to another table state".into()); + } + show(&existing); + return Ok(()); + } + let task = GcTask::plan( + root.context, + identity, + Some(head), + super::runtime::now_ms()?, + limits, + )?; + repository.create(&task).await?; + show(&task); + Ok(()) +} + +async fn start_retired( + catalog: &CatalogRepository, + repository: &GcRepository, + identity: &str, + catalog_id: &str, + epoch: &str, + limits: GcLimits, +) -> Result<(), BoxError> { + let context = CatalogContext { + catalog: catalog_id.parse()?, + activation_epoch: epoch.parse()?, + }; + context.validate()?; + let identity: OperationId = identity.parse()?; + let clear = catalog + .operation(identity) + .await? + .ok_or("clear operation is missing")?; + if clear.phase != ManagementPhase::Complete + || clear.request.action != ManagementAction::Clear + || clear.request.confirmation != Some(context.catalog) + || clear.request.expected_epoch != context.activation_epoch + { + return Err("clear operation does not authorize this retired context".into()); + } + let task = repository + .admit_retired(&clear, super::runtime::now_ms()?, limits) + .await?; + show(&task); + Ok(()) +} + +async fn pin_table( + catalog: &CatalogRepository, + store: &RoutedCatalogStore, + pins: &ReaderPins, + principal: &str, + identity: &str, + table: &str, +) -> Result<(), BoxError> { + let (root, _) = catalog.status().await?; + if root.state != RootState::Ready { + return Err("catalog is not ready".into()); + } + let table: TableId = table.parse()?; + let pin = GcPin { + context: root.context, + identity: identity.parse()?, + head: load_head(store, root.context.catalog, table).await?, + principal: principal.to_owned(), + expires_ms: 0, + released: false, + operator: true, + protects_uploads: true, + }; + pins.acquire(&pin).await?; + println!( + "{}", + serde_json::json!({"pin": pin.identity.to_string(), "table": table.to_string()}) + ); + Ok(()) +} + +async fn unpin_table( + pins: &ReaderPins, + principal: &str, + catalog_id: &str, + table: &str, + identity: &str, +) -> Result<(), BoxError> { + let catalog_id: CatalogId = catalog_id.parse()?; + let table: TableId = table.parse()?; + let identity: OperationId = identity.parse()?; + let pin = pins + .get(catalog_id, table, identity) + .await? + .ok_or("GC pin is missing")?; + if !pin.operator || pin.principal != principal { + return Err("operator pin is owned by another principal".into()); + } + pins.release(&pin).await?; + println!("{}", serde_json::json!({"released": identity.to_string()})); + Ok(()) +} + +async fn load_head( + store: &RoutedCatalogStore, + catalog: CatalogId, + table: TableId, +) -> Result { + let key = head_key(catalog, table); + let value = store.get(&key.encode()?).await?.ok_or("table head is missing")?; + let StorageRecord::TableHead(head) = StorageRecord::decode(&key, &value.bytes)? else { + return Err("table head has an invalid record".into()); + }; + Ok(*head) +} + +fn show(task: &GcTask) { + println!( + "{}", + serde_json::json!({ + "catalog_id": task.context.catalog.to_string(), + "task_id": task.identity.to_string(), + "kind": format!("{:?}", task.kind), + "phase": format!("{:?}", task.phase), + "revision": task.revision, + "paused": task.paused, + "stalled": format!("{:?}", task.stalled), + "attempts": task.attempts, + "retry_at_ms": task.retry_at_ms, + "marked": task.marked, + "deleted": task.deleted, + "reclaimed_bytes": task.reclaimed_bytes, + "deferred_ranges": task.deferred_ranges, + }) + ); +} diff --git a/app/crowdb-access-server/src/iceberg/gc_runtime.rs b/app/crowdb-access-server/src/iceberg/gc_runtime.rs new file mode 100644 index 000000000..abafaa9ea --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/gc_runtime.rs @@ -0,0 +1,355 @@ +use std::{sync::Arc, time::Duration}; + +use crowdb_access_iceberg::{ + catalog::{CatalogContext, CatalogRepository, RootState, RoutedCatalogStore}, + file::FileBlockStore, + gc::{GcLimits, GcPhase, GcRepository, GcScan, GcStore, GcSystemScan, GcTaskKind, GcWorker}, + key::{CatalogId, CatalogScope, IcebergKey, SystemScope}, + operation::{ManagementAction, ManagementPhase}, + record::StorageRecord, +}; +use crowdb_chunk_client::ChunkIoClient; + +pub(super) mod budget; + +#[derive(Clone, Default)] +struct ScanPosition { + task: Vec, + purge: Vec, + system: Vec, + retired_turn: bool, +} + +pub(super) struct GcRuntimeConfig { + pub limits: GcLimits, + pub interval_ms: u64, + pub catalogs: Vec, + pub enabled: bool, + pub kv_bytes: u64, + pub kv_requests: u32, + pub chunk_bytes: u64, + pub chunk_requests: u32, +} + +impl GcRuntimeConfig { + pub fn from_env() -> Result> { + let mut limits = GcLimits::default(); + limits.step_bytes = setting("CROWDB_ICEBERG_GC_STEP_BYTES", limits.step_bytes)?; + limits.step_ms = setting("CROWDB_ICEBERG_GC_STEP_MS", limits.step_ms)?; + limits.page_items = setting("CROWDB_ICEBERG_GC_PAGE_ITEMS", limits.page_items)?; + limits.page_bytes = setting("CROWDB_ICEBERG_GC_PAGE_BYTES", limits.page_bytes)?; + limits.concurrency = setting("CROWDB_ICEBERG_GC_CONCURRENCY", limits.concurrency)?; + limits.retry_base_ms = setting("CROWDB_ICEBERG_GC_RETRY_BASE_MS", limits.retry_base_ms)?; + limits.retry_max_ms = setting("CROWDB_ICEBERG_GC_RETRY_MAX_MS", limits.retry_max_ms)?; + limits.corruption_attempts = setting( + "CROWDB_ICEBERG_GC_CORRUPTION_ATTEMPTS", + limits.corruption_attempts, + )?; + limits.minimum_retention_ms = setting( + "CROWDB_ICEBERG_GC_MINIMUM_RETENTION_MS", + limits.minimum_retention_ms, + )?; + limits.validate()?; + if limits.concurrency != 1 { + return Err("GC scheduler currently supports one concurrent step".into()); + } + if limits.minimum_retention_ms < GcLimits::default().minimum_retention_ms { + return Err("GC retention must be at least seven days".into()); + } + let interval_ms = setting("CROWDB_ICEBERG_GC_INTERVAL_MS", 1000_u64)?; + if !(100..=60_000).contains(&interval_ms) { + return Err("GC interval must be between 100 and 60000 milliseconds".into()); + } + let catalogs = match std::env::var("CROWDB_ICEBERG_GC_CATALOGS") { + Ok(value) => value, + Err(std::env::VarError::NotPresent) => String::new(), + Err(error) => return Err(error.into()), + } + .split(',') + .map(str::trim) + .filter(|value| !value.is_empty()) + .map(str::parse) + .collect::, _>>()?; + if catalogs.len() > 64 { + return Err("too many GC catalog scopes".into()); + } + let mut catalogs: Vec = catalogs; + catalogs.sort_unstable(); + catalogs.dedup(); + let enabled = match std::env::var("CROWDB_ICEBERG_GC_ENABLED").as_deref() { + Ok("1") => true, + Ok("0") | Err(std::env::VarError::NotPresent) => false, + _ => return Err("CROWDB_ICEBERG_GC_ENABLED must be 0 or 1".into()), + }; + let kv_bytes = setting("CROWDB_ICEBERG_GC_KV_BYTES", 64 * 1024 * 1024_u64)?; + let kv_requests = setting("CROWDB_ICEBERG_GC_KV_REQUESTS", 128_u32)?; + let chunk_bytes = setting("CROWDB_ICEBERG_GC_CHUNK_BYTES", 8 * 1024 * 1024_u64)?; + let chunk_requests = setting("CROWDB_ICEBERG_GC_CHUNK_REQUESTS", 128_u32)?; + if !(4 * 1024 * 1024..=256 * 1024 * 1024).contains(&kv_bytes) + || !(8..=4096).contains(&kv_requests) + || !(256 * 1024..=64 * 1024 * 1024).contains(&chunk_bytes) + || !(1..=4096).contains(&chunk_requests) + { + return Err("GC KV or chunk I/O budget is outside supported bounds".into()); + } + Ok(Self { + limits, + interval_ms, + catalogs, + enabled, + kv_bytes, + kv_requests, + chunk_bytes, + chunk_requests, + }) + } +} + +fn setting(name: &str, default: T) -> Result> +where + T: std::str::FromStr, + T::Err: std::error::Error + Send + Sync + 'static, +{ + match std::env::var(name) { + Ok(value) => Ok(value.parse()?), + Err(std::env::VarError::NotPresent) => Ok(default), + Err(error) => Err(error.into()), + } +} + +pub(super) async fn run( + catalog: Arc, + store: Arc, + chunks: ChunkIoClient, + config: GcRuntimeConfig, +) { + if !config.enabled { + return std::future::pending().await; + } + let budget = Arc::new(budget::GcIoBudget::new(&config)); + let metered_store = Arc::new(budget::BudgetedGcStore::new(store.clone(), budget.clone())); + let native = Arc::new(crowdb_access_iceberg::file::NativeFileBlocks::new( + chunks, + metered_store.clone(), + )); + let blocks: Arc = Arc::new(budget::BudgetedGcBlocks::new(native, budget.clone())); + let repository = GcRepository::new(metered_store.clone()); + let worker = match GcWorker::new(repository, blocks, config.limits) { + Ok(worker) => worker, + Err(error) => { + tracing::error!(%error, "GC worker configuration invalid; background processing stopped"); + return; + } + }; + let mut interval = tokio::time::interval(Duration::from_millis(config.interval_ms)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + let mut cursors = Vec::<(CatalogId, ScanPosition)>::new(); + let mut index = 0_usize; + loop { + interval.tick().await; + let Ok(Ok((root, _))) = + tokio::time::timeout(Duration::from_millis(config.interval_ms), catalog.status()).await + else { + tracing::warn!("GC catalog status unavailable; retrying later"); + continue; + }; + if root.state != RootState::Ready { + continue; + } + let mut catalogs = config.catalogs.clone(); + if !catalogs.contains(&root.context.catalog) { + catalogs.push(root.context.catalog); + } + if catalogs.is_empty() { + continue; + } + let selected = catalogs[index % catalogs.len()]; + index = index.wrapping_add(1); + let cursor = cursors.iter_mut().find(|(catalog, _)| *catalog == selected); + let after = cursor + .as_ref() + .map_or_else(ScanPosition::default, |(_, after)| after.clone()); + let result = tokio::time::timeout( + Duration::from_millis(u64::from(config.limits.step_ms) * 2 + 1000), + scan_and_advance( + metered_store.clone(), + &worker, + budget.as_ref(), + selected, + after, + (selected == root.context.catalog).then_some(root.context), + config.limits, + ), + ) + .await; + match result { + Ok(Ok(next)) => { + if let Some((_, cursor)) = cursor { + *cursor = next; + } else { + cursors.push((selected, next)); + } + } + Ok(Err(error)) => { + tracing::error!(catalog = %selected, %error, "GC scheduler scan failed; retrying catalog"); + } + Err(_) => { + tracing::warn!(catalog = %selected, "GC scheduler scan exceeded its budget"); + } + } + } +} + +async fn scan_and_advance( + store: Arc, + worker: &GcWorker, + budget: &budget::GcIoBudget, + catalog: CatalogId, + after: ScanPosition, + active: Option, + limits: GcLimits, +) -> Result> { + budget.reset(); + let next_purge = if let Some(context) = active { + scan_purge(store.clone(), context, after.purge, limits).await? + } else { + Vec::new() + }; + let (next_system, advanced_retired) = if active.is_some() && after.retired_turn { + scan_retired(store.clone(), worker, after.system, limits).await? + } else { + (after.system, false) + }; + let retired_turn = !after.retired_turn; + if advanced_retired { + return Ok(ScanPosition { + task: after.task, + purge: next_purge, + system: next_system, + retired_turn, + }); + } + let scan = GcScan { + catalog, + scope: Some(CatalogScope::GcTask), + prefix: Vec::new(), + after: after.task, + items: usize::from(limits.page_items), + bytes: limits.page_bytes as usize, + }; + let page = store.scan_gc(scan.clone()).await?; + scan.validate_page(&page)?; + let mut next = Vec::new(); + for item in page.items { + next = item.key.clone(); + let key = IcebergKey::decode(&item.key)?; + let StorageRecord::GcTask(task) = StorageRecord::decode(&key, &item.value)? else { + return Err("GC task scan encountered a non-task record".into()); + }; + if task.kind == GcTaskKind::LiveTable && task.phase != GcPhase::Complete { + match GcRepository::new(store.clone()).retire_live(&task).await { + Ok(progress) => { + tracing::info!(catalog = %catalog, task = %task.identity, phase = ?progress.phase, "retired legacy live GC task"); + } + Err(error) => { + tracing::error!(catalog = %catalog, task = %task.identity, %error, "legacy live GC task still requires fence recovery"); + } + } + break; + } + if !matches!(task.phase, GcPhase::Complete | GcPhase::Quarantined) && !task.paused { + let now_ms = super::runtime::now_ms()?; + if now_ms >= task.retry_at_ms { + match worker.run(&task, now_ms).await { + Ok(progress) => { + tracing::debug!(catalog = %catalog, task = %task.identity, phase = ?progress.phase, "GC task advanced"); + } + Err(error) => { + tracing::error!(catalog = %catalog, task = %task.identity, %error, "GC task retained for retry"); + } + } + break; + } + } + } + Ok(ScanPosition { + task: next, + purge: next_purge, + system: next_system, + retired_turn, + }) +} + +async fn scan_retired( + store: Arc, + worker: &GcWorker, + after: Vec, + limits: GcLimits, +) -> Result<(Vec, bool), Box> { + let scan = GcSystemScan { + after, + items: 1, + bytes: limits.page_bytes as usize, + }; + let page = store.scan_gc_system(scan.clone()).await?; + scan.validate_page(&page)?; + let Some(item) = page.items.first() else { + return Ok((Vec::new(), false)); + }; + let mut advanced = false; + let key = IcebergKey::decode(&item.key)?; + if matches!( + key, + IcebergKey::System { + scope: SystemScope::ManagementOperation, + .. + } + ) { + let StorageRecord::Management(operation) = StorageRecord::decode(&key, &item.value)? else { + return Err("management scan encountered a non-management record".into()); + }; + if operation.request.action == ManagementAction::Clear && operation.phase == ManagementPhase::Complete + { + let task = GcRepository::new(store) + .admit_retired(&operation, super::runtime::now_ms()?, limits) + .await?; + if !matches!(task.phase, GcPhase::Complete | GcPhase::Quarantined) && !task.paused { + let now_ms = super::runtime::now_ms()?; + if now_ms >= task.retry_at_ms { + worker.run(&task, now_ms).await?; + advanced = true; + } + } + } + } + Ok((item.key.clone(), advanced)) +} + +async fn scan_purge( + store: Arc, + context: CatalogContext, + after: Vec, + limits: GcLimits, +) -> Result, Box> { + let scan = GcScan { + catalog: context.catalog, + scope: Some(CatalogScope::Reclamation), + prefix: Vec::new(), + after, + items: 1, + bytes: limits.page_bytes as usize, + }; + let page = store.scan_gc(scan.clone()).await?; + scan.validate_page(&page)?; + let Some(item) = page.items.first() else { + return Ok(Vec::new()); + }; + let key = IcebergKey::decode(&item.key)?; + let StorageRecord::TablePurgeTask(marker) = StorageRecord::decode(&key, &item.value)? else { + return Err("purge scan encountered a non-purge record".into()); + }; + GcRepository::new(store) + .admit_purge(context, &marker, super::runtime::now_ms()?, limits) + .await?; + Ok(item.key.clone()) +} diff --git a/app/crowdb-access-server/src/iceberg/gc_runtime/budget.rs b/app/crowdb-access-server/src/iceberg/gc_runtime/budget.rs new file mode 100644 index 000000000..e7f0ff506 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/gc_runtime/budget.rs @@ -0,0 +1,227 @@ +use std::sync::{ + atomic::{AtomicU32, AtomicU64, Ordering}, + Arc, +}; + +use async_trait::async_trait; +use crowdb_access_iceberg::{ + catalog::{CasOutcome, CatalogStore, StoreError, StoredValue}, + file::{ChunkRoot, FileBlockStore, FileIdentity, FileIoError}, + gc::{GcScan, GcStore, GcSystemScan}, + key::{CatalogScope, IcebergKey}, + record::MAX_RECORD_BYTES, +}; +use crowdb_chunk_client::ReclaimOutcome; +use crowdb_chunk_kv_client::MultiScanPage; +use crowdb_protocol::chunk_kv::ClientRequestId; + +use super::GcRuntimeConfig; + +pub struct GcIoBudget { + kv_bytes: AtomicU64, + kv_requests: AtomicU32, + chunk_bytes: AtomicU64, + chunk_requests: AtomicU32, + recovery_bytes: AtomicU64, + recovery_requests: AtomicU32, + max_kv_bytes: u64, + max_kv_requests: u32, + max_chunk_bytes: u64, + max_chunk_requests: u32, +} + +impl GcIoBudget { + pub(super) fn new(config: &GcRuntimeConfig) -> Self { + Self { + kv_bytes: AtomicU64::new(0), + kv_requests: AtomicU32::new(0), + chunk_bytes: AtomicU64::new(0), + chunk_requests: AtomicU32::new(0), + recovery_bytes: AtomicU64::new(0), + recovery_requests: AtomicU32::new(0), + max_kv_bytes: config.kv_bytes, + max_kv_requests: config.kv_requests, + max_chunk_bytes: config.chunk_bytes, + max_chunk_requests: config.chunk_requests, + } + } + + #[cfg(feature = "test-util")] + #[must_use] + pub fn for_tests(kv_bytes: u64, kv_requests: u32, chunk_bytes: u64, chunk_requests: u32) -> Self { + Self { + kv_bytes: AtomicU64::new(0), + kv_requests: AtomicU32::new(0), + chunk_bytes: AtomicU64::new(0), + chunk_requests: AtomicU32::new(0), + recovery_bytes: AtomicU64::new(0), + recovery_requests: AtomicU32::new(0), + max_kv_bytes: kv_bytes, + max_kv_requests: kv_requests, + max_chunk_bytes: chunk_bytes, + max_chunk_requests: chunk_requests, + } + } + + pub fn reset(&self) { + self.kv_bytes.store(0, Ordering::Release); + self.kv_requests.store(0, Ordering::Release); + self.chunk_bytes.store(0, Ordering::Release); + self.chunk_requests.store(0, Ordering::Release); + self.recovery_bytes.store(0, Ordering::Release); + self.recovery_requests.store(0, Ordering::Release); + } + + fn reserve_kv(&self, bytes: usize) -> Result<(), StoreError> { + reserve(&self.kv_requests, 1, self.max_kv_requests).map_err(|()| StoreError::Budget)?; + reserve(&self.kv_bytes, bytes as u64, self.max_kv_bytes).map_err(|()| StoreError::Budget) + } + + fn reserve_kv_key(&self, key: &[u8], bytes: usize) -> Result<(), StoreError> { + match self.reserve_kv(bytes) { + Ok(()) => Ok(()), + Err(error) if is_task_key(key) => { + reserve(&self.recovery_requests, 1, 16).map_err(|()| error)?; + reserve(&self.recovery_bytes, bytes as u64, 2 * 1024 * 1024).map_err(|()| StoreError::Budget) + } + Err(error) => Err(error), + } + } + + fn reserve_chunk(&self, bytes: u64) -> Result<(), FileIoError> { + reserve(&self.chunk_requests, 1, self.max_chunk_requests).map_err(|()| FileIoError::Bounds)?; + reserve(&self.chunk_bytes, bytes, self.max_chunk_bytes).map_err(|()| FileIoError::Bounds) + } +} + +fn is_task_key(key: &[u8]) -> bool { + matches!( + IcebergKey::decode(key), + Ok(IcebergKey::Catalog { + scope: CatalogScope::GcTask, + .. + }) + ) +} + +fn reserve(counter: &Counter, amount: Counter::Value, maximum: Counter::Value) -> Result<(), ()> +where + Counter: BudgetCounter, +{ + counter.reserve(amount, maximum) +} + +trait BudgetCounter { + type Value: Copy; + fn reserve(&self, amount: Self::Value, maximum: Self::Value) -> Result<(), ()>; +} + +impl BudgetCounter for AtomicU32 { + type Value = u32; + + fn reserve(&self, amount: u32, maximum: u32) -> Result<(), ()> { + self.fetch_update(Ordering::AcqRel, Ordering::Acquire, |used| { + used.checked_add(amount).filter(|next| *next <= maximum) + }) + .map(|_| ()) + .map_err(|_| ()) + } +} + +impl BudgetCounter for AtomicU64 { + type Value = u64; + + fn reserve(&self, amount: u64, maximum: u64) -> Result<(), ()> { + self.fetch_update(Ordering::AcqRel, Ordering::Acquire, |used| { + used.checked_add(amount).filter(|next| *next <= maximum) + }) + .map(|_| ()) + .map_err(|_| ()) + } +} + +pub struct BudgetedGcStore { + inner: Arc, + budget: Arc, +} + +impl BudgetedGcStore { + pub fn new(inner: Arc, budget: Arc) -> Self { + Self { inner, budget } + } +} + +#[async_trait] +impl CatalogStore for BudgetedGcStore { + async fn get(&self, key: &[u8]) -> Result, StoreError> { + self.budget.reserve_kv_key(key, key.len() + MAX_RECORD_BYTES)?; + self.inner.get(key).await + } + + async fn compare_exchange( + &self, + key: &[u8], + expected: Option<&[u8]>, + value: &[u8], + identity: ClientRequestId, + ) -> Result { + self.budget.reserve_kv_key( + key, + key.len() + expected.map_or(0, <[u8]>::len) + value.len() + MAX_RECORD_BYTES, + )?; + self.inner.compare_exchange(key, expected, value, identity).await + } +} + +#[async_trait] +impl GcStore for BudgetedGcStore { + async fn scan_gc(&self, request: GcScan) -> Result { + self.budget.reserve_kv(request.bytes)?; + self.inner.scan_gc(request).await + } + + async fn scan_gc_system(&self, request: GcSystemScan) -> Result { + self.budget.reserve_kv(request.bytes)?; + self.inner.scan_gc_system(request).await + } + + async fn delete_gc_record( + &self, + key: &[u8], + expected: &[u8], + identity: ClientRequestId, + ) -> Result { + self.budget + .reserve_kv(key.len() + expected.len() + MAX_RECORD_BYTES)?; + self.inner.delete_gc_record(key, expected, identity).await + } +} + +pub struct BudgetedGcBlocks { + inner: Arc, + budget: Arc, +} + +impl BudgetedGcBlocks { + pub fn new(inner: Arc, budget: Arc) -> Self { + Self { inner, budget } + } +} + +#[async_trait] +impl FileBlockStore for BudgetedGcBlocks { + async fn put(&self, owner: FileIdentity, height: u8, bytes: &[u8]) -> Result { + self.budget.reserve_chunk(bytes.len() as u64)?; + self.inner.put(owner, height, bytes).await + } + + async fn read(&self, root: &ChunkRoot) -> Result, FileIoError> { + self.budget.reserve_chunk(root.logical_length)?; + self.inner.read(root).await + } + + async fn reclaim(&self, root: &ChunkRoot) -> Result { + self.budget.reserve_chunk(root.physical_length)?; + self.inner.reclaim(root).await + } +} diff --git a/app/crowdb-access-server/src/iceberg/http.rs b/app/crowdb-access-server/src/iceberg/http.rs new file mode 100644 index 000000000..a823f1631 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/http.rs @@ -0,0 +1,458 @@ +use std::convert::Infallible; +use std::future::Future; +use std::sync::Arc; +use std::time::Duration; + +use super::body::IcebergBody; +use super::connection::{ActiveIo, ConnectionActivity}; +use super::file_http::FileHttp; +use super::metrics::{self, IcebergMetrics, IcebergMetricsSnapshot, RequestObservation}; +use super::namespace_read::NamespaceHttp; +use super::routes::{InstalledRoutes, Route}; +use crowdb_access_iceberg::catalog::{ + Capabilities, CatalogError, CatalogLifecycle, CatalogRepository, ManagementPrivilege, RootState, +}; +use crowdb_access_iceberg::wire::{BearerAuthenticator, CatalogConfig, IcebergErrorResponse}; +use hyper::body::Incoming; +use hyper::server::conn::http1; +use hyper::service::service_fn; +use hyper::{Request, Response, StatusCode}; +use hyper_util::rt::TokioIo; +use tokio::net::TcpListener; +use tokio::task::JoinSet; + +pub struct IcebergHttpService { + repository: Arc, + authentication: BearerAuthenticator, + request_timeout: Duration, + namespaces: Option, + files: Option>, + tables: Option, + table_writes: Option, + table_credentials: Option, + metrics: Arc, +} + +impl IcebergHttpService { + #[must_use] + pub fn new( + repository: Arc, + authentication: BearerAuthenticator, + request_timeout: Duration, + ) -> Self { + Self { + repository, + authentication, + request_timeout, + namespaces: None, + files: None, + tables: None, + table_writes: None, + table_credentials: None, + metrics: Arc::new(IcebergMetrics::default()), + } + } + + #[must_use] + pub fn metrics_snapshot(&self) -> IcebergMetricsSnapshot { + self.metrics.snapshot() + } + + /// # Errors + /// Rejects invalid native file listener limits or signing configuration. + pub fn with_fileio( + mut self, + store: Arc, + blocks: Arc, + region: String, + ) -> Result { + self.files = Some(Arc::new(FileHttp::new( + store, + blocks, + self.authentication.namespace_token_key(), + region, + )?)); + Ok(self) + } + + /// # Errors + /// Rejects invalid namespace token signing configuration. + pub fn with_namespaces( + mut self, + store: Arc, + ) -> Result { + self.namespaces = Some(NamespaceHttp::new( + store, + &self.authentication.namespace_token_key(), + )?); + Ok(self) + } + + /// Installs and advertises only generation-qualified reads for fixture-backed tests. + /// Runtime activation awaits complete commit validation and credential vending. + /// # Errors + /// Rejects invalid table-list token signing configuration. + #[cfg(feature = "test-util")] + pub fn with_table_reads_for_tests( + mut self, + store: Arc, + blocks: Arc, + ) -> Result { + self.tables = Some(super::table_read::TableHttp::new( + store, + blocks, + &self.authentication.namespace_token_key(), + )?); + Ok(self) + } + + /// # Errors + /// Rejects invalid table token configuration. + pub fn with_tables( + mut self, + store: Arc, + blocks: Arc, + ) -> Result { + self.tables = Some(super::table_read::TableHttp::new( + store.clone(), + blocks.clone(), + &self.authentication.namespace_token_key(), + )?); + self.table_writes = Some(super::table_write::TableWrites::new(store, blocks)); + Ok(self) + } + + /// # Errors + /// Requires installed table access and a configured external HTTP/S origin. + pub fn with_table_credentials( + mut self, + store: Arc, + endpoint: String, + ) -> Result { + let config = super::table_credentials::TableFileConfig::new(endpoint)?; + self.tables + .as_mut() + .ok_or(crowdb_access_iceberg::error::ValidationError::Record)? + .file_config = Some(config.clone()); + self.table_writes + .as_mut() + .ok_or(crowdb_access_iceberg::error::ValidationError::Record)? + .file_config = Some(config); + self.table_credentials = Some( + super::table_credentials::TableCredentials::new(store, self.authentication.namespace_token_key()) + .map_err(|_| crowdb_access_iceberg::error::ValidationError::Record)?, + ); + Ok(self) + } + + async fn handle( + &self, + request: Request, + deadline: tokio::time::Instant, + ) -> Result, Infallible> { + let observation = RequestObservation::new( + self.metrics.clone(), + metrics::route_index(request.method(), request.uri().path()), + ); + let head = request.method() == hyper::Method::HEAD; + let result = Box::pin(tokio::time::timeout_at( + deadline, + metrics::observe(observation.clone(), self.dispatch(request)), + )) + .await; + let mut response = match result { + Ok(Ok(response)) => response, + Ok(Err(error)) => response(error.error.code, serde_json::to_vec(&error).unwrap_or_default()), + Err(_) => { + tracing::warn!("Iceberg request deadline exhausted; durable recovery remains active"); + unavailable() + } + }; + if head { + *response.body_mut() = IcebergBody::new(Vec::new()); + } + observation.dispatched(response.status().as_u16()); + let body = std::mem::replace(response.body_mut(), IcebergBody::new(Vec::new())); + *response.body_mut() = body.with_observation(observation); + Ok(response) + } + + async fn dispatch( + &self, + request: Request, + ) -> Result, IcebergErrorResponse> { + if request.uri().path().starts_with("/iceberg-") { + return Ok(match &self.files { + Some(files) => { + Box::pin(files.dispatch(&self.repository, request, self.request_timeout)).await + } + None => super::file_http::unavailable(request.uri().path()), + }); + } + let mut authorizations = request.headers().get_all(hyper::header::AUTHORIZATION).iter(); + let authorization = match (authorizations.next(), authorizations.next()) { + (Some(value), None) => value.to_str().unwrap_or_default(), + _ => "", + }; + let Some(principal) = self.authentication.authenticate(authorization) else { + return Err(IcebergErrorResponse::new( + 401, + "NotAuthorizedException", + "Valid bearer authentication is required", + )); + }; + if request.uri().to_string().len() > 32 * 1024 { + return Err(bad_request()); + } + let route = Route::classify(request.method(), request.uri().path()) + .filter(|route| route.enabled(&self.installed_routes())) + .ok_or_else(super::table_read::unsupported)?; + if route == Route::AdminMetrics { + if principal.management != ManagementPrivilege::Manage { + return Err(IcebergErrorResponse::new( + 403, + "ForbiddenException", + "Management privilege is required", + )); + } + return Ok(response( + 200, + serde_json::to_vec(&self.metrics.snapshot()).map_err(|_| service_unavailable())?, + )); + } + let (root, authority) = self + .repository + .status() + .await + .map_err(|_| service_unavailable())?; + if root.state != RootState::Ready + || authority.lifecycle != CatalogLifecycle::Ready + || self.request_timeout.is_zero() + || self.request_timeout > Duration::from_millis(authority.admission_bounds.request_ms) + { + return Err(service_unavailable()); + } + if route == Route::Config { + if authority.capabilities.bits() == 0 { + return Err(service_unavailable()); + } + return self.config(request.uri().query(), authority.capabilities); + } + if !route.supported(authority.capabilities) { + return Err(super::table_read::unsupported()); + } + match route { + Route::TableCredentials => { + self.table_credentials + .as_ref() + .ok_or_else(super::table_read::unsupported)? + .load(&self.repository, root.context, principal, &request) + .await + } + Route::TableCreate | Route::TableUpdate | Route::TableDrop | Route::TableRename => { + let writes = self + .table_writes + .as_ref() + .ok_or_else(super::table_read::unsupported)?; + Box::pin(writes.execute(root.context, authority.capabilities, principal, request)).await + } + Route::TableList | Route::TableLoad | Route::TableExists => { + self.tables + .as_ref() + .ok_or_else(super::table_read::unsupported)? + .read(root.context, authority.capabilities, &request) + .await + } + _ => { + self.namespaces + .as_ref() + .ok_or_else(super::table_read::unsupported)? + .dispatch(root.context, principal, request) + .await + } + } + } + + fn installed_routes(&self) -> InstalledRoutes { + let mut bits = 0; + if self.namespaces.is_some() { + bits |= InstalledRoutes::NAMESPACES; + } + if self.tables.is_some() { + bits |= InstalledRoutes::TABLES; + } + if self.table_writes.is_some() { + bits |= InstalledRoutes::WRITES; + } + if self.table_credentials.is_some() { + bits |= InstalledRoutes::CREDENTIALS; + } + InstalledRoutes(bits) + } + + fn config( + &self, + query: Option<&str>, + capabilities: Capabilities, + ) -> Result, IcebergErrorResponse> { + let warehouse = warehouse(query)?; + let mut config = CatalogConfig::for_capabilities(warehouse.as_deref(), capabilities)?; + config.endpoints = Route::endpoints(&self.installed_routes(), capabilities); + if self.namespaces.is_some() { + config.idempotency_key_lifetime = Some("PT24H".into()); + } + Ok(response( + 200, + serde_json::to_vec(&config).map_err(|_| service_unavailable())?, + )) + } +} + +/// # Errors +/// Returns listener failures after stopping admission and draining connections. +pub async fn serve( + listener: TcpListener, + service: Arc, + shutdown: impl Future, +) -> std::io::Result<()> { + tokio::pin!(shutdown); + let recovery = reconcile(&service.repository); + tokio::pin!(recovery); + let mut connections = JoinSet::new(); + let mut failure = None; + loop { + tokio::select! { + () = &mut shutdown => break, + () = &mut recovery => break, + Some(_) = connections.join_next(), if !connections.is_empty() => {}, + accepted = listener.accept(), if connections.len() < 128 => { + let (stream, peer) = match accepted { Ok(value) => value, Err(error) => { failure = Some(error); break; } }; + let service = Arc::clone(&service); + connections.spawn(async move { + let request_timeout = service.request_timeout; + let activity = ConnectionActivity::new(); + let deadline = activity.dispatch_deadline(request_timeout); + let stream = ActiveIo::new(stream, activity.clone()); + let request_activity = activity.clone(); + let handler = service_fn(move |request| { + request_activity.mark_request_started(); + let service = Arc::clone(&service); + async move { Box::pin(service.handle(request, deadline)).await } + }); + let connection = http1::Builder::new().keep_alive(false).max_buf_size(64 * 1024) + .serve_connection(TokioIo::new(stream), handler); + tokio::select! { + result = connection => { + if let Err(error) = result { + tracing::debug!(%peer, %error, "Iceberg HTTP connection failed"); + } + } + () = activity.expired(Duration::from_secs(300)) => { + tracing::debug!(%peer, "Iceberg HTTP connection idle deadline exhausted"); + } + () = activity.header_expired(deadline) => { + tracing::debug!(%peer, "Iceberg HTTP request header deadline exhausted"); + } + } + }); + } + } + } + drop(listener); + if tokio::time::timeout(Duration::from_secs(300), async { + while connections.join_next().await.is_some() {} + }) + .await + .is_err() + { + connections.abort_all(); + while connections.join_next().await.is_some() {} + } + failure.map_or(Ok(()), Err) +} + +async fn reconcile(repository: &CatalogRepository) { + let mut interval = tokio::time::interval(Duration::from_secs(1)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + loop { + interval.tick().await; + let Ok(elapsed) = std::time::SystemTime::now().duration_since(std::time::UNIX_EPOCH) else { + tracing::error!("Iceberg recovery paused: system clock precedes Unix epoch"); + continue; + }; + let Ok(now_ms) = u64::try_from(elapsed.as_millis()) else { + tracing::error!("Iceberg recovery paused: system clock exceeds supported range"); + continue; + }; + match repository.recover(now_ms).await { + Ok(()) | Err(CatalogError::Busy) => {} + Err(error) => tracing::error!(%error, "Iceberg recovery failed; retrying on next interval"), + } + } +} + +fn warehouse(query: Option<&str>) -> Result, IcebergErrorResponse> { + let mut warehouse = None; + for pair in query + .unwrap_or_default() + .split('&') + .filter(|value| !value.is_empty()) + { + let (name, value) = pair.split_once('=').unwrap_or((pair, "")); + let name = decode_query(name)?; + if name == "warehouse" { + if warehouse.is_some() { + return Err(bad_request()); + } + warehouse = Some(decode_query(value)?); + } + } + Ok(warehouse) +} + +pub(super) fn decode_query(value: &str) -> Result { + for (index, byte) in value.bytes().enumerate() { + if byte == b'%' + && !value + .as_bytes() + .get(index + 1..index + 3) + .is_some_and(|bytes| bytes.iter().all(u8::is_ascii_hexdigit)) + { + return Err(bad_request()); + } + } + let value = value.replace('+', " "); + percent_encoding::percent_decode_str(&value) + .decode_utf8() + .map(std::borrow::Cow::into_owned) + .map_err(|_| bad_request()) +} + +pub(super) fn response(status: u16, bytes: Vec) -> Response { + let mut response = Response::new(IcebergBody::new(bytes)); + *response.status_mut() = StatusCode::from_u16(status).unwrap_or(StatusCode::INTERNAL_SERVER_ERROR); + response.headers_mut().insert( + hyper::header::CONTENT_TYPE, + hyper::header::HeaderValue::from_static("application/json"), + ); + if status == 401 { + response.headers_mut().insert( + hyper::header::WWW_AUTHENTICATE, + hyper::header::HeaderValue::from_static("Bearer"), + ); + } + response +} + +pub(super) fn bad_request() -> IcebergErrorResponse { + IcebergErrorResponse::new(400, "BadRequestException", "Invalid request parameters") +} +pub(super) fn service_unavailable() -> IcebergErrorResponse { + IcebergErrorResponse::new(503, "ServiceUnavailableException", "Catalog is not ready") +} +fn unavailable() -> Response { + response( + 503, + serde_json::to_vec(&service_unavailable()).unwrap_or_default(), + ) +} diff --git a/app/crowdb-access-server/src/iceberg/metrics.rs b/app/crowdb-access-server/src/iceberg/metrics.rs new file mode 100644 index 000000000..02722c428 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/metrics.rs @@ -0,0 +1,227 @@ +use std::{ + array, + sync::{ + atomic::{AtomicU16, AtomicU64, AtomicU8, Ordering}, + Arc, + }, + time::Instant, +}; + +use super::routes::Route; + +const ROUTE_COUNT: usize = 9; +const OUTCOME_COUNT: usize = 7; + +pub const ICEBERG_ROUTE_NAMES: [&str; ROUTE_COUNT] = [ + "config", + "namespace_read", + "namespace_write", + "table_read", + "table_write", + "credentials", + "file", + "admin_metrics", + "unsupported", +]; +pub const ICEBERG_OUTCOME_NAMES: [&str; OUTCOME_COUNT] = [ + "success", + "unauthorized", + "conflict", + "client_error", + "unavailable", + "server_error", + "cancelled", +]; + +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq, serde::Serialize)] +pub struct MetricCounts { + pub requests: u64, + pub request_bytes: u64, + pub response_bytes: u64, + pub dispatch_latency_ns: u64, + pub lifetime_ns: u64, +} + +#[derive(Clone, Debug, Eq, PartialEq, serde::Serialize)] +pub struct IcebergMetricsSnapshot { + pub routes: [[MetricCounts; OUTCOME_COUNT]; ROUTE_COUNT], + pub retry_new: u64, + pub retry_resume: u64, + pub retry_replay: u64, + pub selected_versions: [u64; 3], +} + +struct Counters { + requests: AtomicU64, + request_bytes: AtomicU64, + response_bytes: AtomicU64, + dispatch_latency_ns: AtomicU64, + lifetime_ns: AtomicU64, +} + +impl Counters { + fn new() -> Self { + Self { + requests: AtomicU64::new(0), + request_bytes: AtomicU64::new(0), + response_bytes: AtomicU64::new(0), + dispatch_latency_ns: AtomicU64::new(0), + lifetime_ns: AtomicU64::new(0), + } + } + + fn snapshot(&self) -> MetricCounts { + MetricCounts { + requests: self.requests.load(Ordering::Relaxed), + request_bytes: self.request_bytes.load(Ordering::Relaxed), + response_bytes: self.response_bytes.load(Ordering::Relaxed), + dispatch_latency_ns: self.dispatch_latency_ns.load(Ordering::Relaxed), + lifetime_ns: self.lifetime_ns.load(Ordering::Relaxed), + } + } +} + +pub(super) struct IcebergMetrics { + routes: [[Counters; OUTCOME_COUNT]; ROUTE_COUNT], + retry: [AtomicU64; 3], + selected_versions: [AtomicU64; 3], +} + +impl Default for IcebergMetrics { + fn default() -> Self { + Self { + routes: array::from_fn(|_| array::from_fn(|_| Counters::new())), + retry: array::from_fn(|_| AtomicU64::new(0)), + selected_versions: array::from_fn(|_| AtomicU64::new(0)), + } + } +} + +impl IcebergMetrics { + pub(super) fn snapshot(&self) -> IcebergMetricsSnapshot { + IcebergMetricsSnapshot { + routes: array::from_fn(|route| array::from_fn(|outcome| self.routes[route][outcome].snapshot())), + retry_new: self.retry[0].load(Ordering::Relaxed), + retry_resume: self.retry[1].load(Ordering::Relaxed), + retry_replay: self.retry[2].load(Ordering::Relaxed), + selected_versions: array::from_fn(|index| self.selected_versions[index].load(Ordering::Relaxed)), + } + } +} + +pub(super) struct RequestObservation { + metrics: Arc, + route: usize, + started: Instant, + request_bytes: AtomicU64, + response_bytes: AtomicU64, + dispatch_ns: AtomicU64, + status: AtomicU16, + retry: AtomicU8, + version: AtomicU8, +} + +impl RequestObservation { + pub(super) fn new(metrics: Arc, route: usize) -> Arc { + Arc::new(Self { + metrics, + route, + started: Instant::now(), + request_bytes: AtomicU64::new(0), + response_bytes: AtomicU64::new(0), + dispatch_ns: AtomicU64::new(0), + status: AtomicU16::new(0), + retry: AtomicU8::new(0), + version: AtomicU8::new(0), + }) + } + + pub(super) fn dispatched(&self, status: u16) { + self.status.store(status, Ordering::Relaxed); + self.dispatch_ns + .store(elapsed_ns(self.started), Ordering::Relaxed); + } + + pub(super) fn response_bytes(&self, length: usize) { + self.response_bytes.fetch_add(length as u64, Ordering::Relaxed); + } +} + +impl Drop for RequestObservation { + fn drop(&mut self) { + let status = self.status.load(Ordering::Relaxed); + let outcome = match status { + 0 => 6, + 200..=399 => 0, + 401 | 403 => 1, + 409 | 412 => 2, + 400..=499 => 3, + 503 | 504 => 4, + _ => 5, + }; + let counters = &self.metrics.routes[self.route][outcome]; + counters.requests.fetch_add(1, Ordering::Relaxed); + counters + .request_bytes + .fetch_add(self.request_bytes.load(Ordering::Relaxed), Ordering::Relaxed); + counters + .response_bytes + .fetch_add(self.response_bytes.load(Ordering::Relaxed), Ordering::Relaxed); + counters + .dispatch_latency_ns + .fetch_add(self.dispatch_ns.load(Ordering::Relaxed), Ordering::Relaxed); + counters + .lifetime_ns + .fetch_add(elapsed_ns(self.started), Ordering::Relaxed); + let retry = self.retry.load(Ordering::Relaxed); + if retry != 0 { + self.metrics.retry[usize::from(retry - 1)].fetch_add(1, Ordering::Relaxed); + } + let version = self.version.load(Ordering::Relaxed); + if (1..=3).contains(&version) { + self.metrics.selected_versions[usize::from(version - 1)].fetch_add(1, Ordering::Relaxed); + } + } +} + +fn elapsed_ns(started: Instant) -> u64 { + u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX) +} + +tokio::task_local! { + static REQUEST_OBSERVATION: Arc; +} + +pub(super) async fn observe(span: Arc, future: F) -> F::Output { + REQUEST_OBSERVATION.scope(span, future).await +} + +pub(super) fn record_request_bytes(length: usize) { + let _ = REQUEST_OBSERVATION.try_with(|span| { + span.request_bytes.fetch_add(length as u64, Ordering::Relaxed); + }); +} + +pub(super) fn record_retry(kind: u8) { + let _ = REQUEST_OBSERVATION.try_with(|span| span.retry.store(kind, Ordering::Relaxed)); +} + +pub(super) fn record_selected_version(version: u8) { + let _ = REQUEST_OBSERVATION.try_with(|span| span.version.store(version, Ordering::Relaxed)); +} + +pub(super) fn route_index(method: &hyper::Method, path: &str) -> usize { + if path.starts_with("/iceberg-") { + return 6; + } + match Route::classify(method, path) { + Some(Route::Config) => 0, + Some(Route::AdminMetrics) => 7, + Some(Route::NamespaceList | Route::NamespaceLoad | Route::NamespaceExists) => 1, + Some(Route::NamespaceCreate | Route::NamespaceProperties | Route::NamespaceDrop) => 2, + Some(Route::TableList | Route::TableLoad | Route::TableExists) => 3, + Some(Route::TableCreate | Route::TableUpdate | Route::TableDrop | Route::TableRename) => 4, + Some(Route::TableCredentials) => 5, + None => 8, + } +} diff --git a/app/crowdb-access-server/src/iceberg/namespace_read.rs b/app/crowdb-access-server/src/iceberg/namespace_read.rs new file mode 100644 index 000000000..20906a77c --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/namespace_read.rs @@ -0,0 +1,206 @@ +use std::sync::{atomic::AtomicUsize, Arc}; + +use crowdb_access_iceberg::catalog::{CatalogContext, CatalogError}; +use crowdb_access_iceberg::namespace::{ + NamespaceIdentifier, NamespaceLister, NamespaceRepository, NamespaceStore, +}; +use crowdb_access_iceberg::wire::IcebergErrorResponse; +use hyper::{Method, Response, Uri}; + +use super::body::{IcebergBody, SpoolPermit}; +use super::http::{bad_request, decode_query, response, service_unavailable}; + +pub(super) struct NamespaceHttp { + repository: NamespaceRepository, + lister: NamespaceLister, + spools: Arc, + writes: super::namespace_write::NamespaceWrites, +} + +impl NamespaceHttp { + pub(super) fn new( + store: Arc, + secret: &[u8; 32], + ) -> Result { + Ok(Self { + writes: super::namespace_write::NamespaceWrites::new(store.clone()), + repository: NamespaceRepository::new(store.clone()), + lister: NamespaceLister::new(store, secret)?, + spools: Arc::new(AtomicUsize::new(0)), + }) + } + + pub(super) async fn dispatch( + &self, + context: CatalogContext, + principal: crowdb_access_iceberg::wire::Principal, + request: hyper::Request, + ) -> Result, IcebergErrorResponse> { + if request.method() == Method::GET || request.method() == Method::HEAD { + self.read(context, request.method(), request.uri()).await + } else { + self.writes.execute(context, principal, request).await + } + } + + pub(super) async fn read( + &self, + context: CatalogContext, + method: &Method, + uri: &Uri, + ) -> Result, IcebergErrorResponse> { + if method == Method::GET && uri.path() == "/v1/namespaces" { + return self.list(context, uri.query()).await; + } + if method != Method::GET && method != Method::HEAD { + return Err(unsupported()); + } + let encoded = uri + .path() + .strip_prefix("/v1/namespaces/") + .ok_or_else(unsupported)?; + if encoded.is_empty() || encoded.contains('/') || uri.query().is_some() { + return Err(bad_request()); + } + let identifier = NamespaceIdentifier::from_rest(&decode_path(encoded)?).map_err(|_| bad_request())?; + let authority = self + .repository + .load(context, &identifier) + .await + .map_err(|error| storage_error(&error))? + .ok_or_else(not_found)?; + let bytes = if method == Method::HEAD { + Vec::new() + } else { + serde_json::to_vec(&serde_json::json!({"namespace": authority.identifier.components(), "properties": authority.properties.entries()})).map_err(|_| service_unavailable())? + }; + Ok(response(if method == Method::HEAD { 204 } else { 200 }, bytes)) + } + + async fn list( + &self, + context: CatalogContext, + query: Option<&str>, + ) -> Result, IcebergErrorResponse> { + let (parent, limit, token) = parameters(query)?; + if let Some(token) = token { + let page = self + .lister + .page(context, parent.as_ref(), limit, &token) + .await + .map_err(|error| storage_error(&error))? + .ok_or_else(not_found)?; + let namespaces: Vec<_> = page + .namespaces + .iter() + .map(NamespaceIdentifier::components) + .collect(); + let bytes = serde_json::to_vec( + &serde_json::json!({"namespaces": namespaces, "next-page-token": page.next_page_token}), + ) + .map_err(|_| service_unavailable())?; + if bytes.len() > 2 * 1024 * 1024 { + return Err(service_unavailable()); + } + return Ok(response(200, bytes)); + } + let permit = SpoolPermit::acquire(&self.spools).ok_or_else(service_unavailable)?; + let mut bytes = b"{\"namespaces\":[".to_vec(); + let mut token = String::new(); + let mut items = 0; + let mut scanned = 0; + loop { + let page = self + .lister + .page(context, parent.as_ref(), 100, &token) + .await + .map_err(|error| storage_error(&error))? + .ok_or_else(not_found)?; + scanned += page.scanned; + if scanned > 4096 { + return Err(service_unavailable()); + } + for identifier in page.namespaces { + let encoded = + serde_json::to_vec(identifier.components()).map_err(|_| service_unavailable())?; + items += 1; + if items > 1024 || bytes.len() + encoded.len() + 64 > 2 * 1024 * 1024 { + return Err(service_unavailable()); + } + if items > 1 { + bytes.push(b','); + } + bytes.extend_from_slice(&encoded); + } + match page.next_page_token { + Some(next) => token = next, + None => break, + } + } + bytes.extend_from_slice(b"],\"next-page-token\":null}"); + let mut result = response(200, Vec::new()); + *result.body_mut() = IcebergBody::with_permit(bytes, permit); + Ok(result) + } +} + +fn parameters( + query: Option<&str>, +) -> Result<(Option, usize, Option), IcebergErrorResponse> { + let mut values = std::collections::BTreeMap::new(); + for pair in query + .unwrap_or_default() + .split('&') + .filter(|pair| !pair.is_empty()) + { + let (name, value) = pair.split_once('=').unwrap_or((pair, "")); + let name = decode_query(name)?; + if !matches!(name.as_str(), "parent" | "pageSize" | "pageToken") + || values.insert(name, decode_query(value)?).is_some() + { + return Err(bad_request()); + } + } + let parent = values + .remove("parent") + .filter(|value| !value.is_empty()) + .map(|value| NamespaceIdentifier::from_rest(&value)) + .transpose() + .map_err(|_| bad_request())?; + let limit = values + .remove("pageSize") + .map(|value| value.parse::()) + .transpose() + .map_err(|_| bad_request())? + .unwrap_or(100); + if limit == 0 || limit > 100 { + return Err(bad_request()); + } + Ok((parent, limit, values.remove("pageToken"))) +} + +pub(super) fn decode_path(value: &str) -> Result { + decode_query(&value.replace('+', "%2B")) +} + +fn not_found() -> IcebergErrorResponse { + IcebergErrorResponse::new(404, "NoSuchNamespaceException", "Namespace does not exist") +} +fn unsupported() -> IcebergErrorResponse { + IcebergErrorResponse::new( + 406, + "UnsupportedOperationException", + "This endpoint is not implemented", + ) +} +fn storage_error(error: &CatalogError) -> IcebergErrorResponse { + match error { + CatalogError::Invalid( + crowdb_access_iceberg::error::ValidationError::Key + | crowdb_access_iceberg::error::ValidationError::KeyTooLarge + | crowdb_access_iceberg::error::ValidationError::IdentityMismatch + | crowdb_access_iceberg::error::ValidationError::Text, + ) => bad_request(), + _ => service_unavailable(), + } +} diff --git a/app/crowdb-access-server/src/iceberg/namespace_request.rs b/app/crowdb-access-server/src/iceberg/namespace_request.rs new file mode 100644 index 000000000..8e771afce --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/namespace_request.rs @@ -0,0 +1,81 @@ +use std::collections::BTreeMap; + +use crowdb_access_iceberg::namespace::{NamespaceIdentifier, NamespaceProperties, PropertyChanges}; +use crowdb_access_iceberg::wire::IcebergErrorResponse; +use hyper::{Method, Uri}; +use serde::Deserialize; + +use super::http::bad_request; +use super::namespace_read::decode_path; + +pub(super) enum NamespaceMutation { + Create(NamespaceIdentifier, NamespaceProperties), + Update(NamespaceIdentifier, PropertyChanges), + Drop(NamespaceIdentifier), +} + +#[derive(Deserialize)] +struct CreateBody { + namespace: Vec, + #[serde(default)] + properties: BTreeMap, +} + +#[derive(Deserialize)] +struct UpdateBody { + #[serde(default)] + removals: Vec, + #[serde(default)] + updates: BTreeMap, +} + +pub(super) fn route(method: &Method, uri: &Uri) -> Option<&'static str> { + if method == Method::POST && uri.path() == "/v1/namespaces" { + return Some("POST /namespaces"); + } + let tail = uri.path().strip_prefix("/v1/namespaces/")?; + if method == Method::POST && tail.ends_with("/properties") { + return Some("POST /namespaces/{namespace}/properties"); + } + (method == Method::DELETE).then_some("DELETE /namespaces/{namespace}") +} + +pub(super) fn parse(route: &str, uri: &Uri, bytes: &[u8]) -> Result { + if uri.query().is_some() { + return Err(bad_request()); + } + if route == "POST /namespaces" { + let body: CreateBody = serde_json::from_slice(bytes).map_err(|_| bad_request())?; + let identifier = NamespaceIdentifier::new(body.namespace).map_err(|_| bad_request())?; + let properties = NamespaceProperties::new(body.properties).map_err(|_| bad_request())?; + return Ok(NamespaceMutation::Create(identifier, properties)); + } + let tail = uri + .path() + .strip_prefix("/v1/namespaces/") + .ok_or_else(bad_request)?; + let encoded = if route.starts_with("POST") { + tail.strip_suffix("/properties").ok_or_else(bad_request)? + } else { + tail + }; + if encoded.is_empty() || encoded.contains('/') { + return Err(bad_request()); + } + let identifier = NamespaceIdentifier::from_rest(&decode_path(encoded)?).map_err(|_| bad_request())?; + if route.starts_with("DELETE") { + if !bytes.is_empty() { + return Err(bad_request()); + } + return Ok(NamespaceMutation::Drop(identifier)); + } + let body: UpdateBody = serde_json::from_slice(bytes).map_err(|_| bad_request())?; + let changes = PropertyChanges { + removals: body.removals, + updates: body.updates, + }; + changes + .validate() + .map_err(|error| super::namespace_write::mutation_error(&error.into()))?; + Ok(NamespaceMutation::Update(identifier, changes)) +} diff --git a/app/crowdb-access-server/src/iceberg/namespace_write.rs b/app/crowdb-access-server/src/iceberg/namespace_write.rs new file mode 100644 index 000000000..26fee4a8e --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/namespace_write.rs @@ -0,0 +1,239 @@ +use std::sync::Arc; +use std::time::{SystemTime, UNIX_EPOCH}; + +use crowdb_access_iceberg::catalog::{CatalogContext, CatalogError}; +use crowdb_access_iceberg::error::ValidationError; +use crowdb_access_iceberg::namespace::{ + NamespaceCreateRequest, NamespaceCreator, NamespaceDropRequest, NamespaceDropper, + NamespacePropertyRequest, NamespaceRepository, NamespaceStore, +}; +use crowdb_access_iceberg::operation::{PayloadStore, RetryAdmission, RetryLedger, RetryRecord}; +use crowdb_access_iceberg::wire::{IcebergErrorResponse, Principal, RequestKey}; +use http_body_util::BodyExt; +use hyper::body::Incoming; +use hyper::{Request, Response}; +use sha2::{Digest, Sha256}; + +use super::body::IcebergBody; +use super::http::{bad_request, response, service_unavailable}; +use super::namespace_request::{self, NamespaceMutation}; + +pub(super) struct NamespaceWrites { + creator: NamespaceCreator, + dropper: NamespaceDropper, + repository: NamespaceRepository, + ledger: RetryLedger, + payloads: PayloadStore, +} + +impl NamespaceWrites { + pub(super) fn new(store: Arc) -> Self { + Self { + creator: NamespaceCreator::new(store.clone()), + dropper: NamespaceDropper::new(store.clone()), + repository: NamespaceRepository::new(store.clone()), + ledger: RetryLedger::new(store.clone()), + payloads: PayloadStore::new(store), + } + } + + pub(super) async fn execute( + &self, + context: CatalogContext, + principal: Principal, + request: Request, + ) -> Result, IcebergErrorResponse> { + let route = namespace_request::route(request.method(), request.uri()).ok_or_else(|| { + IcebergErrorResponse::new( + 406, + "UnsupportedOperationException", + "This endpoint is not implemented", + ) + })?; + if !principal.namespace_write { + return Err(IcebergErrorResponse::new( + 403, + "ForbiddenException", + "Namespace write privilege is required", + )); + } + let now = now_ms()?; + if request.headers().get_all("idempotency-key").iter().count() > 1 { + return Err(bad_request()); + } + let header = request + .headers() + .get("idempotency-key") + .map(|value| value.to_str()) + .transpose() + .map_err(|_| bad_request())?; + let request_key = RequestKey::parse(header, now).map_err(|_| bad_request())?; + let uri = request.uri().clone(); + let bytes = read_body(request.into_body()).await?; + let mut digest = Sha256::new(); + for value in [route.as_bytes(), uri.to_string().as_bytes(), bytes.as_slice()] { + digest.update((value.len() as u64).to_be_bytes()); + digest.update(value); + } + let retry = RetryRecord { + identity: request_key.identity(), + principal: principal.name.into(), + route: route.into(), + digest: digest.finalize().into(), + context, + retained_until_ms: 0, + status: 0, + body: Vec::new(), + }; + let retry = match self.admit(retry, request_key, now).await? { + RetryAdmission::Replay(record) => { + super::metrics::record_retry(3); + return Ok(response(record.status, record.body)); + } + RetryAdmission::New(record) => { + super::metrics::record_retry(1); + record + } + RetryAdmission::Resume(record) => { + super::metrics::record_retry(2); + record + } + }; + let result = match namespace_request::parse(route, &uri, &bytes) { + Ok(mutation) => self.mutate(&retry, mutation).await, + Err(error) => Err(error), + }; + let (status, body) = match result { + Ok(result) => result, + Err(error) if error.error.code < 500 => ( + error.error.code, + serde_json::to_vec(&error).map_err(|_| service_unavailable())?, + ), + Err(error) => return Err(error), + }; + self.ledger + .finish(retry, status, body.clone(), now_ms()?) + .await + .map_err(|error| mutation_error(&error))?; + Ok(response(status, body)) + } + + async fn admit( + &self, + mut retry: RetryRecord, + request_key: RequestKey, + now: u64, + ) -> Result { + for _ in 0..8 { + match self.ledger.begin(retry.clone(), now).await { + Err(CatalogError::Busy) if matches!(request_key, RequestKey::Internal(_)) => { + retry.identity = RequestKey::parse(None, now) + .map_err(|_| bad_request())? + .identity(); + } + result => return result.map_err(|error| mutation_error(&error)), + } + } + Err(service_unavailable()) + } + + async fn mutate( + &self, + retry: &RetryRecord, + mutation: NamespaceMutation, + ) -> Result<(u16, Vec), IcebergErrorResponse> { + let outcome = match mutation { + NamespaceMutation::Create(identifier, properties) => Some( + self.creator + .create(&NamespaceCreateRequest { + context: retry.context, + identity: retry.identity, + principal: retry.principal.clone(), + identifier, + properties, + }) + .await + .map_err(|error| mutation_error(&error))?, + ), + NamespaceMutation::Update(identifier, changes) => self + .repository + .update_properties(&NamespacePropertyRequest { + context: retry.context, + identity: retry.identity, + principal: retry.principal.clone(), + identifier, + changes, + }) + .await + .map_err(|error| mutation_error(&error))?, + NamespaceMutation::Drop(identifier) => self + .dropper + .drop_namespace(&NamespaceDropRequest { + context: retry.context, + identity: retry.identity, + principal: retry.principal.clone(), + identifier, + }) + .await + .map_err(|error| mutation_error(&error))?, + } + .ok_or_else(|| { + IcebergErrorResponse::new(404, "NoSuchNamespaceException", "Namespace does not exist") + })?; + let bytes = self + .payloads + .get(&outcome.body) + .await + .map_err(|_| service_unavailable())?; + Ok((outcome.status, bytes)) + } +} + +pub(super) async fn read_body(mut body: Incoming) -> Result, IcebergErrorResponse> { + let mut bytes = Vec::new(); + while let Some(frame) = body.frame().await { + let frame = frame.map_err(|_| bad_request())?; + if let Ok(data) = frame.into_data() { + super::metrics::record_request_bytes(data.len()); + if bytes.len() + data.len() > 2 * 1024 * 1024 { + return Err(bad_request()); + } + bytes.extend_from_slice(&data); + } + } + Ok(bytes) +} + +pub(super) fn now_ms() -> Result { + let elapsed = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_err(|_| service_unavailable())?; + u64::try_from(elapsed.as_millis()).map_err(|_| service_unavailable()) +} + +pub(super) fn mutation_error(error: &CatalogError) -> IcebergErrorResponse { + match error { + CatalogError::Invalid(ValidationError::PropertyOverlap) => IcebergErrorResponse::new( + 422, + "UnprocessableEntityException", + "Property removals and updates overlap", + ), + CatalogError::Invalid( + ValidationError::Text + | ValidationError::KeyTooLarge + | ValidationError::RecordTooLarge + | ValidationError::GenerationExhausted + | ValidationError::Deadline + | ValidationError::Identity, + ) => bad_request(), + CatalogError::Conflict => IcebergErrorResponse::new( + 409, + "CommitFailedException", + "Request identity or catalog context conflicts", + ), + _ => { + tracing::error!(%error, "namespace mutation remains recoverable; retry with the same request key"); + service_unavailable() + } + } +} diff --git a/app/crowdb-access-server/src/iceberg/recovery.rs b/app/crowdb-access-server/src/iceberg/recovery.rs new file mode 100644 index 000000000..5b8025608 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/recovery.rs @@ -0,0 +1,59 @@ +use std::sync::Arc; +use std::time::Duration; + +use crowdb_access_iceberg::catalog::{CatalogRepository, RootState}; +use crowdb_access_iceberg::namespace::NamespaceRecovery; + +pub(super) async fn run(repository: Arc, recovery: NamespaceRecovery) { + let mut interval = tokio::time::interval(Duration::from_secs(1)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + let mut context = None; + let mut continuations = [None, None]; + let mut phase = 0; + loop { + interval.tick().await; + let result = tokio::time::timeout(Duration::from_secs(1), async { + let (root, _) = repository.status().await?; + if root.state != RootState::Ready { + continuations = [None, None]; + return Ok(None); + } + if context != Some(root.context) { + context = Some(root.context); + continuations = [None, None]; + } + if phase == 0 { + recovery + .recover_page(root.context, continuations[phase].clone()) + .await + .map(Some) + } else { + recovery + .repair_page(root.context, continuations[phase].clone()) + .await + .map(Some) + } + }) + .await; + match result { + Ok(Ok(Some(page))) => { + continuations[phase] = page.continuation; + for (operation, error) in page.failures { + tracing::error!(%operation, %error, "namespace recovery failed; retrying on a later sweep"); + } + tracing::debug!( + completed = page.completed, + deferred = page.deferred, + "namespace recovery page processed" + ); + } + Ok(Ok(None)) => {} + Ok(Err(error)) => { + continuations[phase] = None; + tracing::error!(%error, "namespace recovery scan failed; restarting sweep"); + } + Err(_) => tracing::warn!("namespace recovery time budget exhausted; retrying page"), + } + phase = 1 - phase; + } +} diff --git a/app/crowdb-access-server/src/iceberg/routes.rs b/app/crowdb-access-server/src/iceberg/routes.rs new file mode 100644 index 000000000..5fd5061bb --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/routes.rs @@ -0,0 +1,179 @@ +use crowdb_access_iceberg::catalog::{Capabilities, FormatAction}; +use hyper::Method; + +pub(super) struct InstalledRoutes(pub u8); + +impl InstalledRoutes { + pub(super) const NAMESPACES: u8 = 1; + pub(super) const TABLES: u8 = 2; + pub(super) const WRITES: u8 = 4; + pub(super) const CREDENTIALS: u8 = 8; + + fn contains(&self, flags: u8) -> bool { + self.0 & flags == flags + } +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(super) enum Route { + Config, + AdminMetrics, + NamespaceList, + NamespaceCreate, + NamespaceLoad, + NamespaceExists, + NamespaceProperties, + NamespaceDrop, + TableList, + TableCreate, + TableLoad, + TableExists, + TableUpdate, + TableDrop, + TableRename, + TableCredentials, +} + +impl Route { + const ADVERTISED: [Self; 14] = [ + Self::NamespaceList, + Self::NamespaceLoad, + Self::NamespaceExists, + Self::NamespaceCreate, + Self::NamespaceProperties, + Self::NamespaceDrop, + Self::TableList, + Self::TableLoad, + Self::TableExists, + Self::TableCreate, + Self::TableUpdate, + Self::TableDrop, + Self::TableRename, + Self::TableCredentials, + ]; + + pub(super) fn classify(method: &Method, path: &str) -> Option { + if path == "/_crowdb/metrics" { + return (method == Method::GET).then_some(Self::AdminMetrics); + } + if path == "/v1/config" { + return (method == Method::GET).then_some(Self::Config); + } + if path == "/v1/tables/rename" { + return (method == Method::POST).then_some(Self::TableRename); + } + if path == "/v1/namespaces" { + return match *method { + Method::GET => Some(Self::NamespaceList), + Method::POST => Some(Self::NamespaceCreate), + _ => None, + }; + } + let mut parts = path.strip_prefix("/v1/namespaces/")?.split('/'); + if parts.next()?.is_empty() { + return None; + } + match ( + parts.next(), + parts.next(), + parts.next(), + parts.next(), + parts.next(), + ) { + (None, None, None, None, None) => match *method { + Method::GET => Some(Self::NamespaceLoad), + Method::HEAD => Some(Self::NamespaceExists), + Method::DELETE => Some(Self::NamespaceDrop), + _ => None, + }, + (Some("properties"), None, None, None, None) if method == Method::POST => { + Some(Self::NamespaceProperties) + } + (Some("tables"), None, None, None, None) => match *method { + Method::GET => Some(Self::TableList), + Method::POST => Some(Self::TableCreate), + _ => None, + }, + (Some("tables"), Some(table), None, None, None) if !table.is_empty() => match *method { + Method::GET => Some(Self::TableLoad), + Method::HEAD => Some(Self::TableExists), + Method::POST => Some(Self::TableUpdate), + Method::DELETE => Some(Self::TableDrop), + _ => None, + }, + (Some("tables"), Some(table), Some("credentials"), None, None) + if !table.is_empty() && method == Method::GET => + { + Some(Self::TableCredentials) + } + _ => None, + } + } + + pub(super) fn enabled(self, installed: &InstalledRoutes) -> bool { + match self { + Self::Config | Self::AdminMetrics => true, + Self::NamespaceList + | Self::NamespaceCreate + | Self::NamespaceLoad + | Self::NamespaceExists + | Self::NamespaceProperties + | Self::NamespaceDrop => installed.contains(InstalledRoutes::NAMESPACES), + Self::TableList | Self::TableLoad | Self::TableExists => { + installed.contains(InstalledRoutes::NAMESPACES | InstalledRoutes::TABLES) + } + Self::TableCreate | Self::TableUpdate | Self::TableDrop | Self::TableRename => { + installed.contains(InstalledRoutes::NAMESPACES | InstalledRoutes::WRITES) + } + Self::TableCredentials => { + installed.contains(InstalledRoutes::NAMESPACES | InstalledRoutes::CREDENTIALS) + } + } + } + + pub(super) fn supported(self, capabilities: Capabilities) -> bool { + let any = |action| (1..=3).any(|version| capabilities.supports(version, action)); + match self { + Self::TableList | Self::TableLoad | Self::TableExists | Self::TableCredentials => { + any(FormatAction::Read) + } + Self::TableCreate => any(FormatAction::Create), + Self::TableUpdate => { + any(FormatAction::Write) || capabilities.upgrade_v1_v2 || capabilities.upgrade_v2_v3 + } + Self::TableDrop | Self::TableRename => any(FormatAction::Write), + _ => true, + } + } + + pub(super) fn endpoints(installed: &InstalledRoutes, capabilities: Capabilities) -> Vec { + Self::ADVERTISED + .iter() + .filter(|route| route.enabled(installed) && route.supported(capabilities)) + .filter_map(|route| route.template()) + .map(str::to_owned) + .collect() + } + + fn template(self) -> Option<&'static str> { + match self { + Self::Config | Self::AdminMetrics => None, + Self::NamespaceList => Some("GET /v1/{prefix}/namespaces"), + Self::NamespaceCreate => Some("POST /v1/{prefix}/namespaces"), + Self::NamespaceLoad => Some("GET /v1/{prefix}/namespaces/{namespace}"), + Self::NamespaceExists => Some("HEAD /v1/{prefix}/namespaces/{namespace}"), + Self::NamespaceProperties => Some("POST /v1/{prefix}/namespaces/{namespace}/properties"), + Self::NamespaceDrop => Some("DELETE /v1/{prefix}/namespaces/{namespace}"), + Self::TableList => Some("GET /v1/{prefix}/namespaces/{namespace}/tables"), + Self::TableCreate => Some("POST /v1/{prefix}/namespaces/{namespace}/tables"), + Self::TableLoad => Some("GET /v1/{prefix}/namespaces/{namespace}/tables/{table}"), + Self::TableExists => Some("HEAD /v1/{prefix}/namespaces/{namespace}/tables/{table}"), + Self::TableUpdate => Some("POST /v1/{prefix}/namespaces/{namespace}/tables/{table}"), + Self::TableDrop => Some("DELETE /v1/{prefix}/namespaces/{namespace}/tables/{table}"), + Self::TableRename => Some("POST /v1/{prefix}/tables/rename"), + Self::TableCredentials => { + Some("GET /v1/{prefix}/namespaces/{namespace}/tables/{table}/credentials") + } + } + } +} diff --git a/app/crowdb-access-server/src/iceberg/runtime.rs b/app/crowdb-access-server/src/iceberg/runtime.rs new file mode 100644 index 000000000..cc2c78e4d --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/runtime.rs @@ -0,0 +1,300 @@ +use std::sync::Arc; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use crowdb_access_iceberg::catalog::{ + Capabilities, CatalogError, CatalogLifecycle, CatalogRepository, ClearBounds, ManagementPrivilege, + RootState, RoutedCatalogStore, +}; +use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; +use crowdb_access_iceberg::wire::BearerAuthenticator; +use crowdb_chunk_client::{ChunkIoClient, ChunkIoClientConfig, SmallWritePolicy}; +use crowdb_chunk_kv_client::{ + ChunkKvClient, ChunkKvRpcTransport, ClientConfig, Group0ChunkKvRangeCatalogSource, +}; +use crowdb_kv_client::{ClientConfig as KvConfig, CrowdbKvClient}; +use tokio::net::TcpListener; + +use super::{serve, IcebergHttpService}; + +type BoxError = Box; + +pub struct IcebergRuntimeConfig { + pub listen: String, + pub management_seeds: Vec, + pub authentication: BearerAuthenticator, +} + +impl IcebergRuntimeConfig { + /// # Errors + /// Rejects missing/invalid credentials, seeds or listener configuration. + pub fn from_env() -> Result { + let seeds = std::env::var("CROWDB_MANAGEMENT_SEEDS")?; + if seeds.len() > 8192 { + return Err("management seed configuration is oversized".into()); + } + let management_seeds: Vec<_> = seeds + .split(',') + .map(str::trim) + .filter(|value| !value.is_empty()) + .map(str::to_owned) + .collect(); + if management_seeds.is_empty() || management_seeds.len() > 16 { + return Err("one to sixteen management seeds are required".into()); + } + let authentication = BearerAuthenticator::new( + &std::env::var("CROWDB_ICEBERG_READ_TOKEN")?, + &std::env::var("CROWDB_ICEBERG_WRITE_TOKEN") + .map_err(|_| "CROWDB_ICEBERG_WRITE_TOKEN must be set")?, + &std::env::var("CROWDB_ICEBERG_MANAGE_TOKEN")?, + &std::env::var("CROWDB_ICEBERG_CLEAR_TOKEN")?, + ) + .map_err(|error| format!("invalid Iceberg bearer credential configuration: {error}"))?; + let listen = std::env::var("CROWDB_ICEBERG_LISTEN").unwrap_or_else(|_| "127.0.0.1:8181".into()); + let _: std::net::SocketAddr = listen.parse()?; + Ok(Self { + listen, + management_seeds, + authentication, + }) + } +} + +/// # Errors +/// Returns configuration, authentication, storage, management or listener failures. +pub async fn run() -> Result<(), BoxError> { + let config = IcebergRuntimeConfig::from_env()?; + let arguments: Vec<_> = std::env::args().skip(1).collect(); + if arguments.len() > 7 { + return Err("too many Iceberg command arguments".into()); + } + let (repository, store, chunks) = connect(config.management_seeds.clone()).await?; + let result = if arguments.is_empty() || arguments == ["serve"] { + Box::pin(start_listener( + &config.listen, + repository, + store, + config.authentication, + chunks.clone(), + config.management_seeds, + )) + .await + } else if arguments.first().is_some_and(|argument| argument == "gc") { + super::gc_control::manage( + &repository, + store.clone(), + &config.authentication, + &arguments[1..], + ) + .await + } else { + manage(&repository, &config.authentication, &arguments).await + }; + let shutdown = chunks.shutdown_small_writes().await; + result?; + shutdown?; + Ok(()) +} + +async fn connect( + seeds: Vec, +) -> Result<(Arc, Arc, ChunkIoClient), BoxError> { + let control = Arc::new(CrowdbKvClient::new(KvConfig::new(seeds.clone()))); + let client_config = ClientConfig::default(); + let source = Arc::new(Group0ChunkKvRangeCatalogSource::from_shared(Arc::clone(&control))); + let transport = Arc::new(ChunkKvRpcTransport::new( + client_config.max_owner_connections, + 1, + 2, + )); + let client = Arc::new(ChunkKvClient::new(client_config, source, transport)?); + client.refresh_catalog().await?; + let chunks = ChunkIoClient::connect_with_kv( + ChunkIoClientConfig { + management_seeds: seeds, + diskio_connections_per_endpoint: 2, + diskio_rpc_workers: 2, + small_write: SmallWritePolicy::default(), + }, + control, + ) + .await?; + let store = Arc::new(RoutedCatalogStore::new(client)); + let repository = Arc::new(CatalogRepository::new( + store.clone(), + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + )?); + Ok((repository, store, chunks)) +} + +async fn start_listener( + address: &str, + repository: Arc, + store: Arc, + authentication: BearerAuthenticator, + chunks: ChunkIoClient, + management_seeds: Vec, +) -> Result<(), BoxError> { + let gc_config = super::gc_runtime::GcRuntimeConfig::from_env()?; + for _ in 0..600 { + match repository.recover(now_ms()?).await { + Ok(()) => break, + Err(CatalogError::Busy) => tokio::time::sleep(Duration::from_millis(100)).await, + Err(error) => return Err(error.into()), + } + } + let (root, authority) = repository.status().await?; + if root.state != RootState::Ready || authority.lifecycle != CatalogLifecycle::Ready { + return Err("Iceberg catalog is not ready for this server".into()); + } + let timeout = Duration::from_millis(authority.admission_bounds.request_ms); + if timeout.is_zero() || timeout > Duration::from_secs(300) { + return Err("catalog request timeout is outside server bounds".into()); + } + let blocks: Arc = Arc::new( + crowdb_access_iceberg::file::NativeFileBlocks::new(chunks.clone(), store.clone()), + ); + let mut service = IcebergHttpService::new(repository.clone(), authentication, timeout) + .with_namespaces(store.clone())? + .with_fileio(store.clone(), blocks.clone(), "us-east-1".into())?; + if authority.admission_bounds.delegated_access_ms >= 900_000 { + let endpoint = + std::env::var("CROWDB_ICEBERG_PUBLIC_URI").unwrap_or_else(|_| format!("http://{address}")); + service = service + .with_tables(store.clone(), blocks.clone())? + .with_table_credentials(store.clone(), endpoint)?; + } else { + tracing::warn!("table routes disabled: persisted catalog delegation bound is below fifteen minutes"); + } + let service = Arc::new(service); + let listener = TcpListener::bind(address).await?; + tracing::info!(%address, "Iceberg listener ready"); + let serving = serve(listener, service, async { + let _ = tokio::signal::ctrl_c().await; + }); + let multipart = Box::pin(super::file_recovery::run( + repository.clone(), + store.clone(), + blocks.clone(), + )); + let tables = super::table_recovery::run(repository.clone(), store.clone(), blocks.clone()); + let (gc_store, gc_chunks) = if gc_config.enabled { + let (_, gc_store, gc_chunks) = connect(management_seeds).await?; + (gc_store, Some(gc_chunks)) + } else { + (store.clone(), None) + }; + let gc_client = gc_chunks.clone().unwrap_or_else(|| chunks.clone()); + let gc = super::gc_runtime::run(repository.clone(), gc_store, gc_client, gc_config); + tokio::select! { + result = serving => result?, + () = super::recovery::run(repository, crowdb_access_iceberg::namespace::NamespaceRecovery::new(store)) => {} + () = multipart => {} + () = tables => {} + () = gc => {} + } + if let Some(gc_chunks) = gc_chunks { + gc_chunks.shutdown_small_writes().await?; + } + tracing::info!("Iceberg listener drained"); + Ok(()) +} + +async fn manage( + repository: &CatalogRepository, + authentication: &BearerAuthenticator, + arguments: &[String], +) -> Result<(), BoxError> { + let token = std::env::var("CROWDB_ICEBERG_TOKEN")?; + let principal = authentication + .authenticate(&format!("Bearer {token}")) + .ok_or("invalid management bearer token")?; + if principal.management == ManagementPrivilege::None { + return Err("management privilege is required".into()); + } + if arguments == ["status"] { + let (root, authority) = repository.status().await?; + println!( + "{}", + serde_json::json!({"catalog_id": authority.catalog.to_string(), "display_name": authority.display_name, + "activation_epoch": root.context.activation_epoch, "state": format!("{:?}", root.state), + "capability_bits": format!("0x{:04x}", authority.capabilities.bits()), + "config_generation": authority.config_generation}) + ); + return Ok(()); + } + if arguments == ["inspect"] { + match repository.status().await { + Ok((root, authority)) => println!( + "{}", + serde_json::json!({"initialized": true, "catalog_id": authority.catalog.to_string(), + "display_name": authority.display_name, "activation_epoch": root.context.activation_epoch, + "state": format!("{:?}", root.state), "capability_bits": format!("0x{:04x}", authority.capabilities.bits()), + "root_operation_id": root.operation.to_string()}) + ), + Err(CatalogError::Uninitialized) => println!("{{\"initialized\":false}}"), + Err(error) => return Err(error.into()), + } + return Ok(()); + } + let action = match arguments.first().map(String::as_str) { + Some("initialize") if arguments.len() == 3 => ManagementAction::Initialize, + Some("rename") if arguments.len() == 4 => ManagementAction::Rename, + Some("clear") if arguments.len() == 5 => ManagementAction::Clear, + Some("activate") if arguments.len() == 5 => ManagementAction::Activate, + _ => return Err("usage: crowdb-iceberg initialize UUIDv7 NAME | rename UUIDv7 NAME EPOCH | clear UUIDv7 NAME EPOCH CONFIRM_CATALOG_ID | activate UUIDv7 NAME EPOCH CAPABILITY_BITS_HEX | status | inspect | serve".into()), + }; + let request = ManagementRequest { + identity: RequestIdentity::parse(&arguments[1], now_ms()?)?, + principal: principal.name.into(), + action, + display_name: arguments[2].clone(), + expected_epoch: arguments + .get(3) + .map(|value| value.parse()) + .transpose()? + .unwrap_or(0), + confirmation: if action == ManagementAction::Clear { + arguments.get(4).map(|value| value.parse()).transpose()? + } else { + None + }, + capabilities: if action == ManagementAction::Activate { + Some(Capabilities::from_bits(u16::from_str_radix( + arguments[4].trim_start_matches("0x"), + 16, + )?)?) + } else { + None + }, + }; + for _ in 0..600 { + match repository + .execute(request.clone(), principal.management, now_ms()?) + .await + { + Ok(authority) => { + println!( + "{}", + serde_json::json!({"catalog_id": authority.catalog.to_string(), "display_name": authority.display_name, + "name_generation": authority.name_generation, "config_generation": authority.config_generation, + "capability_bits": format!("0x{:04x}", authority.capabilities.bits()), + "operation_id": request.identity.operation.to_string()}) + ); + return Ok(()); + } + Err(CatalogError::Busy) => tokio::time::sleep(Duration::from_millis(100)).await, + Err(error) => return Err(error.into()), + } + } + Err("management operation is still pending; retry with the same identity and input".into()) +} + +pub(super) fn now_ms() -> Result { + Ok(u64::try_from( + SystemTime::now().duration_since(UNIX_EPOCH)?.as_millis(), + )?) +} diff --git a/app/crowdb-access-server/src/iceberg/table_credentials.rs b/app/crowdb-access-server/src/iceberg/table_credentials.rs new file mode 100644 index 000000000..9d40b2dd1 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/table_credentials.rs @@ -0,0 +1,260 @@ +use std::{collections::BTreeMap, sync::Arc}; + +use crowdb_access_iceberg::{ + catalog::{CatalogContext, CatalogLifecycle, CatalogRepository, CatalogStore, FormatAction, RootState}, + commit::{TableCreateJournal, TableCreatePhase}, + file::FileGrantIssuer, + key::{OperationId, TableId}, + namespace::{NamespaceIdentifier, NamespaceRepository, NamespaceStore}, + table::TableRepository, + wire::{ + FileDelegationLimits, FileDelegationTarget, IcebergErrorResponse, LoadCredentialsResponse, Principal, + }, +}; +use hyper::{Request, Response}; + +use super::{ + body::IcebergBody, + http::{bad_request, decode_query, response, service_unavailable}, + namespace_write::now_ms, + table_read::{missing_table, unsupported}, +}; + +pub(super) struct TableCredentials { + store: Arc, + namespaces: NamespaceRepository, + tables: TableRepository, + issuer: FileGrantIssuer, +} + +#[derive(Clone)] +pub(super) struct TableFileConfig { + endpoint: String, +} + +impl TableFileConfig { + pub(super) fn append( + &self, + bytes: &mut Vec, + namespace: &NamespaceIdentifier, + name: &str, + table: TableId, + ) -> Result<(), IcebergErrorResponse> { + let config = serde_json::to_vec(&self.properties(namespace, name, table)) + .map_err(|_| service_unavailable())?; + if bytes.last() != Some(&b'}') + || bytes.len() + config.len() + 16 > crowdb_access_iceberg::operation::MAX_PAYLOAD_BYTES + { + return Err(service_unavailable()); + } + bytes.pop(); + bytes.extend_from_slice(b",\"config\":"); + bytes.extend_from_slice(&config); + bytes.push(b'}'); + Ok(()) + } + pub(super) fn new(mut endpoint: String) -> Result { + let uri: hyper::Uri = endpoint + .parse() + .map_err(|_| crowdb_access_iceberg::error::ValidationError::Text)?; + if !matches!(uri.scheme_str(), Some("http" | "https")) + || uri.authority().is_none() + || uri + .authority() + .is_some_and(|authority| authority.as_str().contains('@')) + || uri.path() != "/" + || uri.query().is_some() + || endpoint.len() > 2048 + { + return Err(crowdb_access_iceberg::error::ValidationError::Text); + } + endpoint.truncate(endpoint.trim_end_matches('/').len()); + Ok(Self { endpoint }) + } + + pub(super) fn properties( + &self, + namespace: &NamespaceIdentifier, + name: &str, + table: TableId, + ) -> BTreeMap<&'static str, String> { + let namespace = namespace.components().join("\u{1f}"); + let namespace = percent_encoding::utf8_percent_encode(&namespace, percent_encoding::NON_ALPHANUMERIC); + let name = percent_encoding::utf8_percent_encode(name, percent_encoding::NON_ALPHANUMERIC); + BTreeMap::from([ + ("s3.endpoint", self.endpoint.clone()), + ("s3.path-style-access", "true".into()), + ("client.region", "us-east-1".into()), + ( + "client.refresh-credentials-endpoint", + format!("/v1/namespaces/{namespace}/tables/{name}/credentials?table-id={table}"), + ), + ]) + } +} + +impl TableCredentials { + pub(super) fn new( + store: Arc, + secret: [u8; 32], + ) -> Result { + Ok(Self { + store: store.clone(), + namespaces: NamespaceRepository::new(store.clone()), + tables: TableRepository::new(store), + issuer: FileGrantIssuer::new(secret, 900_000)?, + }) + } + + pub(super) async fn load( + &self, + repository: &CatalogRepository, + context: CatalogContext, + principal: Principal, + request: &Request, + ) -> Result, IcebergErrorResponse> { + if request.method() != hyper::Method::GET { + return Err(super::table_read::unsupported()); + } + let path = request + .uri() + .path() + .strip_suffix("/credentials") + .ok_or_else(bad_request)?; + let target = super::table_write::request::parse(&path.parse().map_err(|_| bad_request())?)?; + let name = target.name.ok_or_else(bad_request)?; + let selector = selector(request.uri().query())?; + let now = now_ms()?; + let parent = self + .namespaces + .load(context, &target.namespace) + .await + .map_err(|_| service_unavailable())? + .ok_or_else(missing_table)?; + let selected = self + .tables + .select(context, parent.namespace, &name) + .await + .map_err(|_| service_unavailable())?; + let mut ttl_ms = 900_000; + let (table, version, staged) = if let Some(selected) = + selected.filter(|selected| selector.map_or(true, |table| table == selected.head.table)) + { + self.tables + .ensure_current(context, &selected) + .await + .map_err(|_| service_unavailable())?; + (selected.head.table, selected.head.format_version, false) + } else { + let table = selector.ok_or_else(missing_table)?; + let identity = OperationId::from_bytes(table.as_bytes()).map_err(|_| bad_request())?; + let journal = TableCreateJournal::new(self.store.clone()); + let operation = journal + .load(context, identity) + .await + .map_err(|_| service_unavailable())? + .ok_or_else(missing_table)?; + let stage = operation.stage.as_ref().ok_or_else(missing_table)?; + if !principal.namespace_write + || operation.principal != principal.name + || operation.namespace != target.namespace + || operation.candidate.namespace != parent.namespace + || operation.candidate.name != name + || operation.candidate.table != table + || operation.phase != TableCreatePhase::Staged + { + return Err(missing_table()); + } + let expires = u64::try_from(stage.expires_ms).map_err(|_| service_unavailable())?; + ttl_ms = ttl_ms.min( + expires + .checked_sub(now) + .filter(|remaining| *remaining > 0) + .ok_or_else(missing_table)?, + ); + if journal + .load(context, identity) + .await + .map_err(|_| service_unavailable())? + .as_ref() + != Some(&operation) + { + return Err(service_unavailable()); + } + (table, operation.candidate.format_version, true) + }; + self.issue_response( + repository, + context, + principal, + FileDelegationTarget { + table, + format_version: version, + staged, + }, + ttl_ms, + now, + ) + .await + } + + async fn issue_response( + &self, + repository: &CatalogRepository, + context: CatalogContext, + principal: Principal, + target: FileDelegationTarget, + ttl_ms: u64, + now: u64, + ) -> Result, IcebergErrorResponse> { + let (root, authority) = repository.status().await.map_err(|_| service_unavailable())?; + if root.context != context + || root.state != RootState::Ready + || authority.lifecycle != CatalogLifecycle::Ready + { + return Err(service_unavailable()); + } + if !authority + .capabilities + .supports(target.format_version, FormatAction::Read) + { + return Err(unsupported()); + } + let credentials = FileDelegationLimits { + ttl_ms, + max_request_bytes: 1024 * 1024 * 1024, + max_file_bytes: 1024 * 1024 * 1024 * 1024, + } + .issue(&self.issuer, principal, context, &authority, target, now) + .map_err(|_| service_unavailable())?; + let pins = crowdb_access_iceberg::gc::ReaderPins::new(self.store.clone()); + let expires_ms = pins + .request_expiry(context, credentials.grant().expires_ms) + .await + .map_err(|_| service_unavailable())?; + pins.protect_files(context, target.table, principal.name, expires_ms, now) + .await + .map_err(|_| service_unavailable())?; + Ok(response( + 200, + serde_json::to_vec(&LoadCredentialsResponse::from(credentials)) + .map_err(|_| service_unavailable())?, + )) + } +} + +fn selector(query: Option<&str>) -> Result, IcebergErrorResponse> { + let mut selector = None; + for pair in query + .unwrap_or_default() + .split('&') + .filter(|pair| !pair.is_empty()) + { + let (name, value) = pair.split_once('=').ok_or_else(bad_request)?; + if decode_query(name)? != "table-id" || selector.is_some() { + return Err(bad_request()); + } + selector = Some(decode_query(value)?.parse().map_err(|_| bad_request())?); + } + Ok(selector) +} diff --git a/app/crowdb-access-server/src/iceberg/table_limits.rs b/app/crowdb-access-server/src/iceberg/table_limits.rs new file mode 100644 index 000000000..b16a12722 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/table_limits.rs @@ -0,0 +1,126 @@ +use crowdb_access_iceberg::{ + commit::{ + CandidateAuxiliaryLimits, CandidateSnapshotLimits, CommitPreparationLimits, CommitProofLimits, + CommitRequestLimits, EvaluationLimits, PriorManifestLimits, RequirementLimits, + }, + file::{AvroDatumLimits, AvroLimits, DeletionVectorLimits, ParquetMetadataLimits, ParquetPageLimits}, + manifest::{ + PositionDeleteLimits, SnapshotDvLimits, SnapshotFileLimits, SnapshotIdentityLimits, + SnapshotManifestLimits, + }, + table::TableMetadataLimits, +}; + +pub(super) const RESPONSE_RESERVE: usize = 64 * 1024; + +pub(super) fn metadata() -> TableMetadataLimits { + TableMetadataLimits { + bytes: 2 * 1024 * 1024 - RESPONSE_RESERVE, + values: 200_000, + depth: 64, + string_bytes: 1024 * 1024, + collection_entries: 10_000, + } +} + +pub(super) fn commits() -> CommitProofLimits { + let manifests = SnapshotManifestLimits { + framing: AvroLimits { + header_bytes: 256 * 1024, + metadata_entries: 64, + block_bytes: 4 * 1024 * 1024, + records_per_block: 100_000, + }, + datum: AvroDatumLimits { + depth: 64, + values: 1_000_000, + value_bytes: 1024 * 1024, + }, + decoded_bytes: 8 * 1024 * 1024, + manifests: 1000, + entries: 100_000, + manifest_bytes: 64 * 1024 * 1024, + identity: SnapshotIdentityLimits { + keys: 100_000, + key_bytes: 16 * 1024 * 1024, + }, + }; + let parquet = ParquetMetadataLimits { + footer_bytes: 1024 * 1024, + values: 100_000, + depth: 32, + schema_elements: 4096, + row_groups: 10_000, + }; + CommitProofLimits { + preparation: CommitPreparationLimits { + request: CommitRequestLimits { + json: metadata(), + requirements: 1000, + updates: 1000, + }, + evaluation: EvaluationLimits { + metadata: metadata(), + requirements: RequirementLimits { + count: 1000, + text_bytes: 4096, + }, + updates: 1000, + work_bytes: 16 * 1024 * 1024, + }, + }, + prior: PriorManifestLimits { + snapshots: 1000, + references: 100_000, + index_bytes: 16 * 1024 * 1024, + manifests, + }, + snapshots: CandidateSnapshotLimits { + snapshots: 1000, + entries: 100_000, + manifest_bytes: 64 * 1024 * 1024, + ranges: 100_000, + files: SnapshotFileLimits { + manifests, + data_files: 100_000, + index_bytes: 16 * 1024 * 1024, + position_deletes: PositionDeleteLimits { + metadata: parquet, + page: ParquetPageLimits { + bytes: 8 * 1024 * 1024, + values: 1_000_000, + pages: 100_000, + }, + rows: 1_000_000, + }, + vectors: SnapshotDvLimits { + vectors: 10_000, + blob_bytes: 16 * 1024 * 1024, + vector: DeletionVectorLimits { + blob_bytes: 16 * 1024 * 1024, + bitmaps: 100_000, + }, + }, + delete_rows: 1_000_000, + }, + }, + auxiliary: CandidateAuxiliaryLimits { + manifests, + files: 1000, + bytes: 64 * 1024 * 1024, + work: 1_000_000, + puffin_encoded_bytes: 1024 * 1024, + puffin_decoded_bytes: 1024 * 1024, + parquet, + partition_rows: crowdb_access_iceberg::manifest::PartitionStatisticsRowLimits { + page: ParquetPageLimits { + bytes: 1024 * 1024, + values: 100_000, + pages: 100_000, + }, + rows: 1_000_000, + buffered_bytes: 64 * 1024 * 1024, + }, + }, + } +} diff --git a/app/crowdb-access-server/src/iceberg/table_read.rs b/app/crowdb-access-server/src/iceberg/table_read.rs new file mode 100644 index 000000000..d8d80e422 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/table_read.rs @@ -0,0 +1,310 @@ +use std::collections::BTreeMap; +use std::sync::{atomic::AtomicUsize, Arc}; + +use crowdb_access_iceberg::{ + catalog::{Capabilities, CatalogContext, CatalogError, FormatAction}, + error::ValidationError, + key::NameSuffix, + namespace::NamespaceIdentifier, + table::{SnapshotLoadingMode, TableListLimits, TableLister, TableLoad, TableLoader}, + wire::IcebergErrorResponse, +}; +use crowdb_access_iceberg::{file::FileBlockStore, namespace::NamespaceStore, table::TableMetadataLimits}; +use hyper::{header, Method, Request, Response}; +use sha2::{Digest, Sha256}; + +use super::{ + body::{IcebergBody, SpoolPermit}, + http::{bad_request, decode_query, response, service_unavailable}, + namespace_read::decode_path, +}; + +const MAX_RESPONSE_BYTES: usize = 4 * 1024 * 1024; + +pub(super) struct TableHttp { + loader: TableLoader, + lister: TableLister, + spools: Arc, + pub(super) file_config: Option, +} + +impl TableHttp { + pub(super) fn new( + store: Arc, + blocks: Arc, + secret: &[u8; 32], + ) -> Result { + Ok(Self { + loader: TableLoader::new( + store.clone(), + blocks, + TableMetadataLimits { + bytes: 2 * 1024 * 1024, + values: 200_000, + depth: 64, + string_bytes: 1024 * 1024, + collection_entries: 10_000, + }, + ) + .with_catalog_reader_pins(), + lister: TableLister::new(store, secret)?, + spools: Arc::new(AtomicUsize::new(0)), + file_config: None, + }) + } + + pub(super) async fn read( + &self, + context: CatalogContext, + capabilities: Capabilities, + request: &Request, + ) -> Result, IcebergErrorResponse> { + if request.method() != Method::GET && request.method() != Method::HEAD { + return Err(unsupported()); + } + let suffix = request + .uri() + .path() + .strip_prefix("/v1/namespaces/") + .ok_or_else(bad_request)?; + let mut parts = suffix.split('/'); + let namespace = NamespaceIdentifier::from_rest(&decode_path(parts.next().ok_or_else(bad_request)?)?) + .map_err(|_| bad_request())?; + if parts.next() != Some("tables") { + return Err(bad_request()); + } + let name = parts.next().map(decode_path).transpose()?; + if parts.next().is_some() { + return Err(unsupported()); + } + let mut parameters = parameters(request.uri().query())?; + let permit = SpoolPermit::acquire(&self.spools).ok_or_else(service_unavailable)?; + let mut result = if let Some(name) = name { + NameSuffix { + parent: None, + name: &name, + } + .encode() + .map_err(|_| bad_request())?; + if request.method() == Method::HEAD { + if !parameters.is_empty() { + return Err(bad_request()); + } + let head = self + .loader + .head(context, &namespace, &name) + .await + .map_err(|_| service_unavailable())? + .ok_or_else(missing_table)?; + if !capabilities.supports(head.format_version, FormatAction::Read) { + return Err(unsupported()); + } + response(204, Vec::new()) + } else { + self.load( + context, + capabilities, + &namespace, + &name, + &mut parameters, + request.headers(), + ) + .await? + } + } else { + if request.method() != Method::GET { + return Err(unsupported()); + } + self.list(context, &namespace, &mut parameters).await? + }; + let body = std::mem::replace(result.body_mut(), IcebergBody::new(Vec::new())); + *result.body_mut() = body.with_spool_permit(permit); + Ok(result) + } + + async fn load( + &self, + context: CatalogContext, + capabilities: Capabilities, + namespace: &NamespaceIdentifier, + name: &str, + parameters: &mut BTreeMap, + headers: &hyper::HeaderMap, + ) -> Result, IcebergErrorResponse> { + let mode = match parameters.remove("snapshots").as_deref() { + None | Some("all") => SnapshotLoadingMode::All, + Some("refs") => SnapshotLoadingMode::Refs, + _ => return Err(bad_request()), + }; + if !parameters.is_empty() { + return Err(bad_request()); + } + let condition = condition(headers)?; + let loaded = self + .loader + .load_with_capabilities(context, namespace, name, mode, capabilities) + .await + .map_err(|error| match error { + crowdb_access_iceberg::table::TableLoadError::UnsupportedVersion => unsupported(), + _ => service_unavailable(), + })?; + let (mut result, etag) = match loaded { + TableLoad::Missing => return Err(missing_table()), + TableLoad::NotModified { etag } => (response(304, Vec::new()), etag), + TableLoad::Loaded { head, etag, metadata } => { + if !capabilities.supports(head.format_version, FormatAction::Read) { + return Err(unsupported()); + } + super::metrics::record_selected_version(head.format_version); + let location = serde_json::to_vec(&head.metadata_location.to_string()) + .map_err(|_| service_unavailable())?; + let mut bytes = b"{\"metadata-location\":".to_vec(); + append(&mut bytes, &location)?; + append(&mut bytes, b",\"metadata\":")?; + append(&mut bytes, &metadata)?; + append(&mut bytes, b"}")?; + let etag = if let Some(config) = &self.file_config { + config.append(&mut bytes, namespace, name, head.table)?; + let mut digest = Sha256::new(); + digest.update(etag.as_bytes()); + digest.update(&bytes); + format!("\"{:x}\"", digest.finalize()) + } else { + etag + }; + let unchanged = condition.as_deref().is_some_and(|header| { + header.split(',').any(|tag| { + let tag = tag.trim(); + tag == "*" || tag.strip_prefix("W/").unwrap_or(tag) == etag + }) + }); + ( + if unchanged { + response(304, Vec::new()) + } else { + response(200, bytes) + }, + etag, + ) + } + }; + result + .headers_mut() + .insert(header::ETAG, etag.parse().map_err(|_| service_unavailable())?); + Ok(result) + } + + async fn list( + &self, + context: CatalogContext, + namespace: &NamespaceIdentifier, + parameters: &mut BTreeMap, + ) -> Result, IcebergErrorResponse> { + let page_size = parameters + .remove("pageSize") + .map(|value| value.parse::()) + .transpose() + .map_err(|_| bad_request())? + .unwrap_or(100); + let token = parameters.remove("pageToken"); + if !parameters.is_empty() || !(1..=100).contains(&page_size) { + return Err(bad_request()); + } + let page = self + .lister + .list( + context, + namespace, + TableListLimits { + page_size, + scanned: 4096, + names_bytes: 256 * 1024, + }, + token.as_deref(), + ) + .await + .map_err(|error| list_error(&error))? + .ok_or_else(|| { + IcebergErrorResponse::new(404, "NoSuchNamespaceException", "Namespace does not exist") + })?; + let mut bytes = b"{\"identifiers\":[".to_vec(); + for (index, name) in page.names.iter().enumerate() { + if index > 0 { + append(&mut bytes, b",")?; + } + let identifier = + serde_json::to_vec(&serde_json::json!({"namespace":namespace.components(),"name":name})) + .map_err(|_| service_unavailable())?; + append(&mut bytes, &identifier)?; + } + append(&mut bytes, b"],\"next-page-token\":")?; + append( + &mut bytes, + &serde_json::to_vec(&page.next_page_token).map_err(|_| service_unavailable())?, + )?; + append(&mut bytes, b"}")?; + Ok(response(200, bytes)) + } +} + +fn parameters(query: Option<&str>) -> Result, IcebergErrorResponse> { + let mut parameters = BTreeMap::new(); + for pair in query + .unwrap_or_default() + .split('&') + .filter(|value| !value.is_empty()) + { + let (name, value) = pair.split_once('=').unwrap_or((pair, "")); + if parameters + .insert(decode_query(name)?, decode_query(value)?) + .is_some() + { + return Err(bad_request()); + } + } + Ok(parameters) +} + +fn condition(headers: &hyper::HeaderMap) -> Result, IcebergErrorResponse> { + let mut result = String::new(); + for value in headers.get_all(header::IF_NONE_MATCH) { + let value = value.to_str().map_err(|_| bad_request())?; + if result.len() + value.len() + 1 > 8192 { + return Err(bad_request()); + } + if !result.is_empty() { + result.push(','); + } + result.push_str(value); + } + Ok((!result.is_empty()).then_some(result)) +} + +fn append(bytes: &mut Vec, value: &[u8]) -> Result<(), IcebergErrorResponse> { + if value.len() > MAX_RESPONSE_BYTES.saturating_sub(bytes.len()) { + return Err(service_unavailable()); + } + bytes.extend_from_slice(value); + Ok(()) +} + +pub(super) fn missing_table() -> IcebergErrorResponse { + IcebergErrorResponse::new(404, "NoSuchTableException", "Table does not exist") +} + +pub(super) fn unsupported() -> IcebergErrorResponse { + IcebergErrorResponse::new( + 406, + "UnsupportedOperationException", + "This table endpoint is not enabled", + ) +} + +fn list_error(error: &CatalogError) -> IcebergErrorResponse { + match error { + CatalogError::Invalid( + ValidationError::Key | ValidationError::KeyTooLarge | ValidationError::IdentityMismatch, + ) => bad_request(), + _ => service_unavailable(), + } +} diff --git a/app/crowdb-access-server/src/iceberg/table_recovery.rs b/app/crowdb-access-server/src/iceberg/table_recovery.rs new file mode 100644 index 000000000..7d99afccf --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/table_recovery.rs @@ -0,0 +1,63 @@ +use crowdb_access_iceberg::{ + catalog::{CatalogRepository, RootState, RoutedCatalogStore}, + commit::{TableRecovery, TableRecoveryKind}, + file::FileBlockStore, +}; +use std::{sync::Arc, time::Duration}; + +pub(super) async fn run( + catalog: Arc, + store: Arc, + blocks: Arc, +) { + let recovery = TableRecovery::new(store, blocks, super::table_limits::commits()); + let mut interval = tokio::time::interval(Duration::from_secs(1)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + let mut context = None; + let mut continuations = [None, None, None]; + let mut index = 0; + loop { + interval.tick().await; + let Ok(Ok((root, authority))) = tokio::time::timeout(Duration::from_secs(1), catalog.status()).await + else { + continuations = [None, None, None]; + continue; + }; + if context != Some(root.context) || root.state != RootState::Ready { + context = Some(root.context); + continuations = [None, None, None]; + } + if root.state != RootState::Ready { + continue; + } + let Ok(now) = super::namespace_write::now_ms() + .and_then(|now| i64::try_from(now).map_err(|_| super::http::service_unavailable())) + else { + continue; + }; + let kind = [ + TableRecoveryKind::Create, + TableRecoveryKind::Update, + TableRecoveryKind::Lifecycle, + ][index]; + let result = tokio::time::timeout( + Duration::from_millis(authority.admission_bounds.request_ms), + recovery.recover_page(root.context, kind, continuations[index].clone(), now), + ) + .await; + match result { + Ok(Ok(page)) => { + continuations[index] = page.continuation; + for (operation, error) in page.failures { + tracing::debug!(%operation, %error, "table recovery deferred; durable intent retained"); + } + } + Ok(Err(error)) => { + continuations[index] = None; + tracing::error!(%error, "table recovery scan failed; restarting sweep"); + } + Err(_) => tracing::warn!("table recovery page deadline exhausted; retaining cursor"), + } + index = (index + 1) % continuations.len(); + } +} diff --git a/app/crowdb-access-server/src/iceberg/table_write.rs b/app/crowdb-access-server/src/iceberg/table_write.rs new file mode 100644 index 000000000..254300531 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/table_write.rs @@ -0,0 +1,235 @@ +use std::sync::{atomic::AtomicUsize, Arc}; + +use crowdb_access_iceberg::{ + catalog::{Capabilities, CatalogContext, CatalogStore}, + commit::{CommitProofLimits, StagedCommitLimits, TableCreator}, + file::FileBlockStore, + namespace::{NamespaceRepository, NamespaceStore}, + operation::{PayloadStore, RetryAdmission, RetryLedger, RetryRecord}, + table::TableRepository, + wire::{IcebergErrorResponse, Principal, RequestKey}, +}; +use hyper::{body::Incoming, Method, Request, Response}; +use sha2::{Digest, Sha256}; + +use super::{ + body::{IcebergBody, SpoolPermit}, + http::{bad_request, response, service_unavailable}, + namespace_write::{mutation_error, now_ms, read_body}, + table_limits, +}; + +mod lifecycle; +mod mutation; +pub(super) mod request; + +pub(super) struct TableWrites { + store: Arc, + blocks: Arc, + creator: TableCreator, + namespaces: NamespaceRepository, + tables: TableRepository, + lifecycles: crowdb_access_iceberg::table::TableLifecycles, + ledger: RetryLedger, + payloads: PayloadStore, + limits: CommitProofLimits, + active: Arc, + pub(super) file_config: Option, +} + +impl TableWrites { + pub(super) fn new( + store: Arc, + blocks: Arc, + ) -> Self { + let limits = table_limits::commits(); + Self { + lifecycles: crowdb_access_iceberg::table::TableLifecycles::new(store.clone()), + store: store.clone(), + blocks: blocks.clone(), + creator: TableCreator::new(store.clone(), blocks) + .with_response_reserve(table_limits::RESPONSE_RESERVE) + .with_staged_limits(StagedCommitLimits { + evaluation: limits.preparation.evaluation, + snapshots: limits.snapshots, + auxiliary: limits.auxiliary, + }), + namespaces: NamespaceRepository::new(store.clone()), + tables: TableRepository::new(store.clone()), + ledger: RetryLedger::new(store.clone()), + payloads: PayloadStore::new(store), + limits, + active: Arc::new(AtomicUsize::new(0)), + file_config: None, + } + } + + pub(super) async fn execute( + &self, + context: CatalogContext, + capabilities: Capabilities, + principal: Principal, + request: Request, + ) -> Result, IcebergErrorResponse> { + if request.method() != Method::POST && request.method() != Method::DELETE { + return Err(super::table_read::unsupported()); + } + if !principal.namespace_write { + return Err(IcebergErrorResponse::new( + 403, + "ForbiddenException", + "Table write privilege is required", + )); + } + let _permit = SpoolPermit::acquire(&self.active).ok_or_else(service_unavailable)?; + let now = now_ms()?; + if request.headers().get_all("idempotency-key").iter().count() > 1 { + return Err(bad_request()); + } + let header = request + .headers() + .get("idempotency-key") + .map(|value| value.to_str()) + .transpose() + .map_err(|_| bad_request())?; + let key = RequestKey::parse(header, now).map_err(|_| bad_request())?; + let uri = request.uri().clone(); + let method = request.method().clone(); + let bytes = read_body(request.into_body()).await?; + let route = if method == Method::DELETE { + "DELETE table" + } else { + "POST table" + }; + let mut digest = Sha256::new(); + for value in [route.as_bytes(), uri.to_string().as_bytes(), bytes.as_slice()] { + digest.update((value.len() as u64).to_be_bytes()); + digest.update(value); + } + let mut record = RetryRecord { + identity: key.identity(), + principal: principal.name.into(), + route: route.into(), + digest: digest.finalize().into(), + context, + retained_until_ms: 0, + status: 0, + body: Vec::new(), + }; + let admission = self.admit(&mut record, key, now).await?; + let (record, resuming) = match admission { + RetryAdmission::Replay(record) => { + super::metrics::record_retry(3); + return Ok(response(record.status, record.body)); + } + RetryAdmission::New(record) => { + super::metrics::record_retry(1); + (record, false) + } + RetryAdmission::Resume(record) => { + super::metrics::record_retry(2); + (record, true) + } + }; + if method == Method::DELETE || uri.path() == "/v1/tables/rename" { + let result = self + .mutate_lifecycle(&record, capabilities, resuming, &method, &uri, &bytes) + .await; + let (status, body) = self.outcome_response(result, None, context).await?; + self.ledger + .finish(record, status, body.clone(), now_ms()?) + .await + .map_err(|error| mutation_error(&error))?; + return Ok(response(status, body)); + } + let target = request::parse(&uri); + let configuration_target = target.as_ref().ok().and_then(|target| { + let name = target.name.clone().or_else(|| { + crowdb_access_iceberg::commit::CreateTableRequest::decode( + &bytes, + self.limits.preparation.request.json, + ) + .ok() + .map(|request| request.name().to_owned()) + })?; + Some((target.namespace.clone(), name)) + }); + let result = match target { + Ok(target) => self.mutate(&record, capabilities, target, bytes, now).await, + Err(error) => Err(error), + }; + let (status, body) = self + .outcome_response(result, configuration_target, context) + .await?; + self.ledger + .finish(record, status, body.clone(), now_ms()?) + .await + .map_err(|error| mutation_error(&error))?; + Ok(response(status, body)) + } + + async fn outcome_response( + &self, + result: Result, + configuration_target: Option<(crowdb_access_iceberg::namespace::NamespaceIdentifier, String)>, + context: CatalogContext, + ) -> Result<(u16, Vec), IcebergErrorResponse> { + let (status, mut body) = match result { + Ok(outcome) => ( + outcome.status, + self.payloads + .get(&outcome.body) + .await + .map_err(|_| service_unavailable())?, + ), + Err(error) if error.error.code < 500 => ( + error.error.code, + serde_json::to_vec(&error).map_err(|_| service_unavailable())?, + ), + Err(error) => return Err(error), + }; + if status == 200 { + if let (Some(config), Some((namespace, name))) = (&self.file_config, configuration_target) { + #[derive(serde::Deserialize)] + struct Metadata { + location: String, + } + #[derive(serde::Deserialize)] + struct Envelope { + metadata: Metadata, + } + let envelope: Envelope = serde_json::from_slice(&body).map_err(|_| service_unavailable())?; + let table: crowdb_access_iceberg::file::TableLocation = + format!("{}/", envelope.metadata.location.trim_end_matches('/')) + .parse() + .map_err(|_| service_unavailable())?; + if table.catalog != context.catalog { + return Err(service_unavailable()); + } + config.append(&mut body, &namespace, &name, table.table)?; + } + } + Ok((status, body)) + } + + async fn admit( + &self, + record: &mut RetryRecord, + key: RequestKey, + now: u64, + ) -> Result { + for _ in 0..8 { + match self.ledger.begin(record.clone(), now).await { + Err(crowdb_access_iceberg::catalog::CatalogError::Busy) + if matches!(key, RequestKey::Internal(_)) => + { + record.identity = RequestKey::parse(None, now) + .map_err(|_| bad_request())? + .identity(); + } + result => return result.map_err(|error| mutation_error(&error)), + } + } + Err(service_unavailable()) + } +} diff --git a/app/crowdb-access-server/src/iceberg/table_write/lifecycle.rs b/app/crowdb-access-server/src/iceberg/table_write/lifecycle.rs new file mode 100644 index 000000000..1a45c62df --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/table_write/lifecycle.rs @@ -0,0 +1,136 @@ +use super::{ + super::http::{bad_request, decode_query, service_unavailable}, + TableWrites, +}; +use crowdb_access_iceberg::{ + catalog::{Capabilities, FormatAction}, + commit::TableCommitOutcome, + key::NameSuffix, + namespace::NamespaceIdentifier, + operation::RetryRecord, + table::{TableLifecycleAction, TableLifecycleRequest}, + wire::IcebergErrorResponse, +}; +use hyper::{Method, Uri}; + +#[derive(serde::Deserialize)] +struct Identifier { + namespace: Vec, + name: String, +} + +#[derive(serde::Deserialize)] +struct Rename { + source: Identifier, + destination: Identifier, +} + +impl Identifier { + fn validate(self) -> Result<(NamespaceIdentifier, String), IcebergErrorResponse> { + NameSuffix { + parent: None, + name: &self.name, + } + .encode() + .map_err(|_| bad_request())?; + Ok(( + NamespaceIdentifier::new(self.namespace).map_err(|_| bad_request())?, + self.name, + )) + } +} + +impl TableWrites { + pub(super) async fn mutate_lifecycle( + &self, + record: &RetryRecord, + capabilities: Capabilities, + resuming: bool, + method: &Method, + uri: &Uri, + bytes: &[u8], + ) -> Result { + let (namespace, name, action) = if uri.path() == "/v1/tables/rename" { + if *method != Method::POST { + return Err(super::super::table_read::unsupported()); + } + if uri.query().is_some() { + return Err(bad_request()); + } + let rename: Rename = serde_json::from_slice(bytes).map_err(|_| bad_request())?; + let (namespace, name) = rename.source.validate()?; + let (destination, target) = rename.destination.validate()?; + ( + namespace, + name, + TableLifecycleAction::Rename { + namespace: destination, + name: target, + }, + ) + } else { + if *method != Method::DELETE || !bytes.is_empty() { + return Err(bad_request()); + } + let purge_requested = purge(uri)?; + let path: Uri = uri.path().parse().map_err(|_| bad_request())?; + let target = super::request::parse(&path)?; + ( + target.namespace, + target.name.ok_or_else(super::super::table_read::unsupported)?, + TableLifecycleAction::Drop { purge_requested }, + ) + }; + let existing = if resuming { + self.lifecycles + .load(record.context, record.identity.operation) + .await + .map_err(|_| service_unavailable())? + } else { + None + }; + let version = if let Some(operation) = existing { + operation.before.format_version + } else { + let parent = self + .namespaces + .load(record.context, &namespace) + .await + .map_err(|_| service_unavailable())? + .ok_or_else(super::super::table_read::missing_table)?; + self.tables + .select(record.context, parent.namespace, &name) + .await + .map_err(|_| service_unavailable())? + .ok_or_else(super::super::table_read::missing_table)? + .head + .format_version + }; + if !capabilities.supports(version, FormatAction::Write) { + return Err(super::super::table_read::unsupported()); + } + self.lifecycles.execute(&TableLifecycleRequest { context: record.context, identity: record.identity, + principal: record.principal.clone(), namespace, name, action }).await.map_err(|error| { + tracing::error!(%error, "table lifecycle remains recoverable; retry with the same request key"); + service_unavailable() + }) + } +} + +fn purge(uri: &Uri) -> Result { + let Some(query) = uri.query() else { + return Ok(false); + }; + let (name, value) = query.split_once('=').ok_or_else(bad_request)?; + if decode_query(name)? != "purgeRequested" { + return Err(bad_request()); + } + let value = decode_query(value)?; + if value.eq_ignore_ascii_case("true") { + Ok(true) + } else if value.eq_ignore_ascii_case("false") { + Ok(false) + } else { + Err(bad_request()) + } +} diff --git a/app/crowdb-access-server/src/iceberg/table_write/mutation.rs b/app/crowdb-access-server/src/iceberg/table_write/mutation.rs new file mode 100644 index 000000000..c874e19cb --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/table_write/mutation.rs @@ -0,0 +1,232 @@ +use super::super::{ + http::{bad_request, service_unavailable}, + table_read::missing_table, +}; +use super::{request::Target, TableWrites}; +use crowdb_access_iceberg::{ + catalog::{Capabilities, CatalogError, FormatAction}, + commit::{ + recover_table_commit, CommitPublicationError, CommitRequest, CreateTableRequest, StagedCommitRequest, + TableCommitJournal, TableCommitOperation, TableCommitOutcome, TableCommitPhase, TableCreationRequest, + TableRequirement, + }, + operation::RetryRecord, + wire::IcebergErrorResponse, +}; + +struct CommitInput<'input> { + parsed: &'input CommitRequest, + body: &'input [u8], +} + +impl TableWrites { + pub(super) async fn mutate( + &self, + record: &RetryRecord, + capabilities: Capabilities, + target: Target, + body: Vec, + now: u64, + ) -> Result { + let timestamp_ms = i64::try_from(now).map_err(|_| service_unavailable())?; + let Some(name) = target.name else { + let parsed = CreateTableRequest::decode(&body, self.limits.preparation.request.json) + .map_err(|_| bad_request())?; + if !capabilities.supports( + parsed + .format_version(self.limits.preparation.request.json) + .map_err(|_| bad_request())?, + FormatAction::Create, + ) { + return Err(super::super::table_read::unsupported()); + } + let staged = parsed.stage_create(); + let request = TableCreationRequest { + context: record.context, + identity: record.identity, + principal: record.principal.clone(), + namespace: target.namespace, + body, + timestamp_ms, + }; + return if staged { + let expiry = timestamp_ms + .checked_add(24 * 60 * 60 * 1000) + .ok_or_else(service_unavailable)?; + self.creator.stage(&request, expiry).await + } else { + self.creator.create(&request).await + } + .map_err(creation_error); + }; + let parsed = + CommitRequest::decode(&body, self.limits.preparation.request).map_err(|_| bad_request())?; + parsed + .check_identifier(&target.namespace, &name) + .map_err(|_| bad_request())?; + if parsed + .requirements + .iter() + .any(|requirement| matches!(requirement, TableRequirement::AssertCreate)) + { + if !capabilities.supports(parsed.create_version(), FormatAction::Create) { + return Err(super::super::table_read::unsupported()); + } + return self + .creator + .commit_staged(&StagedCommitRequest { + context: record.context, + identity: record.identity, + principal: record.principal.clone(), + namespace: target.namespace, + name, + body, + timestamp_ms, + }) + .await + .map_err(creation_error); + } + self.update( + record, + capabilities, + &target.namespace, + &name, + CommitInput { + parsed: &parsed, + body: &body, + }, + timestamp_ms, + ) + .await + } + + async fn update( + &self, + record: &RetryRecord, + capabilities: Capabilities, + namespace: &crowdb_access_iceberg::namespace::NamespaceIdentifier, + name: &str, + input: CommitInput<'_>, + timestamp_ms: i64, + ) -> Result { + let journal = TableCommitJournal::new(self.store.clone()); + let operation = if let Some(operation) = journal + .load(record.context, record.identity.operation) + .await + .map_err(|error| storage_error(&error))? + { + operation + } else { + let namespace = self + .namespaces + .load(record.context, namespace) + .await + .map_err(|error| storage_error(&error))? + .ok_or_else(missing_table)?; + let selected = self + .tables + .select(record.context, namespace.namespace, name) + .await + .map_err(|error| storage_error(&error))? + .ok_or_else(missing_table)?; + if selected.head.pending_operation.is_some() { + return Err(service_unavailable()); + } + let mut version = selected.head.format_version; + let mut upgrades = input.parsed.upgrade_targets().peekable(); + if upgrades.peek().is_none() && !capabilities.supports(version, FormatAction::Write) { + return Err(super::super::table_read::unsupported()); + } + for target in upgrades { + if !capabilities.supports_upgrade(version, target) { + return Err(super::super::table_read::unsupported()); + } + version = target; + } + let input = self + .payloads + .put(record.context.catalog, record.identity.operation, input.body) + .await + .map_err(|error| storage_error(&error))?; + let initial = TableCommitOperation { + context: record.context, + identity: record.identity, + principal: record.principal.clone(), + revision: 1, + timestamp_ms, + phase: TableCommitPhase::Prepared, + input, + before: selected.head, + candidate: None, + outcome: None, + }; + match journal.begin(initial).await { + Ok(operation) => operation, + Err(CatalogError::Conflict) => journal + .load(record.context, record.identity.operation) + .await + .map_err(|error| storage_error(&error))? + .ok_or_else(service_unavailable)?, + Err(error) => return Err(storage_error(&error)), + } + }; + if operation.identity != record.identity + || operation.principal != record.principal + || operation.before.name != name + || self + .payloads + .get(&operation.input) + .await + .map_err(|error| storage_error(&error))? + != input.body + { + return Err(IcebergErrorResponse::new( + 409, + "CommitFailedException", + "Commit identity conflicts", + )); + } + recover_table_commit( + self.store.clone(), + self.blocks.clone(), + record.context, + record.identity.operation, + self.limits, + ) + .await + .map_err(|error| { + tracing::error!(%error, "table commit remains recoverable; retry with the same request key"); + service_unavailable() + }) + } +} + +fn storage_error(error: &CatalogError) -> IcebergErrorResponse { + tracing::error!(%error, "table mutation storage remains recoverable; retry with the same request key"); + service_unavailable() +} + +fn creation_error(error: CommitPublicationError) -> IcebergErrorResponse { + match error { + CommitPublicationError::NamespaceMissing => { + IcebergErrorResponse::new(404, "NoSuchNamespaceException", "Namespace does not exist") + } + CommitPublicationError::Unsupported(_) => super::super::table_read::unsupported(), + CommitPublicationError::Catalog(CatalogError::Conflict) => { + IcebergErrorResponse::new(409, "CommitFailedException", "Table creation conflicts") + } + CommitPublicationError::Evaluation(crowdb_access_iceberg::commit::EvaluationError::Requirement( + crowdb_access_iceberg::commit::RequirementError::Failed(_), + )) => IcebergErrorResponse::new(409, "CommitFailedException", "Table requirement failed"), + CommitPublicationError::Evaluation(_) + | CommitPublicationError::Metadata( + crowdb_access_iceberg::table::TableMetadataError::Field(_) + | crowdb_access_iceberg::table::TableMetadataError::Json(_) + | crowdb_access_iceberg::table::TableMetadataError::Bounds, + ) => bad_request(), + error => { + tracing::error!(%error, "table creation remains recoverable; retry with the same request key"); + service_unavailable() + } + } +} diff --git a/app/crowdb-access-server/src/iceberg/table_write/request.rs b/app/crowdb-access-server/src/iceberg/table_write/request.rs new file mode 100644 index 000000000..ca3b281e1 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/table_write/request.rs @@ -0,0 +1,33 @@ +use super::super::{http::bad_request, namespace_read::decode_path}; +use crowdb_access_iceberg::{key::NameSuffix, namespace::NamespaceIdentifier, wire::IcebergErrorResponse}; + +pub(in crate::iceberg) struct Target { + pub namespace: NamespaceIdentifier, + pub name: Option, +} + +pub(in crate::iceberg) fn parse(uri: &hyper::Uri) -> Result { + if uri.query().is_some() { + return Err(bad_request()); + } + let suffix = uri + .path() + .strip_prefix("/v1/namespaces/") + .ok_or_else(bad_request)?; + let mut parts = suffix.split('/'); + let namespace = NamespaceIdentifier::from_rest(&decode_path(parts.next().ok_or_else(bad_request)?)?) + .map_err(|_| bad_request())?; + if parts.next() != Some("tables") { + return Err(bad_request()); + } + let name = parts.next().map(decode_path).transpose()?; + if parts.next().is_some() { + return Err(super::super::table_read::unsupported()); + } + if let Some(name) = &name { + NameSuffix { parent: None, name } + .encode() + .map_err(|_| bad_request())?; + } + Ok(Target { namespace, name }) +} diff --git a/app/crowdb-access-server/src/iceberg_main.rs b/app/crowdb-access-server/src/iceberg_main.rs new file mode 100644 index 000000000..484dfeaef --- /dev/null +++ b/app/crowdb-access-server/src/iceberg_main.rs @@ -0,0 +1,5 @@ +#[tokio::main] +async fn main() -> Result<(), Box> { + tracing_subscriber::fmt::init(); + crowdb_access_server::iceberg::run().await +} diff --git a/app/crowdb-access-server/src/lib.rs b/app/crowdb-access-server/src/lib.rs index 9157e5136..306c81425 100644 --- a/app/crowdb-access-server/src/lib.rs +++ b/app/crowdb-access-server/src/lib.rs @@ -3,6 +3,8 @@ //! Independent listener lifecycle for external access protocols. +pub mod iceberg; + #[cfg(feature = "s3")] pub mod credentials; #[cfg(feature = "s3")] diff --git a/app/crowdb-access-server/src/main.rs b/app/crowdb-access-server/src/main.rs index cd57e890e..1294b36d6 100644 --- a/app/crowdb-access-server/src/main.rs +++ b/app/crowdb-access-server/src/main.rs @@ -34,9 +34,12 @@ use tokio::net::TcpListener; #[tokio::main] async fn main() -> Result<(), Box> { - tracing_subscriber::fmt::init(); + tracing_subscriber::fmt().with_writer(std::io::stderr).init(); #[cfg(feature = "s3")] - if std::env::args().nth(1).as_deref() == Some("issue-user") { + if matches!( + std::env::args().nth(1).as_deref(), + Some("issue-user" | "ensure-user" | "lookup-user") + ) { return issue_user().await; } #[cfg(feature = "s3")] @@ -146,18 +149,26 @@ fn configure_large_write(config: &mut S3ServiceConfig) -> Result<(), Box Result<(), Box> { + let command = std::env::args().nth(1).ok_or("missing S3 user command")?; let user = std::env::args() .nth(2) .filter(|value| !value.is_empty()) - .ok_or("usage: crowdb-access-server issue-user USER")?; + .ok_or("usage: crowdb-access-server issue-user|ensure-user|lookup-user USER")?; if std::env::args().nth(3).is_some() { - return Err("usage: crowdb-access-server issue-user USER".into()); + return Err("usage: crowdb-access-server issue-user|ensure-user|lookup-user USER".into()); } let master_key = MasterKey::from_hex(&required_env("CROWDB_S3_MASTER_KEY")?)?; let cipher = Arc::new(CredentialCipher::new(&master_key)); let control = Arc::new(CrowdbKvClient::new(KvConfig::new(management_seeds()?))); let authority = CredentialAuthority::new(control, cipher); - let token = authority.issue_user(user.as_bytes()).await?; + let token = match command.as_str() { + "ensure-user" => authority.ensure_user(user.as_bytes()).await?, + "lookup-user" => authority + .lookup_user(user.as_bytes()) + .await? + .ok_or("S3 user does not exist")?, + _ => authority.issue_user(user.as_bytes()).await?, + }; println!("AWS_ACCESS_KEY_ID={}", token.access_key_id); println!("AWS_SECRET_ACCESS_KEY={}", token.secret_key); Ok(()) diff --git a/app/crowdb-access-server/tests/common/iceberg_background.rs b/app/crowdb-access-server/tests/common/iceberg_background.rs new file mode 100644 index 000000000..e33fda0eb --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_background.rs @@ -0,0 +1,61 @@ +use std::sync::Arc; +use std::time::Duration; + +use crowdb_access_iceberg::catalog::{CatalogContext, RoutedCatalogStore}; +use crowdb_access_iceberg::key::{NamespaceId, OperationId}; +use crowdb_access_iceberg::namespace::{ + NamespaceAction, NamespaceIdentifier, NamespaceJournal, NamespaceOperation, NamespacePhase, + NamespaceRepository, +}; +use crowdb_access_iceberg::operation::{PayloadStore, RequestIdentity}; + +use super::common::now_ms; + +pub async fn verify(store: Arc, context: CatalogContext) { + let identity = RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }; + let input = PayloadStore::new(store.clone()) + .put(context.catalog, identity.operation, b"{}") + .await + .unwrap(); + let operation = NamespaceOperation { + context, + identity, + principal: "writer".into(), + action: NamespaceAction::Create, + identifier: NamespaceIdentifier::new(vec!["background-recovered".into()]).unwrap(), + namespace: NamespaceId::random(), + parent: None, + phase: NamespacePhase::Prepared, + revision: 1, + input, + mutation: None, + scan_after: Vec::new(), + scan_generation: 0, + outcome: None, + }; + let journal = NamespaceJournal::new(store.clone()); + journal.begin(operation.clone()).await.unwrap(); + tokio::time::timeout(Duration::from_secs(15), async { + loop { + let current = journal.load(context, identity.operation).await.unwrap().unwrap(); + if current.phase == NamespacePhase::Complete { + assert_eq!(current.outcome.unwrap().status, 200); + let authority = NamespaceRepository::new(store.clone()) + .load(context, &operation.identifier) + .await + .unwrap() + .unwrap(); + assert_eq!(authority.namespace, operation.namespace); + if authority.pending_operation.is_none() { + break; + } + } + tokio::time::sleep(Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); +} diff --git a/app/crowdb-access-server/tests/common/iceberg_client.py b/app/crowdb-access-server/tests/common/iceberg_client.py new file mode 100644 index 000000000..f0745c3b8 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_client.py @@ -0,0 +1,99 @@ +import sys +import uuid + +import requests +from pyiceberg.catalog import load_catalog +from pyiceberg.exceptions import RESTError, UnauthorizedError, NamespaceAlreadyExistsError, NamespaceNotEmptyError, NoSuchNamespaceError + + +def main(): + uri = sys.argv[1] + properties = {"type": "rest", "uri": uri, "token": "r" * 32} + for extra in ({}, {"warehouse": ""}, {"token": "w" * 32}): + catalog = load_catalog("crowdb", **(properties | extra)) + assert catalog.properties["crowdb.iceberg.v1.read"] == "true" + assert catalog.properties["crowdb.iceberg.v3.write"] == "true" + for extra, expected in ( + ({"warehouse": "unknown"}, RESTError), + ({"token": "wrong"}, UnauthorizedError), + ): + try: + load_catalog("crowdb", **(properties | extra)) + except expected as error: + if "warehouse" in extra: + assert "NoSuchWarehouseException" in str(error) + else: + raise AssertionError(f"expected {expected.__name__}") + response = requests.get( + uri + "/v1/config", + headers={"Authorization": "Bearer " + "r" * 32}, + timeout=5, + ) + response.raise_for_status() + assert set(response.json()["endpoints"]) == { + "GET /v1/{prefix}/namespaces", + "GET /v1/{prefix}/namespaces/{namespace}", + "HEAD /v1/{prefix}/namespaces/{namespace}", + "POST /v1/{prefix}/namespaces", + "POST /v1/{prefix}/namespaces/{namespace}/properties", + "DELETE /v1/{prefix}/namespaces/{namespace}", + } + assert response.json()["idempotency-key-lifetime"] == "PT24H" + catalog = load_catalog("crowdb", **properties) + namespaces = catalog.list_namespaces() + assert isinstance(namespaces, list) + for namespace in namespaces: + assert isinstance(catalog.load_namespace_properties(namespace), dict) + complete = requests.get(uri + "/v1/namespaces", headers={"Authorization": "Bearer " + "r" * 32}, timeout=5) + complete.raise_for_status() + assert complete.json()["next-page-token"] is None + assert len(complete.json()["namespaces"]) == len(namespaces) + missing = requests.head(uri + "/v1/namespaces/missing-namespace", headers={"Authorization": "Bearer " + "r" * 32}, timeout=5) + assert missing.status_code == 404 and missing.content == b"" + if "--read-only" in sys.argv[2:]: + print("PyIceberg config, authentication and namespace reads passed") + else: + verify_namespaces(uri, properties) + print("PyIceberg config, authentication and namespace CRUD passed") + + +def verify_namespaces(uri, properties): + writer = load_catalog("crowdb", **(properties | {"token": "w" * 32})) + namespace = ("client-" + uuid.uuid4().hex,) + child = namespace + ("冰+a%2F",) + writer.create_namespace(namespace, {"owner": "original"}) + assert writer.namespace_exists(namespace) + assert writer.load_namespace_properties(namespace) == {"owner": "original"} + try: + writer.create_namespace(namespace) + except NamespaceAlreadyExistsError: + pass + else: + raise AssertionError("duplicate namespace was accepted") + writer.create_namespace(child) + assert writer.list_namespaces(namespace) == [child] + update = writer.update_namespace_properties(namespace, removals={"owner", "absent"}, updates={"owner2": "retained"}) + assert update.removed == ["owner"] and update.missing == ["absent"] + assert writer.load_namespace_properties(namespace) == {"owner2": "retained"} + try: + writer.drop_namespace(namespace) + except NamespaceNotEmptyError: + pass + else: + raise AssertionError("nonempty namespace was dropped") + denied = requests.post(uri + "/v1/namespaces", json={"namespace": ["denied"]}, headers={"Authorization": "Bearer " + "r" * 32}, timeout=5) + assert denied.status_code == 403 + writer.drop_namespace(child) + writer.drop_namespace(namespace) + assert not writer.namespace_exists(namespace) + for operation in (writer.load_namespace_properties, writer.drop_namespace, writer.list_namespaces): + try: + operation(namespace) + except NoSuchNamespaceError: + pass + else: + raise AssertionError("missing namespace was accepted") + + +if __name__ == "__main__": + main() diff --git a/app/crowdb-access-server/tests/common/iceberg_commit_case.rs b/app/crowdb-access-server/tests/common/iceberg_commit_case.rs new file mode 100644 index 000000000..c12455c74 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_commit_case.rs @@ -0,0 +1,175 @@ +use reqwest::{Client, Response}; +use serde_json::{json, Value}; +use sha2::{Digest, Sha256}; + +fn identity() -> String { + static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); + static BUCKETS: [std::sync::atomic::AtomicU64; 64] = [const { std::sync::atomic::AtomicU64::new(0) }; 64]; + for _ in 0..100_000 { + let now = super::common::now_ms(); + let key = format!( + "{:08x}-{:04x}-7000-8000-{:012x}", + now >> 16, + now & 0xffff, + NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed) + ); + let operation: crowdb_access_iceberg::key::OperationId = key.parse().unwrap(); + let crowdb_access_iceberg::key::IcebergKey::System { suffix, .. } = + crowdb_access_iceberg::operation::ledger_key( + crowdb_access_iceberg::key::SystemScope::RetryBinding, + operation, + ) + .unwrap() + else { + unreachable!() + }; + let bucket = usize::from(u16::from_be_bytes([suffix[14], suffix[15]])) - 1; + let mask = 1_u64 << (bucket % 64); + if BUCKETS[bucket / 64].fetch_or(mask, std::sync::atomic::Ordering::Relaxed) & mask == 0 { + return key; + } + } + panic!("crash fixture exhausted distinct retry admission buckets") +} + +pub struct TestCommitCase { + pub path: String, + pub body: String, + pub identity: String, + pub name: String, + pub staged: bool, + pub generation: u64, +} + +pub async fn post( + endpoint: &str, + path: &str, + identity: &str, + body: &str, +) -> Result { + Client::new() + .post(format!("{endpoint}{path}")) + .bearer_auth("w".repeat(32)) + .header("content-type", "application/json") + .header("idempotency-key", identity) + .body(body.to_owned()) + .send() + .await +} + +pub async fn success(endpoint: &str, path: &str, body: &Value) -> Value { + let response = post(endpoint, path, &identity(), &body.to_string()) + .await + .unwrap(); + let status = response.status(); + let text = response.text().await.unwrap(); + assert_eq!(status, 200, "{text}"); + serde_json::from_str(&text).unwrap() +} + +impl TestCommitCase { + pub async fn prepare(endpoint: &str, kind: &str, name: String) -> Self { + let mut create = json!({"name":name,"schema":{"type":"struct","schema-id":0,"fields":[ + {"id":1,"name":"id","type":"long","required":true}]},"properties":properties()}); + let path = format!("/v1/namespaces/analytics/tables/{name}"); + let (path, body, staged, generation) = match kind { + "create" => ("/v1/namespaces/analytics/tables".into(), create, false, 1), + "stage" => { + create["stage-create"] = json!(true); + ("/v1/namespaces/analytics/tables".into(), create, true, 0) + } + "update" => { + success(endpoint, "/v1/namespaces/analytics/tables", &create).await; + ( + path, + json!({"requirements":[],"updates":[{"action":"set-properties","updates":{"owner":"after"}}]}), + false, + 2, + ) + } + "publish-stage" => { + create["stage-create"] = json!(true); + let response = success(endpoint, "/v1/namespaces/analytics/tables", &create).await; + (path, staged_commit(&response["metadata"]), false, 1) + } + _ => panic!("unknown case"), + }; + Self { + path, + body: body.to_string(), + identity: identity(), + name, + staged, + generation, + } + } + + pub async fn replay(&self, endpoint: &str) -> Value { + let first = post(endpoint, &self.path, &self.identity, &self.body) + .await + .unwrap(); + let status = first.status(); + let bytes = first.bytes().await.unwrap(); + assert_eq!(status, 200, "{}", String::from_utf8_lossy(&bytes)); + let replay = post(endpoint, &self.path, &self.identity, &self.body) + .await + .unwrap(); + assert_eq!(replay.status(), 200); + assert_eq!(replay.bytes().await.unwrap(), bytes); + let changed = post(endpoint, &self.path, &self.identity, &format!("{} ", self.body)) + .await + .unwrap(); + assert_eq!(changed.status(), 409, "{}", changed.text().await.unwrap()); + let loaded = Client::new() + .get(format!("{endpoint}/v1/namespaces/analytics/tables/{}", self.name)) + .bearer_auth("r".repeat(32)) + .send() + .await + .unwrap(); + let status = loaded.status(); + let loaded = loaded.text().await.unwrap(); + if self.staged { + assert_eq!(status, 404, "{loaded}"); + } else { + assert_eq!(status, 200, "{loaded}"); + let loaded: Value = serde_json::from_str(&loaded).unwrap(); + let result: Value = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(loaded["metadata-location"], result["metadata-location"]); + if self.generation == 2 { + assert_eq!(loaded["metadata"]["properties"]["owner"], "after"); + } + } + serde_json::from_slice(&bytes).unwrap() + } +} + +fn properties() -> Value { + let mut properties = serde_json::Map::new(); + for field in 0..16 { + let mut value = String::new(); + for part in 0..32 { + use std::fmt::Write; + let digest = Sha256::digest(format!("field-{field}-part-{part}").as_bytes()); + for byte in digest { + write!(value, "{byte:02x}").unwrap(); + } + } + properties.insert(format!("random-{field}"), json!(value)); + } + Value::Object(properties) +} + +fn staged_commit(metadata: &Value) -> Value { + json!({"requirements":[{"type":"assert-create"}],"updates":[ + {"action":"assign-uuid","uuid":metadata["table-uuid"]}, + {"action":"upgrade-format-version","format-version":metadata["format-version"]}, + {"action":"add-schema","schema":metadata["schemas"][0]}, + {"action":"set-current-schema","schema-id":-1}, + {"action":"add-spec","spec":metadata["partition-specs"][0]}, + {"action":"set-default-spec","spec-id":-1}, + {"action":"add-sort-order","sort-order":metadata["sort-orders"][0]}, + {"action":"set-default-sort-order","sort-order-id":-1}, + {"action":"set-location","location":metadata["location"]}, + {"action":"set-properties","updates":metadata["properties"]} + ]}) +} diff --git a/app/crowdb-access-server/tests/common/iceberg_commit_child.rs b/app/crowdb-access-server/tests/common/iceberg_commit_child.rs new file mode 100644 index 000000000..1003d790d --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_commit_child.rs @@ -0,0 +1,163 @@ +use std::net::SocketAddr; +use std::path::PathBuf; +use std::process::{Child, Command, Stdio}; +use std::sync::Arc; +use std::time::Duration; + +use crowdb_access_iceberg::{ + catalog::{CatalogRepository, ClearBounds, RoutedCatalogStore}, + file::NativeFileBlocks, + wire::BearerAuthenticator, +}; +use crowdb_access_server::iceberg::{serve, IcebergHttpService}; +use crowdb_chunk_client::{ChunkIoClient, ChunkIoClientConfig, SmallWritePolicy}; +use crowdb_chunk_kv_client::{ + ChunkKvClient, ChunkKvRpcTransport, ClientConfig, Group0ChunkKvRangeCatalogSource, +}; +use crowdb_kv_client::{ClientConfig as KvConfig, CrowdbKvClient}; + +use super::fault::{TestBoundary, TestCommitBlocks, TestCommitStore}; + +pub async fn run() { + let Some(config) = std::env::var_os("CROWDB_TEST_COMMIT_CHILD") else { + return; + }; + let config: serde_json::Value = serde_json::from_str(config.to_str().unwrap()).unwrap(); + let seeds: Vec = serde_json::from_value(config["seeds"].clone()).unwrap(); + let control = Arc::new(CrowdbKvClient::new(KvConfig::new(seeds.clone()))); + let client_config = ClientConfig::default(); + let source = Arc::new(Group0ChunkKvRangeCatalogSource::from_shared(control.clone())); + let transport = Arc::new(ChunkKvRpcTransport::new( + client_config.max_owner_connections, + 1, + 2, + )); + let client = Arc::new(ChunkKvClient::new(client_config, source, transport).unwrap()); + client.refresh_catalog().await.unwrap(); + let chunks = ChunkIoClient::connect_with_kv( + ChunkIoClientConfig { + management_seeds: seeds, + diskio_connections_per_endpoint: 2, + diskio_rpc_workers: 2, + small_write: SmallWritePolicy::default(), + }, + control, + ) + .await + .unwrap(); + let boundary = Arc::new(TestBoundary::new( + usize::try_from(config["target"].as_u64().unwrap()).unwrap(), + config["after"].as_bool().unwrap(), + config["marker"].as_str().unwrap().into(), + )); + let store = Arc::new(TestCommitStore { + inner: Arc::new(RoutedCatalogStore::new(client)), + boundary: boundary.clone(), + }); + let blocks = Arc::new(TestCommitBlocks { + inner: Arc::new(NativeFileBlocks::new(chunks, store.clone())), + boundary, + }); + let repository = Arc::new( + CatalogRepository::new( + store.clone(), + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + ) + .unwrap(), + ); + let authentication = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let address = config["address"].as_str().unwrap(); + let service = IcebergHttpService::new(repository, authentication, Duration::from_secs(300)) + .with_namespaces(store.clone()) + .unwrap() + .with_fileio(store.clone(), blocks.clone(), "us-east-1".into()) + .unwrap() + .with_tables(store.clone(), blocks) + .unwrap() + .with_table_credentials(store, format!("http://{address}")) + .unwrap(); + let listener = tokio::net::TcpListener::bind(address).await.unwrap(); + serve(listener, Arc::new(service), std::future::pending()) + .await + .unwrap(); +} + +pub struct TestCommitChild { + child: Child, + pub address: SocketAddr, + pub marker: PathBuf, +} + +impl TestCommitChild { + pub async fn start(seeds: &[String], marker: PathBuf, target: usize, after: bool) -> Self { + let reservation = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + let address = reservation.local_addr().unwrap(); + drop(reservation); + let configuration = serde_json::json!({"seeds":seeds,"address":address.to_string(),"marker":marker,"target":target,"after":after}); + let child = Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "native_fault_listener_child", + "--ignored", + "--nocapture", + ]) + .env("CROWDB_TEST_COMMIT_CHILD", configuration.to_string()) + .stdout(Stdio::inherit()) + .stderr(Stdio::inherit()) + .spawn() + .unwrap(); + let mut process = Self { + child, + address, + marker, + }; + tokio::time::timeout(Duration::from_secs(30), async { + loop { + assert!( + process.child.try_wait().unwrap().is_none(), + "fault listener exited" + ); + if tokio::net::TcpStream::connect(address).await.is_ok() { + break; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .unwrap(); + process + } + + pub async fn paused(&mut self) -> serde_json::Value { + tokio::time::timeout(Duration::from_secs(60), async { + loop { + assert!( + self.child.try_wait().unwrap().is_none(), + "fault listener exited before boundary" + ); + if let Ok(bytes) = std::fs::read(&self.marker) { + if let Ok(value) = serde_json::from_slice::(&bytes) { + if value["paused"] == true { + return value; + } + } + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap() + } +} + +impl Drop for TestCommitChild { + fn drop(&mut self) { + let _ = self.child.kill(); + let _ = self.child.wait(); + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_commit_fault.rs b/app/crowdb-access-server/tests/common/iceberg_commit_fault.rs new file mode 100644 index 000000000..60c8a01c4 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_commit_fault.rs @@ -0,0 +1,128 @@ +use std::path::PathBuf; +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; + +use async_trait::async_trait; +use crowdb_access_iceberg::{ + catalog::{CasOutcome, CatalogStore, RoutedCatalogStore, StoreError, StoredValue}, + file::{ChunkRoot, FileBlockStore, FileIdentity, FileIoError, MultipartPartScan, MultipartPartStore}, + key::IcebergKey, + namespace::{ChildScan, NamespaceStore}, + record::StorageRecord, +}; +use crowdb_chunk_kv_client::MultiScanPage; +use crowdb_protocol::chunk_kv::ClientRequestId; + +pub struct TestBoundary { + count: AtomicUsize, + target: usize, + after: bool, + marker: PathBuf, +} + +impl TestBoundary { + pub fn new(target: usize, after: bool, marker: PathBuf) -> Self { + Self { + count: AtomicUsize::new(0), + target, + after, + marker, + } + } + + pub async fn before(&self, label: &str) -> usize { + let index = self.count.fetch_add(1, Ordering::SeqCst) + 1; + self.observe(index, label, false).await; + index + } + + pub async fn observe(&self, index: usize, label: &str, after: bool) { + let paused = index == self.target && after == self.after; + let state = serde_json::json!({"index":index,"label":label,"after":after,"paused":paused}); + std::fs::write(&self.marker, serde_json::to_vec(&state).unwrap()).unwrap(); + if paused { + std::future::pending::<()>().await; + } + } +} + +pub struct TestCommitStore { + pub inner: Arc, + pub boundary: Arc, +} + +#[async_trait] +impl MultipartPartStore for TestCommitStore { + async fn scan_multipart_parts(&self, scan: MultipartPartScan) -> Result { + self.inner.scan_multipart_parts(scan).await + } +} + +#[async_trait] +impl NamespaceStore for TestCommitStore { + async fn scan_children(&self, request: ChildScan) -> Result { + self.inner.scan_children(request).await + } + + async fn delete_mapping( + &self, + key: &[u8], + expected: &[u8], + identity: ClientRequestId, + ) -> Result { + let index = self.boundary.before("mapping-delete").await; + let result = self.inner.delete_mapping(key, expected, identity).await; + self.boundary.observe(index, "mapping-delete", true).await; + result + } +} + +#[async_trait] +impl CatalogStore for TestCommitStore { + async fn get(&self, key: &[u8]) -> Result, StoreError> { + self.inner.get(key).await + } + + async fn compare_exchange( + &self, + key: &[u8], + expected: Option<&[u8]>, + value: &[u8], + identity: ClientRequestId, + ) -> Result { + let label = match StorageRecord::decode(&IcebergKey::decode(key)?, value)? { + StorageRecord::TableCreateOperation(operation) => format!("create-{:?}", operation.phase), + StorageRecord::TableCommitOperation(operation) => format!("commit-{:?}", operation.phase), + StorageRecord::TableHead(head) => format!("head-{}", head.generation), + StorageRecord::File(_) => "file-record".into(), + StorageRecord::FileMapping(_) => "file-mapping".into(), + StorageRecord::MultipartSession(session) => format!("multipart-{:?}", session.phase), + _ => "journal-or-fence".into(), + }; + let index = self.boundary.before(&label).await; + let result = self.inner.compare_exchange(key, expected, value, identity).await; + self.boundary.observe(index, &label, true).await; + result + } +} + +pub struct TestCommitBlocks { + pub inner: Arc, + pub boundary: Arc, +} + +#[async_trait] +impl FileBlockStore for TestCommitBlocks { + async fn put(&self, owner: FileIdentity, height: u8, bytes: &[u8]) -> Result { + let index = self.boundary.before("file-block").await; + let result = self.inner.put(owner, height, bytes).await; + self.boundary.observe(index, "file-block", true).await; + result + } + + async fn read(&self, root: &ChunkRoot) -> Result, FileIoError> { + self.inner.read(root).await + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_commit_loser.rs b/app/crowdb-access-server/tests/common/iceberg_commit_loser.rs new file mode 100644 index 000000000..f7a93ea10 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_commit_loser.rs @@ -0,0 +1,112 @@ +use std::path::Path; + +use crowdb_access_iceberg::{ + catalog::CatalogContext, + commit::{TableCommitJournal, TableCommitPhase}, + file::FileRepository, + namespace::{NamespaceIdentifier, NamespaceRepository}, + table::TableRepository, +}; +use serde_json::json; + +use super::{case, child::TestCommitChild, common::TestIcebergStack, process::TestIcebergProcess}; + +pub async fn verify(stack: &TestIcebergStack, context: CatalogContext, directory: &Path, offset: usize) { + let setup = TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let case = case::TestCommitCase::prepare( + &format!("http://{}", setup.address), + "update", + "losing-candidate".into(), + ) + .await; + drop(setup); + let mut loser = TestCommitChild::start( + &stack.cluster.mgmt_endpoints, + directory.join("loser.json"), + offset, + false, + ) + .await; + let endpoint = format!("http://{}", loser.address); + let path = case.path.clone(); + let identity = case.identity.clone(); + let body = case.body.clone(); + let request = tokio::spawn(async move { case::post(&endpoint, &path, &identity, &body).await }); + assert_eq!(loser.paused().await["label"], "head-2"); + let store = stack.store().await; + let journal = TableCommitJournal::new(store.clone()); + let operation = journal + .load(context, case.identity.parse().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(operation.phase, TableCommitPhase::Publishing); + let candidate = operation.candidate.unwrap(); + let files = FileRepository::new(store.clone()); + assert!(files + .load(context, &candidate.metadata_location) + .await + .unwrap() + .is_some()); + let winner = TestCommitChild::start( + &stack.cluster.mgmt_endpoints, + directory.join("winner.json"), + usize::MAX, + false, + ) + .await; + case::success( + &format!("http://{}", winner.address), + &case.path, + &json!({"requirements":[],"updates":[{"action":"set-properties","updates":{"winner":"selected"}}]}), + ) + .await; + drop(winner); + drop(loser); + assert!(request.await.unwrap().is_err()); + let recovery = TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let endpoint = format!("http://{}", recovery.address); + let first = case::post(&endpoint, &case.path, &case.identity, &case.body) + .await + .unwrap(); + assert_eq!(first.status(), 409); + let bytes = first.bytes().await.unwrap(); + let replay = case::post(&endpoint, &case.path, &case.identity, &case.body) + .await + .unwrap(); + assert_eq!(replay.status(), 409); + assert_eq!(replay.bytes().await.unwrap(), bytes); + let changed = case::post(&endpoint, &case.path, &case.identity, &format!("{} ", case.body)) + .await + .unwrap(); + assert_eq!(changed.status(), 409); + assert_eq!( + journal + .load(context, case.identity.parse().unwrap()) + .await + .unwrap() + .unwrap() + .phase, + TableCommitPhase::Rejected + ); + let parent = NamespaceRepository::new(store.clone()) + .load( + context, + &NamespaceIdentifier::new(vec!["analytics".into()]).unwrap(), + ) + .await + .unwrap() + .unwrap(); + let selected = TableRepository::new(store) + .select(context, parent.namespace, &case.name) + .await + .unwrap() + .unwrap(); + assert_eq!(selected.head.generation, 2); + assert_ne!(selected.head.metadata_location, candidate.metadata_location); + assert!(files + .load(context, &candidate.metadata_location) + .await + .unwrap() + .is_some()); +} diff --git a/app/crowdb-access-server/tests/common/iceberg_creation.rs b/app/crowdb-access-server/tests/common/iceberg_creation.rs new file mode 100644 index 000000000..9d7d60a86 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_creation.rs @@ -0,0 +1,137 @@ +use std::sync::{ + atomic::{AtomicBool, Ordering}, + Arc, +}; + +use async_trait::async_trait; +use crowdb_access_iceberg::catalog::{ + CasOutcome, CatalogContext, CatalogStore, RoutedCatalogStore, StoreError, StoredValue, +}; +use crowdb_access_iceberg::key::{IcebergKey, NamespaceId, OperationId}; +use crowdb_access_iceberg::namespace::{ + ChildScan, NamespaceCreateRequest, NamespaceCreator, NamespaceIdentifier, NamespaceJournal, + NamespaceMappingState, NamespacePhase, NamespaceProperties, NamespaceRepository, NamespaceStore, +}; +use crowdb_access_iceberg::operation::RequestIdentity; +use crowdb_access_iceberg::record::StorageRecord; +use crowdb_chunk_kv_client::MultiScanPage; +use crowdb_protocol::chunk_kv::ClientRequestId; + +use super::common::now_ms; + +pub struct TestCreateRecovery { + request: NamespaceCreateRequest, + namespace: NamespaceId, +} + +pub async fn prepare(store: Arc, context: CatalogContext) -> TestCreateRecovery { + let mut request = NamespaceCreateRequest { + context, + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "writer".into(), + identifier: NamespaceIdentifier::new(vec!["native-created".into()]).unwrap(), + properties: NamespaceProperties::default(), + }; + assert_eq!( + NamespaceCreator::new(store.clone()) + .create(&request) + .await + .unwrap() + .status, + 200 + ); + request.identity.operation = OperationId::random(); + request.identifier = NamespaceIdentifier::new(vec!["native-created".into(), "child".into()]).unwrap(); + let fault = Arc::new(TestCreateFaultStore { + inner: store.clone(), + armed: AtomicBool::new(true), + }); + assert!(NamespaceCreator::new(fault).create(&request).await.is_err()); + let operation = NamespaceJournal::new(store) + .load(context, request.identity.operation) + .await + .unwrap() + .unwrap(); + assert_eq!(operation.phase, NamespacePhase::Publishing); + TestCreateRecovery { + request, + namespace: operation.namespace, + } +} + +pub async fn verify(store: Arc, recovery: &TestCreateRecovery) { + let creator = NamespaceCreator::new(store.clone()); + let outcome = creator.create(&recovery.request).await.unwrap(); + assert_eq!(outcome.status, 200); + assert_eq!(creator.create(&recovery.request).await.unwrap(), outcome); + let reader = NamespaceRepository::new(store); + let selected = reader + .load(recovery.request.context, &recovery.request.identifier) + .await + .unwrap() + .unwrap(); + assert_eq!(selected.namespace, recovery.namespace); + assert_eq!(selected.pending_operation, None); + let parent = reader + .load( + recovery.request.context, + &recovery.request.identifier.parent().unwrap(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(parent.pending_operation, None); + assert_eq!(parent.admission_fence, 1); + assert_eq!(parent.property_revision, 1); +} + +struct TestCreateFaultStore { + inner: Arc, + armed: AtomicBool, +} + +#[async_trait] +impl CatalogStore for TestCreateFaultStore { + async fn get(&self, key: &[u8]) -> Result, StoreError> { + self.inner.get(key).await + } + + async fn compare_exchange( + &self, + key: &[u8], + expected: Option<&[u8]>, + value: &[u8], + identity: ClientRequestId, + ) -> Result { + let record = StorageRecord::decode(&IcebergKey::decode(key)?, value)?; + let intercept = expected.is_some() + && matches!(record, StorageRecord::NamespaceMapping(mapping) if mapping.state == NamespaceMappingState::Published); + let result = self + .inner + .compare_exchange(key, expected, value, identity) + .await?; + if intercept && self.armed.swap(false, Ordering::SeqCst) { + return Err(StoreError::Response); + } + Ok(result) + } +} + +#[async_trait] +impl NamespaceStore for TestCreateFaultStore { + async fn scan_children(&self, request: ChildScan) -> Result { + self.inner.scan_children(request).await + } + + async fn delete_mapping( + &self, + key: &[u8], + expected: &[u8], + identity: ClientRequestId, + ) -> Result { + self.inner.delete_mapping(key, expected, identity).await + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_drop.rs b/app/crowdb-access-server/tests/common/iceberg_drop.rs new file mode 100644 index 000000000..dcdc36cce --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_drop.rs @@ -0,0 +1,166 @@ +use std::sync::{ + atomic::{AtomicBool, Ordering}, + Arc, +}; + +use async_trait::async_trait; +use crowdb_access_iceberg::catalog::{ + CasOutcome, CatalogContext, CatalogStore, RoutedCatalogStore, StoreError, StoredValue, +}; +use crowdb_access_iceberg::key::{IcebergKey, NamespaceId, OperationId}; +use crowdb_access_iceberg::namespace::{ + authority_key, ChildScan, NamespaceCreateRequest, NamespaceCreator, NamespaceDropRequest, + NamespaceDropper, NamespaceIdentifier, NamespaceJournal, NamespaceLifecycle, NamespacePhase, + NamespaceProperties, NamespaceRepository, NamespaceStore, +}; +use crowdb_access_iceberg::operation::RequestIdentity; +use crowdb_access_iceberg::record::StorageRecord; +use crowdb_chunk_kv_client::MultiScanPage; +use crowdb_protocol::chunk_kv::ClientRequestId; + +use super::common::now_ms; + +pub struct TestDropRecovery { + request: NamespaceDropRequest, + namespace: NamespaceId, +} + +pub async fn prepare(store: Arc, context: CatalogContext) -> TestDropRecovery { + let request = NamespaceDropRequest { + context, + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "writer".into(), + identifier: NamespaceIdentifier::new(vec!["drop-recovery".into()]).unwrap(), + }; + recreate(store.clone(), &request).await; + let fault = Arc::new(TestDropFaultStore { + inner: store.clone(), + armed: AtomicBool::new(true), + }); + assert!(NamespaceDropper::new(fault) + .drop_namespace(&request) + .await + .is_err()); + let operation = NamespaceJournal::new(store.clone()) + .load(context, request.identity.operation) + .await + .unwrap() + .unwrap(); + assert_eq!(operation.phase, NamespacePhase::Tombstoning); + assert!(!NamespaceRepository::new(store) + .exists(context, &request.identifier) + .await + .unwrap()); + TestDropRecovery { + request, + namespace: operation.namespace, + } +} + +pub async fn verify(store: Arc, recovery: &TestDropRecovery) { + let dropper = NamespaceDropper::new(store.clone()); + let outcome = dropper.drop_namespace(&recovery.request).await.unwrap().unwrap(); + assert_eq!(outcome.status, 204); + assert_eq!( + dropper.drop_namespace(&recovery.request).await.unwrap(), + Some(outcome.clone()) + ); + let key = authority_key(recovery.request.context.catalog, recovery.namespace); + let value = store.get(&key.encode().unwrap()).await.unwrap().unwrap(); + let StorageRecord::NamespaceAuthority(authority) = StorageRecord::decode(&key, &value.bytes).unwrap() + else { + panic!("expected retained namespace tombstone"); + }; + assert_eq!(authority.lifecycle, NamespaceLifecycle::Tombstone); + recreate(store.clone(), &recovery.request).await; + let reader = NamespaceRepository::new(store); + let replacement = reader + .load(recovery.request.context, &recovery.request.identifier) + .await + .unwrap() + .unwrap(); + assert_ne!(replacement.namespace, recovery.namespace); + assert_eq!( + dropper.drop_namespace(&recovery.request).await.unwrap(), + Some(outcome) + ); + assert_eq!( + reader + .load(recovery.request.context, &recovery.request.identifier) + .await + .unwrap(), + Some(replacement) + ); +} + +async fn recreate(store: Arc, request: &NamespaceDropRequest) { + let creation = NamespaceCreateRequest { + context: request.context, + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: request.principal.clone(), + identifier: request.identifier.clone(), + properties: NamespaceProperties::default(), + }; + assert_eq!( + NamespaceCreator::new(store) + .create(&creation) + .await + .unwrap() + .status, + 200 + ); +} + +struct TestDropFaultStore { + inner: Arc, + armed: AtomicBool, +} + +#[async_trait] +impl CatalogStore for TestDropFaultStore { + async fn get(&self, key: &[u8]) -> Result, StoreError> { + self.inner.get(key).await + } + + async fn compare_exchange( + &self, + key: &[u8], + expected: Option<&[u8]>, + value: &[u8], + identity: ClientRequestId, + ) -> Result { + let record = StorageRecord::decode(&IcebergKey::decode(key)?, value)?; + let intercept = matches!(record, StorageRecord::NamespaceAuthority(authority) + if authority.lifecycle == NamespaceLifecycle::Tombstone); + let result = self + .inner + .compare_exchange(key, expected, value, identity) + .await?; + if intercept && self.armed.swap(false, Ordering::SeqCst) { + return Err(StoreError::Response); + } + Ok(result) + } +} + +#[async_trait] +impl NamespaceStore for TestDropFaultStore { + async fn scan_children(&self, request: ChildScan) -> Result { + self.inner.scan_children(request).await + } + + async fn delete_mapping( + &self, + key: &[u8], + expected: &[u8], + identity: ClientRequestId, + ) -> Result { + self.inner.delete_mapping(key, expected, identity).await + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_fault.rs b/app/crowdb-access-server/tests/common/iceberg_fault.rs new file mode 100644 index 000000000..1c597fe73 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_fault.rs @@ -0,0 +1,70 @@ +use std::sync::atomic::{AtomicU8, Ordering}; +use std::sync::Arc; + +use async_trait::async_trait; +use crowdb_access_iceberg::catalog::{ + CasOutcome, CatalogStore, RootState, RoutedCatalogStore, StoreError, StoredValue, +}; +use crowdb_access_iceberg::key::IcebergKey; +use crowdb_access_iceberg::namespace::{ChildScan, NamespaceStore}; +use crowdb_access_iceberg::record::StorageRecord; +use crowdb_chunk_kv_client::MultiScanPage; +use crowdb_protocol::chunk_kv::ClientRequestId; + +pub struct TestFaultStore { + pub inner: Arc, + pub mode: AtomicU8, +} + +#[async_trait] +impl NamespaceStore for TestFaultStore { + async fn scan_children(&self, request: ChildScan) -> Result { + self.inner.scan_children(request).await + } + + async fn delete_mapping( + &self, + key: &[u8], + expected: &[u8], + identity: ClientRequestId, + ) -> Result { + self.inner.delete_mapping(key, expected, identity).await + } +} + +#[async_trait] +impl CatalogStore for TestFaultStore { + async fn get(&self, key: &[u8]) -> Result, StoreError> { + self.inner.get(key).await + } + + async fn compare_exchange( + &self, + key: &[u8], + expected: Option<&[u8]>, + value: &[u8], + identity: ClientRequestId, + ) -> Result { + let record = StorageRecord::decode(&IcebergKey::decode(key)?, value)?; + let intercept = matches!(&record, StorageRecord::Active(root) if root.state == RootState::Fencing) + || (self.mode.load(Ordering::SeqCst) == 3 + && expected.is_some() + && matches!(&record, StorageRecord::NamespaceAuthority(authority) if authority.pending_operation.is_some())); + let mode = if intercept { + self.mode.swap(0, Ordering::SeqCst) + } else { + 0 + }; + if mode == 1 { + return Err(StoreError::Response); + } + let result = self + .inner + .compare_exchange(key, expected, value, identity) + .await?; + if mode == 2 || mode == 3 { + return Err(StoreError::Response); + } + Ok(result) + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_file_blocks.rs b/app/crowdb-access-server/tests/common/iceberg_file_blocks.rs new file mode 100644 index 000000000..c9f893c16 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_file_blocks.rs @@ -0,0 +1,69 @@ +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + +use async_trait::async_trait; +use crowdb_access_iceberg::file::{ + ChunkRoot, ContentFormat, FileBlockStore, FileContent, FileIdentity, FileIoError, FileKind, FileRecord, + TableLocation, +}; +use crowdb_access_iceberg::key::{CatalogId, FileId, TableId}; +use crowdb_protocol::common::ChunkId; +use sha2::{Digest, Sha256}; + +#[derive(Default)] +pub struct TestFileBlocks { + pub bytes: Vec, + pub reads: AtomicUsize, + pub fail: AtomicBool, + pub pause: AtomicBool, + pub entered: tokio::sync::Notify, + pub release: tokio::sync::Notify, +} + +impl TestFileBlocks { + pub fn record(&self) -> FileRecord { + let digest = Sha256::digest(&self.bytes).into(); + FileRecord { + file: FileId::random(), + location: TableLocation { + catalog: CatalogId::random(), + table: TableId::random(), + } + .file("data.parquet") + .unwrap(), + kind: FileKind::Data, + format: ContentFormat::Parquet, + length: self.bytes.len() as u64, + digest, + hint: None, + content: FileContent::Chunks { + root: (!self.bytes.is_empty()).then_some(ChunkRoot { + chunk: ChunkId { high: 1, low: 1 }, + offset: 0, + physical_length: self.bytes.len() as u64 + 64, + logical_offset: 0, + logical_length: self.bytes.len() as u64, + height: 0, + digest, + }), + }, + } + } +} + +#[async_trait] +impl FileBlockStore for TestFileBlocks { + async fn put(&self, _owner: FileIdentity, _height: u8, _bytes: &[u8]) -> Result { + Err(FileIoError::Bounds) + } + async fn read(&self, _root: &ChunkRoot) -> Result, FileIoError> { + self.reads.fetch_add(1, Ordering::SeqCst); + if self.pause.load(Ordering::SeqCst) { + self.entered.notify_one(); + self.release.notified().await; + } + if self.fail.load(Ordering::SeqCst) { + return Err(FileIoError::Bounds); + } + Ok(self.bytes.clone()) + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_file_lifecycle.rs b/app/crowdb-access-server/tests/common/iceberg_file_lifecycle.rs new file mode 100644 index 000000000..3dab531b1 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_file_lifecycle.rs @@ -0,0 +1,304 @@ +use std::time::Duration; + +use crowdb_access_iceberg::{ + catalog::{CatalogContext, CatalogError, CatalogRepository, ClearBounds, ManagementPrivilege, RootState}, + file::{FileCredentials, FileGrantIssuer, TableLocation}, + key::OperationId, + operation::{ManagementAction, ManagementRequest, RequestIdentity}, + wire::BearerAuthenticator, +}; +use reqwest::{Client, Method, Response}; +use serde_json::{json, Value}; + +use super::{common::now_ms, path, process::TestIcebergProcess, setup_with_bounds, TestFileClient}; + +const NAME: &str = "/v1/namespaces/analytics/tables/grants"; +const RENAMED: &str = "/v1/namespaces/analytics/tables/renamed"; + +pub async fn run() { + let bounds = ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }; + let (stack, first, bootstrap, _) = setup_with_bounds(bounds).await; + let second = TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let endpoint = format!("http://{}", first.address); + let other = format!("http://{}", second.address); + let context = bootstrap.credentials.grant().context; + let authentication = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let issuer = FileGrantIssuer::new(authentication.namespace_token_key(), 900_000).unwrap(); + let created = create(&endpoint).await; + let table = location(&created); + let first_grant = refresh(&endpoint, NAME, "w", &issuer, context).await; + let next_grant = refresh(&other, NAME, "w", &issuer, context).await; + assert_ne!(first_grant.access_key_id(), next_grant.access_key_id()); + let writer = TestFileClient { + client: Client::new(), + credentials: next_grant, + address: second.address, + }; + let object = path(table, "metadata/delegated.json"); + let bytes = br#"{"delegated":true}"#; + assert_eq!( + writer.send(Method::PUT, &object, "", bytes, false).await.status(), + 200 + ); + let previous = TestFileClient { + client: Client::new(), + credentials: first_grant, + address: first.address, + }; + assert_eq!( + previous + .send(Method::GET, &object, "", b"", false) + .await + .bytes() + .await + .unwrap() + .as_ref(), + bytes + ); + let reader = TestFileClient { + client: Client::new(), + credentials: refresh(&other, NAME, "r", &issuer, context).await, + address: second.address, + }; + assert_eq!( + reader.send(Method::GET, &object, "", b"", false).await.status(), + 200 + ); + assert_eq!( + reader + .send( + Method::PUT, + &path(table, "metadata/forbidden.json"), + "", + bytes, + false + ) + .await + .status(), + 403 + ); + expire(&writer, &issuer, &object).await; + rename_drop(&endpoint, &other, &writer, &issuer, table, &object).await; + let repository = CatalogRepository::new(stack.store().await, bounds).unwrap(); + clear(&repository, &writer, &previous, &other, &object).await; +} + +async fn rename_drop( + endpoint: &str, + other: &str, + writer: &TestFileClient, + issuer: &FileGrantIssuer, + table: TableLocation, + object: &str, +) { + let context = writer.credentials.grant().context; + let bytes = br#"{"delegated":true}"#; + value( + rest( + endpoint, + Method::POST, + "/v1/tables/rename", + "w", + Some(json!({ + "source":{"namespace":["analytics"],"name":"grants"}, + "destination":{"namespace":["analytics"],"name":"renamed"} + })), + ) + .await, + 204, + ) + .await; + assert_eq!( + rest(other, Method::GET, &format!("{NAME}/credentials"), "w", None) + .await + .status(), + 404 + ); + let renamed = refresh(other, RENAMED, "w", issuer, context).await; + assert_eq!(renamed.grant().table, writer.credentials.grant().table); + assert_eq!( + writer.send(Method::GET, object, "", b"", false).await.status(), + 200 + ); + value(rest(endpoint, Method::DELETE, RENAMED, "w", None).await, 204).await; + assert_eq!( + rest(other, Method::GET, &format!("{RENAMED}/credentials"), "w", None) + .await + .status(), + 404 + ); + assert_eq!( + writer.send(Method::GET, object, "", b"", false).await.status(), + 200 + ); + let recreated = create(other).await; + let replacement = location(&recreated); + assert_ne!(replacement.table, table.table); + assert_eq!( + writer + .send( + Method::PUT, + &path(replacement, "metadata/foreign.json"), + "", + bytes, + false + ) + .await + .status(), + 403 + ); + let fresh = refresh(other, NAME, "w", issuer, context).await; + let fresh = TestFileClient { + client: Client::new(), + credentials: fresh, + address: writer.address, + }; + assert_eq!( + fresh.send(Method::GET, object, "", b"", false).await.status(), + 403 + ); +} + +async fn clear( + repository: &CatalogRepository, + writer: &TestFileClient, + previous: &TestFileClient, + other: &str, + object: &str, +) { + let context = writer.credentials.grant().context; + let request = ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "clearer".into(), + action: ManagementAction::Clear, + expected_epoch: context.activation_epoch, + display_name: "after-clear".into(), + confirmation: Some(context.catalog), + capabilities: None, + }; + assert!(matches!( + repository + .execute(request, ManagementPrivilege::Clear, now_ms()) + .await, + Err(CatalogError::Busy) + )); + assert_ne!(repository.status().await.unwrap().0.state, RootState::Ready); + assert_eq!( + writer.send(Method::GET, object, "", b"", false).await.status(), + 503 + ); + assert_eq!( + previous.send(Method::GET, object, "", b"", false).await.status(), + 503 + ); + assert_eq!( + rest(other, Method::GET, &format!("{NAME}/credentials"), "w", None) + .await + .status(), + 503 + ); +} + +fn location(response: &Value) -> TableLocation { + format!( + "{}/", + response["metadata"]["location"] + .as_str() + .unwrap() + .trim_end_matches('/') + ) + .parse() + .unwrap() +} + +async fn expire(writer: &TestFileClient, issuer: &FileGrantIssuer, object: &str) { + let mut grant = writer.credentials.grant().clone(); + grant.nonce = OperationId::random(); + grant.issued_ms = now_ms(); + grant.expires_ms = grant.issued_ms + 2_000; + let expires = grant.expires_ms; + let short = TestFileClient { + client: Client::new(), + credentials: issuer.issue(grant).unwrap(), + address: writer.address, + }; + assert_eq!( + short.send(Method::GET, object, "", b"", false).await.status(), + 200 + ); + tokio::time::sleep(Duration::from_millis(expires.saturating_sub(now_ms()))).await; + assert!(now_ms() >= expires); + assert_eq!( + short.send(Method::GET, object, "", b"", false).await.status(), + 403 + ); + assert_eq!( + writer.send(Method::GET, object, "", b"", false).await.status(), + 200 + ); +} + +async fn refresh( + endpoint: &str, + name: &str, + role: &str, + issuer: &FileGrantIssuer, + context: CatalogContext, +) -> FileCredentials { + let response = value( + rest(endpoint, Method::GET, &format!("{name}/credentials"), role, None).await, + 200, + ) + .await; + let config = &response["storage-credentials"][0]["config"]; + let credentials = issuer + .verify( + config["s3.access-key-id"].as_str().unwrap(), + config["s3.session-token"].as_str().unwrap(), + context, + now_ms(), + ) + .unwrap(); + assert_eq!( + credentials.secret_access_key(), + config["s3.secret-access-key"].as_str().unwrap() + ); + credentials +} + +async fn create(endpoint: &str) -> Value { + value(rest(endpoint, Method::POST, "/v1/namespaces/analytics/tables", "w", Some(json!({ + "name":"grants","schema":{"type":"struct","schema-id":0,"fields":[{"id":1,"name":"id","type":"long","required":true}]} + }))).await, 200).await +} + +async fn rest(endpoint: &str, method: Method, path: &str, role: &str, body: Option) -> Response { + let mut request = Client::new() + .request(method, format!("{endpoint}{path}")) + .bearer_auth(role.repeat(32)); + if let Some(body) = body { + request = request + .header("content-type", "application/json") + .body(body.to_string()); + } + request.send().await.unwrap() +} + +async fn value(response: Response, expected: u16) -> Value { + let status = response.status(); + let text = response.text().await.unwrap(); + assert_eq!(status, expected, "{text}"); + if expected == 204 { + Value::Null + } else { + serde_json::from_str(&text).unwrap() + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_file_recovery.rs b/app/crowdb-access-server/tests/common/iceberg_file_recovery.rs new file mode 100644 index 000000000..b73c94e07 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_file_recovery.rs @@ -0,0 +1,246 @@ +use std::collections::BTreeSet; +use std::path::Path; +use std::time::Duration; + +use crowdb_access_iceberg::{ + catalog::ClearBounds, + file::{ + FileKind, FileRepository, MultipartAdmission, MultipartPhase, MultipartRepository, MultipartSession, + TableLocation, + }, +}; +use reqwest::Method; +use serde_json::Value; + +use super::{ + child::TestCommitChild, common::TestIcebergStack, path, process::TestIcebergProcess, setup_with_bounds, + TestFileClient, +}; + +pub async fn run() { + let (mut stack, process, mut client, table) = setup_with_bounds(ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }) + .await; + let directory = stack + .cluster + .runtime_mut() + .service_dir("iceberg", "file-faults") + .unwrap(); + drop(process); + for multipart in [false, true] { + let baseline = prepare(&stack, &mut client, table, multipart, "baseline").await; + let child = TestCommitChild::start( + &stack.cluster.mgmt_endpoints, + directory.join(format!("baseline-{multipart}.json")), + usize::MAX, + false, + ) + .await; + client.address = child.address; + baseline.execute(&client).await; + let marker: Value = serde_json::from_slice(&std::fs::read(&child.marker).unwrap()).unwrap(); + let count = usize::try_from(marker["index"].as_u64().unwrap()).unwrap(); + drop(child); + let mut labels = BTreeSet::new(); + for offset in 1..=count { + for after in [false, true] { + let label = interrupt(&stack, &directory, &mut client, table, multipart, offset, after).await; + labels.insert(label); + } + } + assert!(labels.contains("file-record")); + assert!(labels.contains("file-mapping")); + if multipart { + assert!(labels.contains("multipart-Completing")); + assert!(labels.contains("multipart-Publishing")); + assert!(labels.contains("multipart-Published")); + } else { + assert!(labels.contains("file-block")); + } + } +} + +async fn interrupt( + stack: &TestIcebergStack, + directory: &Path, + client: &mut TestFileClient, + table: TableLocation, + multipart: bool, + offset: usize, + after: bool, +) -> String { + let case = prepare(stack, client, table, multipart, &format!("{offset}-{after}")).await; + let mut child = TestCommitChild::start( + &stack.cluster.mgmt_endpoints, + directory.join(format!("{multipart}-{offset}-{after}.json")), + offset, + after, + ) + .await; + client.address = child.address; + let request = case.request(client); + let mut request = tokio::spawn(async move { request.send().await?.bytes().await }); + let boundary = tokio::select! { + boundary = child.paused() => boundary, + result = &mut request => panic!("file boundary {multipart}/{offset}/{after} returned early: {result:?}"), + }; + let label = boundary["label"].as_str().unwrap().to_owned(); + println!("kill multipart={multipart} boundary={offset} after={after}: {label}"); + drop(child); + assert!( + request.await.unwrap().is_err(), + "interruption must lose the response" + ); + let recovery = TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + client.address = recovery.address; + let first = case.execute(client).await; + let files = FileRepository::new(stack.store().await); + let context = client.credentials.grant().context; + let record = files + .load(context, &table.file(&case.key).unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(record.kind, FileKind::Unbound); + assert_eq!(case.execute(client).await, first); + assert_eq!( + files.load(context, &record.location).await.unwrap().unwrap(), + record + ); + let get = client.send(Method::GET, &case.path, "", b"", false).await; + assert_eq!(get.status(), 200); + assert_eq!(get.bytes().await.unwrap().as_ref(), case.bytes); + let mut changed = case.bytes.clone(); + changed[4] ^= 1; + let conflict = client.send(Method::PUT, &case.path, "", &changed, false).await; + assert_eq!(conflict.status(), 409); + if let Some(upload) = &case.upload { + let session = MultipartRepository::new(stack.store().await) + .load(context, upload.parse().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(session.phase, MultipartPhase::Published); + assert_eq!(session.published, Some(record.file)); + released_by_recovery(stack, &session).await; + } + label +} + +async fn released_by_recovery(stack: &TestIcebergStack, published: &MultipartSession) { + let store = stack.store().await; + let sessions = MultipartRepository::new(store.clone()); + let admission = MultipartAdmission::new(store); + tokio::time::timeout(Duration::from_secs(5), async { + loop { + let current = sessions + .load(published.context, published.upload) + .await + .unwrap() + .unwrap(); + assert_eq!(current.phase, MultipartPhase::Published); + assert_eq!(current.published, published.published); + let policy = admission.load(published.context).await.unwrap().unwrap(); + assert!(policy.sessions <= 1); + assert!(policy.reserved_bytes <= published.limits.max_staged_bytes); + if current.credit.unwrap().released && policy.pending.is_none() { + assert_eq!((policy.sessions, policy.reserved_bytes), (0, 0)); + break; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .expect("background recovery must settle terminal credits without another Complete request"); +} + +struct TestFileCase { + key: String, + path: String, + bytes: Vec, + upload: Option, + complete: String, +} + +impl TestFileCase { + fn request(&self, client: &TestFileClient) -> reqwest::RequestBuilder { + match &self.upload { + Some(upload) => client.request( + Method::POST, + &self.path, + &format!("uploadId={upload}"), + self.complete.as_bytes(), + false, + None, + ), + None => client.request(Method::PUT, &self.path, "", &self.bytes, false, None), + } + } + + async fn execute(&self, client: &TestFileClient) -> String { + let response = self.request(client).send().await.unwrap(); + let status = response.status(); + let body = response.text().await.unwrap(); + assert_eq!(status, 200, "{body}"); + if self.upload.is_some() { + assert!(body.ends_with(""), "{body}"); + assert!(!body.contains(""), "{body}"); + } + body.trim().to_owned() + } +} + +async fn prepare( + stack: &TestIcebergStack, + client: &mut TestFileClient, + table: TableLocation, + multipart: bool, + name: &str, +) -> TestFileCase { + let key = format!("objects/{multipart}-{name}.parquet"); + let mut bytes = b"PAR1".to_vec(); + bytes.resize(65540, b'x'); + bytes.extend_from_slice(b"foot\x04\0\0\0PAR1"); + let mut case = TestFileCase { + path: path(table, &key), + key, + bytes, + upload: None, + complete: String::new(), + }; + if multipart { + let setup = TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + client.address = setup.address; + let create = client + .send(Method::POST, &case.path, "uploads=", b"", false) + .await; + let status = create.status(); + let body = create.text().await.unwrap(); + assert_eq!(status, 200, "{body}"); + let upload = body + .split_once("") + .unwrap() + .1 + .split_once("") + .unwrap() + .0 + .to_owned(); + let part = client + .send( + Method::PUT, + &case.path, + &format!("partNumber=1&uploadId={upload}"), + &case.bytes, + false, + ) + .await; + assert_eq!(part.status(), 200, "{}", part.text().await.unwrap()); + let etag = part.headers()["etag"].to_str().unwrap(); + case.complete = format!("{etag}1"); + case.upload = Some(upload); + } + case +} diff --git a/app/crowdb-access-server/tests/common/iceberg_file_worker.rs b/app/crowdb-access-server/tests/common/iceberg_file_worker.rs new file mode 100644 index 000000000..ffdd4b0be --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_file_worker.rs @@ -0,0 +1,129 @@ +use std::net::TcpListener; +use std::process::{Child, Command, Stdio}; +use std::time::Duration; + +use crowdb_access_iceberg::catalog::CatalogContext; +use crowdb_access_iceberg::file::{ + FileIdentity, FileTree, MultipartAdmission, MultipartAdmissionLimits, MultipartLimits, MultipartPart, + MultipartPhase, MultipartRepository, MultipartSession, TableLocation, +}; +use crowdb_access_iceberg::key::{FileId, OperationId}; +use sha2::{Digest, Sha256}; + +use crate::common::TestIcebergStack; + +pub async fn verify(stack: &TestIcebergStack, context: CatalogContext, table: TableLocation) { + let store = stack.store().await; + let initial = MultipartSession { + context, + upload: OperationId::random(), + owner: FileIdentity { + table, + file: FileId::random(), + }, + location: table.file("multipart/expired.bin").unwrap(), + principal: [1; 32], + revision: 1, + created_ms: 1, + expires_ms: 1001, + limits: MultipartLimits { + max_parts: 1, + max_part_bytes: 100, + max_file_bytes: 100, + max_staged_bytes: 100, + ttl_ms: 1000, + }, + phase: MultipartPhase::Open, + part_count: 0, + staged_bytes: 0, + completion: None, + published: None, + pending: None, + credit: None, + }; + let admission = MultipartAdmission::new(store.clone()); + let policy = admission + .initialize( + context, + MultipartAdmissionLimits { + max_sessions: 2, + max_reserved_bytes: 200, + }, + ) + .await + .unwrap(); + assert!(admission.reserve(&policy, &initial, 1).await.unwrap()); + let repository = MultipartRepository::new(store); + let admitted = repository.load(context, initial.upload).await.unwrap().unwrap(); + let part = MultipartPart { + upload: initial.upload, + number: 1, + revision: 1, + modified_ms: 2, + owner: FileIdentity { + table, + file: FileId::random(), + }, + tree: FileTree { + root: None, + length: 0, + digest: Sha256::digest([]).into(), + }, + }; + assert!(repository.reserve_part(&admitted, &part, 2).await.unwrap()); + let mut worker = TestWorker::start(&stack.cluster.mgmt_endpoints); + tokio::time::timeout(Duration::from_secs(20), async { + loop { + assert!(worker.0.try_wait().unwrap().is_none(), "Iceberg worker exited"); + let current = repository.load(context, initial.upload).await.unwrap().unwrap(); + if current.phase == MultipartPhase::Aborted + && current.credit.is_some_and(|credit| credit.released) + { + assert!(current.pending.is_none()); + assert_eq!((current.part_count, current.staged_bytes), (1, 0)); + assert_eq!(repository.part(¤t, 1).await.unwrap().as_ref(), Some(&part)); + let policy = admission.load(context).await.unwrap().unwrap(); + if policy.pending.is_some() { + tokio::time::sleep(Duration::from_millis(20)).await; + continue; + } + assert_eq!((policy.sessions, policy.reserved_bytes), (0, 0)); + break; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .unwrap(); +} + +struct TestWorker(Child); + +impl TestWorker { + fn start(seeds: &[String]) -> Self { + let reservation = TcpListener::bind("127.0.0.1:0").unwrap(); + let address = reservation.local_addr().unwrap(); + drop(reservation); + Self( + Command::new(env!("CARGO_BIN_EXE_crowdb-iceberg")) + .env("CROWDB_MANAGEMENT_SEEDS", seeds.join(",")) + .env("CROWDB_ICEBERG_LISTEN", address.to_string()) + .env("CROWDB_ICEBERG_READ_TOKEN", "r".repeat(32)) + .env("CROWDB_ICEBERG_WRITE_TOKEN", "w".repeat(32)) + .env("CROWDB_ICEBERG_MANAGE_TOKEN", "m".repeat(32)) + .env("CROWDB_ICEBERG_CLEAR_TOKEN", "c".repeat(32)) + .arg("serve") + .stdout(Stdio::inherit()) + .stderr(Stdio::inherit()) + .spawn() + .unwrap(), + ) + } +} + +impl Drop for TestWorker { + fn drop(&mut self) { + let _ = self.0.kill(); + let _ = self.0.wait(); + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_gc_capacity.rs b/app/crowdb-access-server/tests/common/iceberg_gc_capacity.rs new file mode 100644 index 000000000..44e520675 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_gc_capacity.rs @@ -0,0 +1,79 @@ +use std::sync::{ + atomic::{AtomicBool, Ordering}, + Arc, +}; + +use async_trait::async_trait; +use crowdb_access_iceberg::{ + catalog::{CasOutcome, CatalogStore, RoutedCatalogStore, StoreError, StoredValue}, + gc::{GcScan, GcStore, GcSystemScan}, + key::{CatalogScope, IcebergKey}, +}; +use crowdb_chunk_kv_client::MultiScanPage; +use crowdb_protocol::chunk_kv::ClientRequestId; + +pub struct TestGcWorkspace { + inner: Arc, + denied: AtomicBool, +} + +impl TestGcWorkspace { + pub fn new(inner: Arc) -> Self { + Self { + inner, + denied: AtomicBool::new(false), + } + } + + pub fn deny(&self, denied: bool) { + self.denied.store(denied, Ordering::SeqCst); + } +} + +#[async_trait] +impl CatalogStore for TestGcWorkspace { + async fn get(&self, key: &[u8]) -> Result, StoreError> { + self.inner.get(key).await + } + + async fn compare_exchange( + &self, + key: &[u8], + expected: Option<&[u8]>, + value: &[u8], + identity: ClientRequestId, + ) -> Result { + if self.denied.load(Ordering::SeqCst) + && matches!( + IcebergKey::decode(key), + Ok(IcebergKey::Catalog { + scope: CatalogScope::GcClaim | CatalogScope::GcCandidate, + .. + }) + ) + { + return Err(StoreError::Budget); + } + self.inner.compare_exchange(key, expected, value, identity).await + } +} + +#[async_trait] +impl GcStore for TestGcWorkspace { + async fn scan_gc(&self, request: GcScan) -> Result { + self.inner.scan_gc(request).await + } + + async fn scan_gc_system(&self, request: GcSystemScan) -> Result { + self.inner.scan_gc_system(request).await + } + + async fn delete_gc_record( + &self, + key: &[u8], + expected: &[u8], + identity: ClientRequestId, + ) -> Result { + self.inner.delete_gc_record(key, expected, identity).await + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/pom.xml b/app/crowdb-access-server/tests/common/iceberg_java/pom.xml new file mode 100644 index 000000000..f1b5b3154 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/pom.xml @@ -0,0 +1,82 @@ + + 4.0.0 + db.crow.tests + iceberg-fileio-acceptance + 1.0 + + 17 + UTF-8 + 1.11.0 + TestIcebergFileIO + + + + org.apache.iceberg + iceberg-core + ${iceberg.version} + + + org.apache.iceberg + iceberg-aws + ${iceberg.version} + + + org.apache.iceberg + iceberg-aws-bundle + ${iceberg.version} + + + org.apache.iceberg + iceberg-data + ${iceberg.version} + + + org.apache.iceberg + iceberg-parquet + ${iceberg.version} + + + org.apache.iceberg + iceberg-orc + ${iceberg.version} + + + org.apache.parquet + parquet-column + 1.17.1 + + + org.apache.parquet + parquet-hadoop + 1.17.1 + + + org.apache.hadoop + hadoop-common + 3.4.1 + + + org.apache.hadoop + hadoop-mapreduce-client-core + 3.4.1 + + + + + + org.apache.maven.plugins + maven-compiler-plugin + 3.14.0 + + + org.codehaus.mojo + exec-maven-plugin + 3.5.0 + + ${exec.mainClass} + + + + + diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCatalogReads.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCatalogReads.java new file mode 100644 index 000000000..6abdb9ac7 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCatalogReads.java @@ -0,0 +1,80 @@ +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import org.apache.iceberg.Snapshot; +import org.apache.iceberg.Table; +import org.apache.iceberg.catalog.Namespace; +import org.apache.iceberg.catalog.TableIdentifier; +import org.apache.iceberg.io.FileIO; +import org.apache.iceberg.io.InputFile; +import org.apache.iceberg.io.OutputFile; +import org.apache.iceberg.rest.RESTCatalog; + +public final class TestIcebergCatalogReads { + public static void main(String[] args) throws Exception { + Namespace namespace = Namespace.of("analytics"); + TableIdentifier events = TableIdentifier.of(namespace, "events"); + for (String mode : List.of("all", "refs")) { + Map properties = new HashMap<>(); + properties.put("uri", args[0]); + properties.put("token", "r".repeat(32)); + properties.put("io-impl", TestNoFileIO.class.getName()); + properties.put("snapshot-loading-mode", mode); + properties.put("rest-page-size", "1"); + properties.put("rest-metrics-reporting-enabled", "false"); + try (RESTCatalog catalog = new RESTCatalog()) { + catalog.initialize("crowdb", properties); + List tables = catalog.listTables(namespace); + require(tables.size() == 3 && tables.contains(events), "paged list"); + require(catalog.tableExists(events), "exists"); + require(!catalog.tableExists(TableIdentifier.of(namespace, "absent")), "missing table"); + Table table = catalog.loadTable(events); + require(table.currentSnapshot().snapshotId() == 20, "current snapshot"); + require(table.refs().get("tag").snapshotId() == 30, "tag reference"); + Table unchanged = catalog.loadTable(events); + require(unchanged.currentSnapshot().snapshotId() == 20, "conditional load"); + int count = 0; + for (Snapshot snapshot : table.snapshots()) { + require(snapshot.schemaId() == 0, "snapshot schema"); + count++; + } + require(count == 3, "complete snapshots including REFS fallback"); + for (String name : List.of("a+b", "%2F")) { + require(catalog.loadTable(TableIdentifier.of(namespace, name)).schema().columns().size() == 1, + "single-decoded table name"); + } + } + } + System.out.println("Official RESTCatalog read acceptance passed"); + } + + private static void require(boolean valid, String operation) { + if (!valid) { + throw new IllegalStateException("RESTCatalog failed: " + operation); + } + } + + public static final class TestNoFileIO implements FileIO { + public TestNoFileIO() {} + + @Override + public InputFile newInputFile(String path) { + throw new AssertionError("Catalog read unexpectedly accessed a file: " + path); + } + + @Override + public OutputFile newOutputFile(String path) { + throw new AssertionError("Read-only acceptance attempted to create a file"); + } + + @Override + public void deleteFile(String path) { + throw new AssertionError("Read-only acceptance attempted to delete a file"); + } + + @Override + public Map properties() { + return Map.of(); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCatalogWrites.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCatalogWrites.java new file mode 100644 index 000000000..13dd1960e --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCatalogWrites.java @@ -0,0 +1,180 @@ +import java.util.Map; +import java.util.UUID; +import org.apache.iceberg.DataFile; +import org.apache.iceberg.BaseTable; +import org.apache.iceberg.FileScanTask; +import org.apache.iceberg.data.GenericRecord; +import org.apache.iceberg.data.Record; +import org.apache.iceberg.data.parquet.GenericParquetWriter; +import org.apache.iceberg.io.DataWriter; +import org.apache.iceberg.parquet.Parquet; +import org.apache.iceberg.Schema; +import org.apache.iceberg.Table; +import org.apache.iceberg.Transaction; +import org.apache.iceberg.aws.AwsClientProperties; +import org.apache.iceberg.aws.s3.VendedCredentialsProvider; +import org.apache.iceberg.catalog.Namespace; +import org.apache.iceberg.catalog.TableIdentifier; +import org.apache.iceberg.rest.RESTCatalog; +import org.apache.iceberg.types.Types; + +public final class TestIcebergCatalogWrites { + public static void main(String[] args) throws Exception { + Schema schema = new Schema(Types.NestedField.required(91, "id", Types.LongType.get())); + try (RESTCatalog catalog = new RESTCatalog()) { + catalog.initialize("crowdb", Map.of("uri", args[0], "token", "w".repeat(32), + "io-impl", "org.apache.iceberg.aws.s3.S3FileIO", "client.region", "us-east-1", + "rest-metrics-reporting-enabled", "false")); + if (args.length > 1 && args[1].equals("verify")) { + TestIcebergVersionRows.run(catalog, true); + TestIcebergPartitionStatistics.run(catalog, args[0], true); + for (String tableName : new String[] {"immediate", "staged"}) { + Table persisted = catalog.loadTable(TableIdentifier.of(Namespace.of("analytics"), tableName)); + credential(persisted); + verifyFiles(persisted, tableName.equals("immediate") ? 2 : 1); + } + require(!catalog.tableExists(TableIdentifier.of(Namespace.of("analytics"), "lifecycle")), + "dropped lifecycle table remains absent after restart"); + require(!catalog.namespaceExists(Namespace.of("lifecycle_destination")), + "empty destination namespace remains dropped after restart"); + System.out.println("Official RESTCatalog restart read and credential acceptance passed"); + return; + } + TableIdentifier name = TableIdentifier.of(Namespace.of("analytics"), "immediate"); + Table table = catalog.buildTable(name, schema).withProperty("format-version", "1").create(); + require(table.schema().findField("id").fieldId() == 1, "fresh create field IDs"); + if (args.length > 1) { + table.newAppend().appendFile(writeData(table)).commit(); + verifyFiles(table, 1); + } + table.updateProperties().set("owner", "sdk").commit(); + table.updateSchema().addColumn("message", Types.StringType.get()).commit(); + table.updateProperties().set("format-version", "3").commit(); + Table loaded = catalog.loadTable(name); + require(loaded.schema().findField("message") != null, "ordered schema commit"); + require(loaded.properties().get("owner").equals("sdk"), "property commit"); + credential(loaded); + if (args.length > 1) { + loaded.newAppend().appendFile(writeData(loaded)).commit(); + verifyFiles(loaded, 2); + } + TableIdentifier stagedName = TableIdentifier.of(Namespace.of("analytics"), "staged"); + Transaction first = catalog.buildTable(stagedName, schema).createTransaction(); + Transaction second = catalog.buildTable(stagedName, schema).createTransaction(); + require(!catalog.tableExists(stagedName), "draft invisibility"); + require(!first.table().location().equals(second.table().location()), "same-name draft isolation"); + String firstCredential = credential(first.table()); + String secondCredential = credential(second.table()); + require(!firstCredential.equals(secondCredential), "independent vended credentials"); + first.updateProperties().set("draft-owner", "sdk").commit(); + if (args.length > 1) { + first.newAppend().appendFile(writeData(first.table())).commit(); + } + first.commitTransaction(); + require(catalog.loadTable(stagedName).properties().get("draft-owner").equals("sdk"), "staged publication"); + credential(first.table()); + if (args.length > 1) { + verifyFiles(catalog.loadTable(stagedName), 1); + } + lifecycle(catalog, schema, args.length > 1); + if (args.length > 1) { + TestIcebergPartitionStatistics.run(catalog, args[0], false); + TestIcebergVersionRows.run(catalog, false); + } + System.out.println("Official RESTCatalog create, update, upgrade, stage, refresh, rename and drop acceptance passed"); + } + } + + private static void lifecycle(RESTCatalog catalog, Schema schema, boolean nativeFiles) throws Exception { + Namespace destination = Namespace.of("lifecycle_destination"); + catalog.createNamespace(destination); + TableIdentifier source = TableIdentifier.of(Namespace.of("analytics"), "lifecycle"); + TableIdentifier renamed = TableIdentifier.of(Namespace.of("analytics"), "lifecycle_renamed"); + TableIdentifier moved = TableIdentifier.of(destination, "moved"); + Table original = catalog.buildTable(source, schema).create(); + if (nativeFiles) { + original.newAppend().appendFile(writeData(original)).commit(); + } + String location = original.location(); + catalog.renameTable(source, renamed); + require(!catalog.tableExists(source), "old name is not an alias"); + require(catalog.loadTable(renamed).location().equals(location), "rename preserves file location"); + catalog.renameTable(renamed, moved); + require(!catalog.tableExists(renamed), "cross-namespace old name is not an alias"); + Table selected = catalog.loadTable(moved); + require(selected.location().equals(location), "cross-namespace stable identity"); + credential(selected); + selected.updateProperties().set("after-rename", "yes").commit(); + require(catalog.loadTable(moved).properties().get("after-rename").equals("yes"), "commit after rename"); + if (nativeFiles) { + verifyFiles(selected, 1); + } + String metadata = ((BaseTable) selected).operations().current().metadataFileLocation(); + require(catalog.dropTable(moved, false), "logical drop succeeds"); + require(!catalog.tableExists(moved), "dropped table is absent"); + require(catalog.dropNamespace(destination), "moved table does not leave live namespace children"); + if (nativeFiles) { + require(selected.io().newInputFile(metadata).exists(), "drop does not physically delete metadata"); + verifyFiles(selected, 1); + } + Table recreated = catalog.buildTable(source, schema).create(); + require(!recreated.location().equals(location), "recreated name uses a new table identity"); + String recreatedMetadata = ((BaseTable) recreated).operations().current().metadataFileLocation(); + if (nativeFiles) { + require(recreated.io().newInputFile(recreatedMetadata).exists(), "metadata exists before purge request"); + } + require(catalog.dropTable(source, true), "purge request logically drops the table"); + require(!catalog.tableExists(source), "purge request removes name visibility"); + if (nativeFiles) { + require(recreated.io().newInputFile(recreatedMetadata).exists(), "purge is a deferred proof task"); + } + require(!catalog.dropTable(source, false), "missing table follows SDK false contract"); + } + + private static DataFile writeData(Table table) throws Exception { + String path = table.location() + "/data/" + UUID.randomUUID() + ".parquet"; + DataWriter writer = Parquet.writeData(table.io().newOutputFile(path)) + .schema(table.schema()).withSpec(table.spec()) + .createWriterFunc(parquetSchema -> GenericParquetWriter.create(table.schema(), parquetSchema)) + .set("write.parquet.compression-codec", "zstd").build(); + try (writer) { + for (long row = 0; row < 10; row++) { + GenericRecord record = GenericRecord.create(table.schema()); + record.setField("id", row); + if (table.schema().findField("message") != null) { + record.setField("message", "row-" + row); + } + writer.write(record); + } + } + return writer.toDataFile(); + } + + private static void verifyFiles(Table table, int expectedFiles) throws Exception { + int count = 0; + try (var tasks = table.newScan().planFiles()) { + for (FileScanTask task : tasks) { + require(task.file().recordCount() == 10, "selected manifest rows"); + try (var input = table.io().newInputFile(task.file().location()).newStream()) { + require(new String(input.readNBytes(4), java.nio.charset.StandardCharsets.US_ASCII).equals("PAR1"), + "native Parquet read"); + } + count++; + } + } + require(count == expectedFiles, "selected data file count"); + } + + private static String credential(Table table) { + try (VendedCredentialsProvider provider = (VendedCredentialsProvider) + new AwsClientProperties(table.io().properties()).credentialsProvider(null, null, null)) { + return provider.resolveCredentials().accessKeyId(); + } + } + + private static void require(boolean valid, String message) { + if (!valid) { + throw new IllegalStateException(message); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCommitErrors.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCommitErrors.java new file mode 100644 index 000000000..4b2b8ea06 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCommitErrors.java @@ -0,0 +1,170 @@ +import java.util.Collections; +import java.util.List; +import java.util.Map; +import org.apache.iceberg.BaseTable; +import org.apache.iceberg.MetadataUpdate; +import org.apache.iceberg.ImmutableGenericPartitionStatisticsFile; +import org.apache.iceberg.SnapshotParser; +import org.apache.iceberg.Schema; +import org.apache.iceberg.UpdateRequirement; +import org.apache.iceberg.catalog.TableIdentifier; +import org.apache.iceberg.exceptions.AlreadyExistsException; +import org.apache.iceberg.exceptions.BadRequestException; +import org.apache.iceberg.exceptions.CommitFailedException; +import org.apache.iceberg.exceptions.NoSuchTableException; +import org.apache.iceberg.rest.ErrorHandlers; +import org.apache.iceberg.rest.ErrorHandler; +import org.apache.iceberg.rest.HTTPClient; +import org.apache.iceberg.rest.RESTCatalog; +import org.apache.iceberg.rest.auth.AuthSession; +import org.apache.iceberg.rest.requests.UpdateTableRequest; +import org.apache.iceberg.rest.responses.LoadTableResponse; +import org.apache.iceberg.rest.responses.ErrorResponse; +import org.apache.iceberg.types.Types; + +public final class TestIcebergCommitErrors { + private static final TableIdentifier NAME = TableIdentifier.of("analytics", "commit_errors"); + private static final String PATH = "v1/namespaces/analytics/tables/commit_errors"; + + public static void main(String[] args) throws Exception { + Map properties = Map.of("uri", args[0], "token", "w".repeat(32), + "io-impl", TestIcebergCatalogReads.TestNoFileIO.class.getName(), + "rest-metrics-reporting-enabled", "false"); + Schema schema = new Schema(Types.NestedField.required(1, "id", Types.LongType.get())); + try (RESTCatalog catalog = new RESTCatalog(); + HTTPClient root = HTTPClient.builder(properties).uri(args[0]) + .withHeaders(Map.of("Authorization", "Bearer " + "w".repeat(32))).build(); + HTTPClient client = root.withAuthSession(AuthSession.EMPTY)) { + catalog.initialize("crowdb", properties); + catalog.buildTable(NAME, schema).withProperty("format-version", "2").create(); + String initial = metadata(catalog); + expect(AlreadyExistsException.class, () -> catalog.buildTable(NAME, schema).create()); + require(initial.equals(metadata(catalog)), "duplicate create preserves head"); + String uuid = ((BaseTable) catalog.loadTable(NAME)).operations().current().uuid(); + rejected(catalog, client, new UpdateTableRequest( + List.of(new UpdateRequirement.AssertTableUUID("00000000-0000-0000-0000-000000000000")), + List.of(property("invalid"))), 409, "CommitFailedException", CommitFailedException.class); + rejected(catalog, client, new UpdateTableRequest(List.of(), + List.of(property("partial"), new MetadataUpdate.SetCurrentSchema(999))), + 400, "BadRequestException", BadRequestException.class); + rejected(catalog, client, new UpdateTableRequest(List.of(), + List.of(new MetadataUpdate.UpgradeFormatVersion(99))), + 400, "BadRequestException", BadRequestException.class); + rejected(catalog, client, UpdateTableRequest.create(TableIdentifier.of("analytics", "other"), + List.of(), List.of(property("wrong-path"))), + 400, "BadRequestException", BadRequestException.class); + unavailablePartitionStatistics(catalog, client); + counts(catalog, client, uuid); + int oldSchema = catalog.loadTable(NAME).schema().schemaId(); + catalog.loadTable(NAME).updateSchema().addColumn("message", Types.StringType.get()).commit(); + rejected(catalog, client, new UpdateTableRequest( + List.of(new UpdateRequirement.AssertCurrentSchemaID(oldSchema)), List.of(property("stale"))), + 409, "CommitFailedException", CommitFailedException.class); + require(catalog.dropTable(NAME, false), "drop succeeds"); + failure(client, new UpdateTableRequest(List.of(), List.of(property("dropped"))), + 404, "NoSuchTableException", NoSuchTableException.class); + catalog.buildTable(NAME, schema).create(); + rejected(catalog, client, new UpdateTableRequest( + List.of(new UpdateRequirement.AssertTableUUID(uuid)), List.of(property("old-identity"))), + 409, "CommitFailedException", CommitFailedException.class); + } + System.out.println("Official commit errors, atomic rejection and count boundaries passed"); + } + + private static void unavailablePartitionStatistics(RESTCatalog catalog, HTTPClient client) { + String location = catalog.loadTable(NAME).location(); + var snapshot = SnapshotParser.fromJson("{\"snapshot-id\":1,\"sequence-number\":1," + + "\"timestamp-ms\":" + System.currentTimeMillis() + + ",\"schema-id\":0,\"summary\":{\"operation\":\"append\"}," + + "\"manifest-list\":\"" + location + "/metadata/disabled.avro\"}"); + var statistics = ImmutableGenericPartitionStatisticsFile.builder().snapshotId(1) + .path(location + "/metadata/disabled.parquet").fileSizeInBytes(8).build(); + rejected(catalog, client, new UpdateTableRequest(List.of(), List.of(property("disabled"), + new MetadataUpdate.AddSnapshot(snapshot), new MetadataUpdate.SetPartitionStatistics(statistics))), + 400, "BadRequestException", BadRequestException.class); + } + + private static void counts(RESTCatalog catalog, HTTPClient client, String uuid) { + List requirements = Collections.nCopies(1000, + new UpdateRequirement.AssertCurrentSchemaID(catalog.loadTable(NAME).schema().schemaId())); + List updates = Collections.nCopies(1000, property("at-limit")); + client.post(PATH, new UpdateTableRequest(requirements, updates), LoadTableResponse.class, + Map.of(), ErrorHandlers.tableCommitHandler()); + require("at-limit".equals(catalog.loadTable(NAME).properties().get("boundary")), + "exact requirement and update count limits accepted"); + rejected(catalog, client, new UpdateTableRequest(Collections.nCopies(1001, + new UpdateRequirement.AssertCurrentSchemaID(catalog.loadTable(NAME).schema().schemaId())), + List.of(property("over-requirements"))), + 400, "BadRequestException", BadRequestException.class); + rejected(catalog, client, new UpdateTableRequest(Collections.nCopies(1000, + new UpdateRequirement.AssertTableUUID(uuid)), List.of(property("over-text"))), + 400, "BadRequestException", BadRequestException.class); + client.post(PATH, new UpdateTableRequest( + List.of(new UpdateRequirement.AssertRefSnapshotID("a".repeat(4096), null)), + List.of(property("at-text-limit"))), LoadTableResponse.class, + Map.of(), ErrorHandlers.tableCommitHandler()); + require("at-text-limit".equals(catalog.loadTable(NAME).properties().get("boundary")), + "exact requirement text budget accepted"); + rejected(catalog, client, new UpdateTableRequest( + List.of(new UpdateRequirement.AssertRefSnapshotID("a".repeat(4097), null)), + List.of(property("over-text-limit"))), 400, "BadRequestException", BadRequestException.class); + rejected(catalog, client, new UpdateTableRequest(List.of(), + Collections.nCopies(1001, property("over-updates"))), + 400, "BadRequestException", BadRequestException.class); + } + + private static MetadataUpdate property(String value) { + return new MetadataUpdate.SetProperties(Map.of("boundary", value)); + } + + private static String metadata(RESTCatalog catalog) { + return ((BaseTable) catalog.loadTable(NAME)).operations().current().metadataFileLocation(); + } + + private static void rejected(RESTCatalog catalog, HTTPClient client, UpdateTableRequest request, + int status, String type, Class exception) { + String before = metadata(catalog); + failure(client, request, status, type, exception); + require(before.equals(metadata(catalog)), "rejected commit preserves selected metadata"); + } + + private static void failure(HTTPClient client, UpdateTableRequest request, int status, + String type, Class exception) { + boolean[] received = {false}; + ErrorHandler official = (ErrorHandler) ErrorHandlers.tableCommitHandler(); + expect(exception, () -> client.post(PATH, request, LoadTableResponse.class, Map.of(), new ErrorHandler() { + @Override + public ErrorResponse parseResponse(int code, String json) { + require(code == status, "expected HTTP status " + status + ", received " + code); + return official.parseResponse(code, json); + } + + @Override + public void accept(ErrorResponse error) { + received[0] = true; + require(error.code() == status, "expected status " + status + ", received " + error); + require(type.equals(error.type()), "expected error type " + type + ", received " + error); + official.accept(error); + } + })); + require(received[0], "exception must originate from server response"); + } + + private static void expect(Class expected, Runnable action) { + try { + action.run(); + } catch (RuntimeException failure) { + if (failure.getClass().equals(expected)) { + return; + } + throw failure; + } + throw new AssertionError("Expected " + expected.getSimpleName()); + } + + private static void require(boolean condition, String message) { + if (!condition) { + throw new AssertionError(message); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCommitRace.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCommitRace.java new file mode 100644 index 000000000..41c23394c --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergCommitRace.java @@ -0,0 +1,68 @@ +import java.util.List; +import java.util.Map; +import org.apache.iceberg.MetadataUpdate; +import org.apache.iceberg.exceptions.CommitFailedException; +import org.apache.iceberg.rest.ErrorHandler; +import org.apache.iceberg.rest.ErrorHandlers; +import org.apache.iceberg.rest.HTTPClient; +import org.apache.iceberg.rest.auth.AuthSession; +import org.apache.iceberg.rest.requests.UpdateTableRequest; +import org.apache.iceberg.rest.responses.ErrorResponse; +import org.apache.iceberg.rest.responses.LoadTableResponse; + +public final class TestIcebergCommitRace { + private static final String PATH = "v1/namespaces/analytics/tables/events"; + + public static void main(String[] args) throws Exception { + try (HTTPClient root = HTTPClient.builder(Map.of()).uri(args[0]) + .withHeaders(Map.of("Authorization", "Bearer " + "w".repeat(32))).build(); + HTTPClient client = root.withAuthSession(AuthSession.EMPTY)) { + UpdateTableRequest request = new UpdateTableRequest(List.of(), + List.of(new MetadataUpdate.SetProperties(Map.of("loser-only", "never-visible")))); + String first = conflict(client, request, args[1]); + require(first.equals(conflict(client, request, args[1])), "exact durable conflict replay"); + conflict(client, new UpdateTableRequest(List.of(), + List.of(new MetadataUpdate.SetProperties(Map.of("changed-input", "rejected")))), args[1]); + LoadTableResponse loaded = client.get(PATH, LoadTableResponse.class, + Map.of(), ErrorHandlers.tableErrorHandler()); + require("visible".equals(loaded.tableMetadata().properties().get("winner-only")), + "winner selected"); + require(!loaded.tableMetadata().properties().containsKey("loser-only"), "loser never selected"); + require(!loaded.tableMetadata().properties().containsKey("changed-input"), "identity cannot rebind"); + } + System.out.println("Official SDK head CAS conflict and durable replay passed"); + } + + private static String conflict(HTTPClient client, UpdateTableRequest request, String identity) { + ErrorHandler official = (ErrorHandler) ErrorHandlers.tableCommitHandler(); + String[] body = {null}; + try { + client.post(PATH, request, LoadTableResponse.class, Map.of("Idempotency-Key", identity), + new ErrorHandler() { + @Override + public ErrorResponse parseResponse(int code, String json) { + require(code == 409, "HTTP conflict status"); + body[0] = json; + return official.parseResponse(code, json); + } + + @Override + public void accept(ErrorResponse error) { + require(error.code() == 409, "wire conflict status"); + require("CommitFailedException".equals(error.type()), "wire conflict type"); + official.accept(error); + } + }); + } catch (CommitFailedException expected) { + require(body[0] != null, "server-originated conflict"); + return body[0]; + } + throw new AssertionError("CAS loser must fail, not rebase or succeed"); + } + + private static void require(boolean condition, String message) { + if (!condition) { + throw new AssertionError(message); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergDraftCredentials.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergDraftCredentials.java new file mode 100644 index 000000000..48ae86193 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergDraftCredentials.java @@ -0,0 +1,149 @@ +import com.fasterxml.jackson.databind.ObjectMapper; +import com.sun.net.httpserver.HttpExchange; +import com.sun.net.httpserver.HttpServer; +import java.io.IOException; +import java.net.InetSocketAddress; +import java.nio.charset.StandardCharsets; +import java.time.Instant; +import java.util.HashMap; +import java.util.Map; +import java.util.concurrent.atomic.AtomicBoolean; +import java.util.concurrent.atomic.AtomicInteger; +import org.apache.iceberg.PartitionSpec; +import org.apache.iceberg.Schema; +import org.apache.iceberg.TableMetadata; +import org.apache.iceberg.TableMetadataParser; +import org.apache.iceberg.Transaction; +import org.apache.iceberg.aws.AwsClientProperties; +import org.apache.iceberg.aws.s3.VendedCredentialsProvider; +import org.apache.iceberg.catalog.Namespace; +import org.apache.iceberg.catalog.TableIdentifier; +import org.apache.iceberg.rest.RESTCatalog; +import org.apache.iceberg.types.Types; +import software.amazon.awssdk.auth.credentials.AwsSessionCredentials; + +public final class TestIcebergDraftCredentials { + private static final ObjectMapper JSON = new ObjectMapper(); + private static final String TABLES = "/v1/namespaces/analytics/tables"; + private static final String CREDENTIALS = TABLES + "/events/credentials"; + private static final String TOKEN = "w".repeat(32); + private static final AtomicInteger DRAFTS = new AtomicInteger(); + private static final AtomicInteger REFRESHES = new AtomicInteger(); + private static final AtomicBoolean EXPIRED = new AtomicBoolean(); + private static final Schema SCHEMA = + new Schema(Types.NestedField.required(1, "id", Types.LongType.get())); + + public static void main(String[] args) throws Exception { + HttpServer server = HttpServer.create(new InetSocketAddress("127.0.0.1", 0), 0); + server.createContext("/", TestIcebergDraftCredentials::handle); + server.start(); + String endpoint = "http://127.0.0.1:" + server.getAddress().getPort(); + try (RESTCatalog catalog = new RESTCatalog()) { + catalog.initialize("crowdb", Map.of( + "uri", endpoint, + "token", TOKEN, + "io-impl", "org.apache.iceberg.aws.s3.S3FileIO", + "client.region", "us-east-1", + "rest-metrics-reporting-enabled", "false")); + TableIdentifier name = TableIdentifier.of(Namespace.of("analytics"), "events"); + Transaction first = catalog.buildTable(name, SCHEMA).createTransaction(); + Transaction second = catalog.buildTable(name, SCHEMA).createTransaction(); + verify(first, endpoint, "1"); + verify(second, endpoint, "2"); + require(REFRESHES.get() == 2, "one refresh per exact draft"); + require(DRAFTS.get() == 2, "two invisible same-name drafts"); + EXPIRED.set(true); + boolean rejected = false; + try { + verify(first, endpoint, "1"); + } catch (RuntimeException failure) { + require(failure.getMessage().contains("Draft expired"), "precise expired draft failure"); + rejected = true; + } + require(rejected, "expired draft must not fall back to another same-name table"); + System.out.println("Official staged RESTCatalog credential refresh acceptance passed"); + } finally { + server.stop(0); + } + } + + private static void verify(Transaction transaction, String endpoint, String draft) { + Map properties = new HashMap<>(transaction.table().io().properties()); + require(properties.get("client.refresh-credentials-endpoint").equals(CREDENTIALS + "?table-id=" + draft), + "stage response config reaches S3FileIO unchanged"); + require(transaction.table().location().equals("s3://test-bucket/t/" + draft), "draft location"); + require(properties.get("uri").equals(endpoint), "catalog URI inherited by FileIO"); + require(properties.get("token").equals(TOKEN), "bearer inherited by FileIO"); + properties.put("s3.access-key-id", "expired"); + properties.put("s3.secret-access-key", "expired-secret"); + properties.put("s3.session-token", "expired-token"); + properties.put("s3.session-token-expires-at-ms", "1"); + try (VendedCredentialsProvider provider = (VendedCredentialsProvider) + new AwsClientProperties(properties).credentialsProvider("expired", "expired-secret", "expired-token")) { + AwsSessionCredentials credentials = (AwsSessionCredentials) provider.resolveCredentials(); + require(credentials.accessKeyId().equals("draft-" + draft), "exact draft access key"); + require(credentials.sessionToken().equals("session-" + draft), "exact draft session"); + require(provider.resolveCredentials().accessKeyId().equals("draft-" + draft), "cached grant"); + } + } + + private static void handle(HttpExchange exchange) throws IOException { + try { + require(("Bearer " + TOKEN).equals(exchange.getRequestHeaders().getFirst("Authorization")), + "bearer authentication on refresh and REST requests"); + String path = exchange.getRequestURI().getPath(); + if (path.equals("/v1/config")) { + respond(exchange, Map.of("defaults", Map.of(), "overrides", Map.of())); + } else if (path.equals(TABLES) && exchange.getRequestMethod().equals("POST")) { + require(JSON.readTree(exchange.getRequestBody()).get("stage-create").asBoolean(), "staged create"); + String draft = Integer.toString(DRAFTS.incrementAndGet()); + TableMetadata metadata = TableMetadata.newTableMetadata( + SCHEMA, PartitionSpec.unpartitioned(), "s3://test-bucket/t/" + draft, Map.of()); + respond(exchange, Map.of( + "metadata", JSON.readTree(TableMetadataParser.toJson(metadata)), + "config", Map.of("client.refresh-credentials-endpoint", CREDENTIALS + "?table-id=" + draft))); + } else if (path.equals(CREDENTIALS) && exchange.getRequestMethod().equals("GET")) { + String query = exchange.getRequestURI().getRawQuery(); + require(query.equals("table-id=1") || query.equals("table-id=2"), "exact refresh query"); + String draft = query.substring("table-id=".length()); + if (draft.equals("1") && EXPIRED.get()) { + byte[] bytes = JSON.writeValueAsBytes(Map.of("error", Map.of( + "code", 404, "type", "NoSuchTableException", "message", "Draft expired"))); + exchange.getResponseHeaders().set("Content-Type", "application/json"); + exchange.sendResponseHeaders(404, bytes.length); + exchange.getResponseBody().write(bytes); + return; + } + REFRESHES.incrementAndGet(); + respond(exchange, Map.of("storage-credentials", new Object[] {Map.of( + "prefix", "s3://test-bucket/t/" + draft + "/", + "config", Map.of( + "s3.access-key-id", "draft-" + draft, + "s3.secret-access-key", "secret-" + draft, + "s3.session-token", "session-" + draft, + "s3.session-token-expires-at-ms", Long.toString(Instant.now().plusSeconds(900).toEpochMilli())))})); + } else { + throw new IllegalStateException("Unexpected SDK request: " + exchange.getRequestURI()); + } + } catch (Exception failure) { + byte[] bytes = failure.toString().getBytes(StandardCharsets.UTF_8); + exchange.sendResponseHeaders(400, bytes.length); + exchange.getResponseBody().write(bytes); + } finally { + exchange.close(); + } + } + + private static void respond(HttpExchange exchange, Object value) throws IOException { + byte[] bytes = JSON.writeValueAsBytes(value); + exchange.getResponseHeaders().set("Content-Type", "application/json"); + exchange.sendResponseHeaders(200, bytes.length); + exchange.getResponseBody().write(bytes); + } + + private static void require(boolean valid, String message) { + if (!valid) { + throw new IllegalStateException(message); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileIO.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileIO.java new file mode 100644 index 000000000..13125128d --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileIO.java @@ -0,0 +1,112 @@ +import com.sun.net.httpserver.HttpServer; +import java.net.InetSocketAddress; +import java.net.URI; +import java.nio.charset.StandardCharsets; +import java.util.Arrays; +import java.util.Base64; +import java.util.HashMap; +import java.util.Map; +import java.util.Properties; +import java.util.concurrent.atomic.AtomicInteger; +import org.apache.iceberg.aws.s3.S3FileIO; +import org.apache.iceberg.io.InputFile; +import org.apache.iceberg.io.PositionOutputStream; +import org.apache.iceberg.io.SeekableInputStream; +import software.amazon.awssdk.core.sync.RequestBody; +import software.amazon.awssdk.services.s3.S3Client; +import software.amazon.awssdk.services.s3.model.CompletedPart; +import software.amazon.awssdk.services.s3.model.S3Exception; + +public class TestIcebergFileIO { + public static void main(String[] args) throws Exception { + Properties configuration = new Properties(); + configuration.load(System.in); + Map properties = new HashMap<>(); + properties.put("s3.endpoint", configuration.getProperty("endpoint")); + properties.put("client.region", "us-east-1"); + properties.put("s3.path-style-access", "true"); + properties.put("s3.multipart.part-size-bytes", "5242880"); + properties.put("s3.multipart.threshold", "1.0"); + AtomicInteger credentialRequests = new AtomicInteger(); + HttpServer credentials = HttpServer.create(new InetSocketAddress("127.0.0.1", 0), 0); + byte[] credentialResponse = Base64.getDecoder().decode(configuration.getProperty("credentials")); + credentials.createContext("/v1/namespaces/test/tables/test/credentials", exchange -> { + if (!"GET".equals(exchange.getRequestMethod())) { + exchange.sendResponseHeaders(405, -1); + exchange.close(); + return; + } + credentialRequests.incrementAndGet(); + exchange.getResponseHeaders().set("Content-Type", "application/json"); + exchange.sendResponseHeaders(200, credentialResponse.length); + try (var output = exchange.getResponseBody()) { + output.write(credentialResponse); + } + }); + credentials.start(); + properties.put("uri", "http://127.0.0.1:" + credentials.getAddress().getPort()); + properties.put("client.refresh-credentials-endpoint", "/v1/namespaces/test/tables/test/credentials"); + properties.put("rest.auth.type", "none"); + try (S3FileIO files = new S3FileIO()) { + files.initialize(properties); + String prefix = configuration.getProperty("location"); + byte[] small = "{\"client\":\"iceberg-java-1.11.0\"}".getBytes(StandardCharsets.UTF_8); + verify(files, prefix + "metadata/sdk-small.json", small); + byte[] large = new byte[6 * 1024 * 1024]; + Arrays.fill(large, (byte) 'x'); + byte[] start = "{\"data\":\"".getBytes(StandardCharsets.UTF_8); + System.arraycopy(start, 0, large, 0, start.length); + large[large.length - 2] = '"'; + large[large.length - 1] = '}'; + verify(files, prefix + "metadata/sdk-multipart.json", large); + verifyLateError(files.client(), prefix + "metadata/sdk-invalid.json"); + TestIcebergFileOperations.run(files.client(), prefix); + if (credentialRequests.get() != 1) { + throw new AssertionError("SDK did not fetch and cache the delegated credential response"); + } + } finally { + credentials.stop(0); + } + System.out.println("Apache Iceberg 1.11.0 S3FileIO PUT, multipart, HEAD, GET, seek and embedded error passed"); + } + + private static void verifyLateError(S3Client client, String location) { + URI uri = URI.create(location); + String bucket = uri.getHost(); + String key = uri.getPath().substring(1); + String upload = client.createMultipartUpload(request -> request.bucket(bucket).key(key)).uploadId(); + String etag = client.uploadPart( + request -> request.bucket(bucket).key(key).uploadId(upload).partNumber(1), + RequestBody.fromString("{invalid-json")).eTag(); + try { + client.completeMultipartUpload(request -> request.bucket(bucket).key(key).uploadId(upload) + .multipartUpload(parts -> parts.parts(CompletedPart.builder().partNumber(1).eTag(etag).build()))); + throw new AssertionError("SDK accepted an embedded Complete error as success"); + } catch (S3Exception error) { + if (!"InvalidRequest".equals(error.awsErrorDetails().errorCode())) { + throw error; + } + } finally { + client.abortMultipartUpload(request -> request.bucket(bucket).key(key).uploadId(upload)); + } + } + + private static void verify(S3FileIO files, String location, byte[] bytes) throws Exception { + try (PositionOutputStream output = files.newOutputFile(location).create()) { + output.write(bytes); + } + InputFile file = files.newInputFile(location); + if (!file.exists() || file.getLength() != bytes.length) { + throw new AssertionError("HEAD returned incorrect file state"); + } + try (SeekableInputStream input = file.newStream()) { + if (!Arrays.equals(input.readAllBytes(), bytes)) { + throw new AssertionError("GET changed canonical bytes"); + } + input.seek(bytes.length - 2L); + if (input.read() != bytes[bytes.length - 2] || input.read() != bytes[bytes.length - 1]) { + throw new AssertionError("Range GET returned incorrect tail"); + } + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileOperations.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileOperations.java new file mode 100644 index 000000000..02cecc576 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileOperations.java @@ -0,0 +1,53 @@ +import java.net.URI; +import java.nio.charset.StandardCharsets; +import java.util.Arrays; +import software.amazon.awssdk.core.sync.RequestBody; +import software.amazon.awssdk.services.s3.S3Client; +import software.amazon.awssdk.services.s3.model.S3Exception; +import software.amazon.awssdk.services.s3.model.Tag; + +public final class TestIcebergFileOperations { + public static void run(S3Client client, String prefix) { + URI location = URI.create(prefix); + String bucket = location.getHost(); + String root = location.getPath().substring(1); + String key = root + "metadata/operations.json"; + byte[] bytes = "{\"immutable\":true}".getBytes(StandardCharsets.UTF_8); + client.putObject(request -> request.bucket(bucket).key(key), RequestBody.fromBytes(bytes)); + String etag = client.headObject(request -> request.bucket(bucket).key(key)).eTag(); + client.putObject(request -> request.bucket(bucket).key(key), RequestBody.fromBytes(bytes)); + reject(409, "OperationAborted", () -> client.putObject(request -> request.bucket(bucket).key(key), + RequestBody.fromString("{\"immutable\":false}"))); + reject(400, "InvalidRequest", () -> client.createBucket(request -> request.bucket(bucket))); + reject(400, "InvalidRequest", () -> client.deleteBucket(request -> request.bucket(bucket))); + reject(400, "InvalidRequest", () -> client.listObjectsV2(request -> request.bucket(bucket))); + reject(400, "InvalidRequest", () -> client.getBucketLifecycleConfiguration(request -> request.bucket(bucket))); + reject(400, "InvalidRequest", () -> client.deleteBucketLifecycle(request -> request.bucket(bucket))); + reject(400, "InvalidRequest", () -> client.getObjectTagging(request -> request.bucket(bucket).key(key))); + reject(400, "InvalidRequest", () -> client.putObjectTagging(request -> request.bucket(bucket).key(key) + .tagging(tags -> tags.tagSet(Tag.builder().key("owner").value("changed").build())))); + reject(400, "InvalidRequest", () -> client.deleteObjectTagging(request -> request.bucket(bucket).key(key))); + reject(400, "InvalidRequest", () -> client.deleteObject(request -> request.bucket(bucket).key(key))); + reject(400, "InvalidRequest", () -> client.putObject(request -> request.bucket(bucket) + .key(root + "../escape.json"), RequestBody.fromBytes(bytes))); + String foreign = "t/" + (root.charAt(2) == '0' ? '1' : '0') + root.substring(3) + "metadata/foreign.json"; + reject(403, "AccessDenied", () -> client.putObject(request -> request.bucket(bucket).key(foreign), + RequestBody.fromBytes(bytes))); + if (!etag.equals(client.headObject(request -> request.bucket(bucket).key(key)).eTag()) + || !Arrays.equals(bytes, client.getObjectAsBytes(request -> request.bucket(bucket).key(key)).asByteArray())) { + throw new AssertionError("unsupported operations changed immutable authority"); + } + System.out.println("Official S3 operation restrictions, immutable replay and exact table scope passed"); + } + + private static void reject(int status, String code, Runnable operation) { + try { + operation.run(); + throw new AssertionError("unsupported operation succeeded"); + } catch (S3Exception failure) { + if (failure.statusCode() != status || !code.equals(failure.awsErrorDetails().errorCode())) { + throw failure; + } + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergNamespaces.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergNamespaces.java new file mode 100644 index 000000000..ad2c8cdd4 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergNamespaces.java @@ -0,0 +1,45 @@ +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import org.apache.iceberg.catalog.Namespace; +import org.apache.iceberg.rest.ErrorHandlers; +import org.apache.iceberg.rest.HTTPClient; +import org.apache.iceberg.rest.RESTCatalog; +import org.apache.iceberg.rest.auth.AuthSession; +import org.apache.iceberg.rest.responses.ListNamespacesResponse; + +public class TestIcebergNamespaces { + public static void main(String[] args) throws Exception { + Map properties = new HashMap<>(); + properties.put("uri", args[0]); + properties.put("token", "r".repeat(32)); + properties.put("rest-page-size", "1"); + properties.put("rest-metrics-reporting-enabled", "false"); + properties.put("io-impl", TestIcebergCatalogReads.TestNoFileIO.class.getName()); + try (HTTPClient root = HTTPClient.builder(properties).uri(args[0]) + .withHeaders(Map.of("Authorization", "Bearer " + "r".repeat(32))).build(); + HTTPClient client = root.withAuthSession(AuthSession.EMPTY)) { + ListNamespacesResponse complete = client.get("v1/namespaces", Map.of(), + ListNamespacesResponse.class, Map.of(), ErrorHandlers.namespaceErrorHandler()); + require(complete.nextPageToken() == null, "complete response has no continuation"); + require(complete.namespaces().equals(List.of(Namespace.of("analytics"))), "complete contents"); + ListNamespacesResponse first = client.get("v1/namespaces", + Map.of("pageToken", "", "pageSize", "1"), ListNamespacesResponse.class, + Map.of(), ErrorHandlers.namespaceErrorHandler()); + require(first.namespaces().isEmpty(), "stale first page is empty"); + require(first.nextPageToken() != null, "stale first page can continue"); + } + try (RESTCatalog catalog = new RESTCatalog()) { + catalog.initialize("crowdb", properties); + require(catalog.listNamespaces().equals(List.of(Namespace.of("analytics"))), + "official catalog follows empty pages through the live namespace and final stale page"); + } + System.out.println("Java namespace pagination passed"); + } + + private static void require(boolean condition, String message) { + if (!condition) { + throw new AssertionError(message); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergPartitionStatistics.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergPartitionStatistics.java new file mode 100644 index 000000000..230735d8e --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergPartitionStatistics.java @@ -0,0 +1,125 @@ +import java.util.List; +import java.util.Map; +import java.util.UUID; +import org.apache.iceberg.BaseTable; +import org.apache.iceberg.DataFile; +import org.apache.iceberg.MetadataUpdate; +import org.apache.iceberg.PartitionKey; +import org.apache.iceberg.PartitionSpec; +import org.apache.iceberg.PartitionStatisticsFile; +import org.apache.iceberg.PartitionStatsHandler; +import org.apache.iceberg.Schema; +import org.apache.iceberg.Table; +import org.apache.iceberg.UpdateRequirement; +import org.apache.iceberg.catalog.TableIdentifier; +import org.apache.iceberg.data.GenericRecord; +import org.apache.iceberg.data.Record; +import org.apache.iceberg.data.parquet.GenericParquetWriter; +import org.apache.iceberg.expressions.Expressions; +import org.apache.iceberg.io.DataWriter; +import org.apache.iceberg.parquet.Parquet; +import org.apache.iceberg.rest.ErrorHandlers; +import org.apache.iceberg.rest.HTTPClient; +import org.apache.iceberg.rest.RESTCatalog; +import org.apache.iceberg.rest.auth.AuthSession; +import org.apache.iceberg.rest.requests.UpdateTableRequest; +import org.apache.iceberg.rest.responses.LoadTableResponse; +import org.apache.iceberg.types.Types; + +public final class TestIcebergPartitionStatistics { + private static final TableIdentifier NAME = TableIdentifier.of("analytics", "partition_statistics"); + + public static void run(RESTCatalog catalog, String endpoint, boolean verifyOnly) throws Exception { + if (verifyOnly) { + verify(catalog.loadTable(NAME), 20, 2); + verify(catalog.loadTable(TableIdentifier.of("analytics", "staged_statistics")), 10, 1); + return; + } + Schema schema = new Schema( + Types.NestedField.required(1, "id", Types.LongType.get()), + Types.NestedField.optional(2, "category", Types.StringType.get())); + PartitionSpec spec = PartitionSpec.builderFor(schema).identity("id").build(); + Table table = catalog.buildTable(NAME, schema).withPartitionSpec(spec) + .withProperty("format-version", "2").create(); + table.newAppend().appendFile(writeData(table)).commit(); + PartitionStatisticsFile statistics = PartitionStatsHandler.computeAndWriteStatsFile(table); + publishAndReplay(table, endpoint, statistics); + verify(table, 10, 1); + table.updateSchema().addColumn("extra", Types.StringType.get()).commit(); + table.updateSpec().addField(Expressions.bucket("category", 8)).commit(); + table.updateProperties().set("format-version", "3").commit(); + verify(table, 10, 1); + table.newAppend().appendFile(writeData(table)).commit(); + table.updatePartitionStatistics() + .setPartitionStatistics(PartitionStatsHandler.computeAndWriteStatsFile(table)).commit(); + verify(table, 20, 2); + var transaction = catalog.buildTable(TableIdentifier.of("analytics", "staged_statistics"), schema) + .withPartitionSpec(spec).withProperty("format-version", "3").createTransaction(); + transaction.newAppend().appendFile(writeData(transaction.table())).commit(); + transaction.updatePartitionStatistics().setPartitionStatistics( + PartitionStatsHandler.computeAndWriteStatsFile(transaction.table())).commit(); + transaction.commitTransaction(); + verify(catalog.loadTable(TableIdentifier.of("analytics", "staged_statistics")), 10, 1); + System.out.println("Official partition statistics publication, replay, evolution and staged creation passed"); + } + + private static void publishAndReplay(Table table, String endpoint, PartitionStatisticsFile statistics) + throws java.io.IOException { + String path = "v1/namespaces/analytics/tables/partition_statistics"; + var request = new UpdateTableRequest( + List.of(new UpdateRequirement.AssertTableUUID(((BaseTable) table).operations().current().uuid())), + List.of(new MetadataUpdate.SetPartitionStatistics(statistics))); + UUID random = UUID.randomUUID(); + UUID identity = new UUID((System.currentTimeMillis() << 16) | 0x7000 + | (random.getMostSignificantBits() & 0xfff), random.getLeastSignificantBits()); + Map headers = Map.of("Idempotency-Key", identity.toString()); + try (HTTPClient root = HTTPClient.builder(Map.of()).uri(endpoint) + .withHeaders(Map.of("Authorization", "Bearer " + "w".repeat(32))).build(); + HTTPClient client = root.withAuthSession(AuthSession.EMPTY)) { + LoadTableResponse first = client.post(path, request, LoadTableResponse.class, headers, + ErrorHandlers.tableCommitHandler()); + LoadTableResponse replay = client.post(path, request, LoadTableResponse.class, headers, + ErrorHandlers.tableCommitHandler()); + require(first.metadataLocation().equals(replay.metadataLocation()), "exact statistics publication replay"); + } + table.refresh(); + } + + private static DataFile writeData(Table table) throws Exception { + GenericRecord row = GenericRecord.create(table.schema()); + row.setField("id", 7L); + row.setField("category", "same"); + PartitionKey partition = new PartitionKey(table.spec(), table.schema()); + partition.partition(row); + DataWriter writer = Parquet.writeData(table.io().newOutputFile( + table.location() + "/data/" + UUID.randomUUID() + ".parquet")) + .schema(table.schema()).withSpec(table.spec()).withPartition(partition) + .createWriterFunc(parquetSchema -> GenericParquetWriter.create(table.schema(), parquetSchema)) + .set("write.parquet.compression-codec", "zstd").build(); + try (writer) { + for (int index = 0; index < 10; index++) { + writer.write(row); + } + } + return writer.toDataFile(); + } + + private static void verify(Table table, long expectedRecords, int expectedFiles) throws Exception { + long records = 0; + int files = 0; + try (var statistics = table.newPartitionStatisticsScan().scan()) { + for (var row : statistics) { + records += row.dataRecordCount(); + files += row.dataFileCount(); + require(row.dvCount() == null || row.dvCount() == 0, "missing historical DV defaults to zero"); + } + } + require(records == expectedRecords && files == expectedFiles, "selected partition statistics counts"); + } + + private static void require(boolean valid, String message) { + if (!valid) { + throw new AssertionError(message); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergResponseLoss.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergResponseLoss.java new file mode 100644 index 000000000..0d5050fd4 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergResponseLoss.java @@ -0,0 +1,33 @@ +import java.util.Map; +import org.apache.iceberg.Schema; +import org.apache.iceberg.catalog.Namespace; +import org.apache.iceberg.catalog.TableIdentifier; +import org.apache.iceberg.rest.RESTCatalog; +import org.apache.iceberg.types.Types; + +public final class TestIcebergResponseLoss { + public static void main(String[] args) throws Exception { + TableIdentifier table = TableIdentifier.of(Namespace.of("analytics"), "java_lost_reply"); + Schema schema = new Schema(Types.NestedField.required(1, "id", Types.LongType.get())); + try (RESTCatalog first = new RESTCatalog(); RESTCatalog second = new RESTCatalog()) { + first.initialize("crowdb", Map.of("uri", args[0], "token", "w".repeat(32), + "io-impl", "org.apache.iceberg.aws.s3.S3FileIO", "rest-metrics-reporting-enabled", "false")); + second.initialize("crowdb", Map.of("uri", args[1], "token", "w".repeat(32), + "io-impl", "org.apache.iceberg.aws.s3.S3FileIO", "rest-metrics-reporting-enabled", "false")); + boolean failed = false; + try { + first.buildTable(table, schema).create(); + } catch (RuntimeException expected) { + failed = true; + } + if (!failed || !second.tableExists(table)) { + throw new AssertionError("lost Java create response did not preserve one visible table"); + } + second.loadTable(table); + if (!second.dropTable(table)) { + throw new AssertionError("second listener could not drop the committed table"); + } + System.out.println("Official Java RESTCatalog response-loss acceptance passed"); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergSelectedFiles.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergSelectedFiles.java new file mode 100644 index 000000000..7970dc4cb --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergSelectedFiles.java @@ -0,0 +1,118 @@ +import java.util.Map; +import java.util.UUID; +import org.apache.iceberg.BaseTable; +import org.apache.iceberg.DataFile; +import org.apache.iceberg.DataFiles; +import org.apache.iceberg.DeleteFile; +import org.apache.iceberg.FileMetadata; +import org.apache.iceberg.Schema; +import org.apache.iceberg.Table; +import org.apache.iceberg.catalog.TableIdentifier; +import org.apache.iceberg.data.GenericRecord; +import org.apache.iceberg.data.Record; +import org.apache.iceberg.data.parquet.GenericParquetWriter; +import org.apache.iceberg.deletes.EqualityDeleteWriter; +import org.apache.iceberg.exceptions.BadRequestException; +import org.apache.iceberg.io.DataWriter; +import org.apache.iceberg.parquet.Parquet; +import org.apache.iceberg.rest.RESTCatalog; +import org.apache.iceberg.types.Types; + +public final class TestIcebergSelectedFiles { + public static void main(String[] args) throws Exception { + try (RESTCatalog catalog = new RESTCatalog()) { + catalog.initialize("crowdb", Map.of("uri", args[0], "token", "w".repeat(32), + "io-impl", "org.apache.iceberg.aws.s3.S3FileIO", "client.region", "us-east-1", + "rest-metrics-reporting-enabled", "false")); + Schema schema = new Schema( + Types.NestedField.required(1, "id", Types.LongType.get()), + Types.NestedField.required(2, "message", Types.StringType.get())); + Table table = catalog.buildTable(TableIdentifier.of("analytics", "selected_files"), schema) + .withProperty("format-version", "2").create(); + DataFile data = data(table); + DeleteFile equality = equality(table); + require(table.io().newInputFile(data.location()).exists(), "ordinary data upload"); + require(table.io().newInputFile(equality.location()).exists(), "ordinary equality-delete upload"); + table.newAppend().appendFile(data).commit(); + long beforeDelete = table.currentSnapshot().snapshotId(); + TestIcebergVersionRows.rows(org.apache.iceberg.data.IcebergGenerics.read(table), java.util.List.of(1L)); + DataFile wrongData = DataFiles.builder(table.spec()).withPath(equality.location()) + .withFormat("PARQUET").withFileSizeInBytes(equality.fileSizeInBytes()) + .withRecordCount(equality.recordCount()).build(); + rejected(table, () -> table.newAppend().appendFile(wrongData).commit()); + DeleteFile wrongPosition = FileMetadata.deleteFileBuilder(table.spec()).ofPositionDeletes() + .withPath(equality.location()).withFormat("PARQUET") + .withFileSizeInBytes(equality.fileSizeInBytes()).withRecordCount(equality.recordCount()).build(); + rejected(table, () -> table.newRowDelta().addDeletes(wrongPosition).commit()); + DeleteFile wrongEquality = FileMetadata.deleteFileBuilder(table.spec()) + .ofEqualityDeletes(table.schema().findField("message").fieldId()) + .withPath(equality.location()).withFormat("PARQUET") + .withFileSizeInBytes(equality.fileSizeInBytes()).withRecordCount(equality.recordCount()).build(); + rejected(table, () -> table.newRowDelta().addDeletes(wrongEquality).commit()); + table.newRowDelta().addDeletes(equality).commit(); + table.refresh(); + int files = 0; + try (var tasks = table.newScan().planFiles()) { + for (var task : tasks) { + require(task.file().location().equals(data.location()), "original data remains selected"); + require(task.deletes().size() == 1 + && task.deletes().get(0).location().equals(equality.location()), "equality delete is selected"); + files++; + } + } + require(files == 1, "wrong uses never add files"); + TestIcebergVersionRows.rows(org.apache.iceberg.data.IcebergGenerics.read(table), java.util.List.of()); + TestIcebergVersionRows.rows(org.apache.iceberg.data.IcebergGenerics.read(table).useSnapshot(beforeDelete), + java.util.List.of(1L)); + System.out.println("Official identical S3 uploads and selected data/delete validation passed"); + } + } + + private static DataFile data(Table table) throws Exception { + DataWriter writer = Parquet.writeData(table.io().newOutputFile(location(table))) + .schema(table.schema()).withSpec(table.spec()) + .createWriterFunc(parquet -> GenericParquetWriter.create(table.schema(), parquet)).build(); + try (writer) { + GenericRecord row = GenericRecord.create(table.schema()); + row.setField("id", 1L); + row.setField("message", "one"); + writer.write(row); + } + return writer.toDataFile(); + } + + private static DeleteFile equality(Table table) throws Exception { + Schema schema = table.schema().select("id"); + EqualityDeleteWriter writer = Parquet.writeDeletes(table.io().newOutputFile(location(table))) + .rowSchema(schema).withSpec(table.spec()).equalityFieldIds(schema.findField("id").fieldId()) + .createWriterFunc(parquet -> GenericParquetWriter.create(schema, parquet)).buildEqualityWriter(); + try (writer) { + GenericRecord row = GenericRecord.create(schema); + row.setField("id", 1L); + writer.write(row); + } + return writer.toDeleteFile(); + } + + private static String location(Table table) { + return table.location() + "/objects/" + UUID.randomUUID() + ".parquet"; + } + + private static void rejected(Table table, Runnable operation) { + String before = ((BaseTable) table).operations().current().metadataFileLocation(); + try { + operation.run(); + throw new AssertionError("wrong selected file use was accepted"); + } catch (BadRequestException expected) { + table.refresh(); + require(before.equals(((BaseTable) table).operations().current().metadataFileLocation()), + "rejection must preserve the exact selected metadata file"); + } + } + + private static void require(boolean valid, String message) { + if (!valid) { + throw new AssertionError(message); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergVersionRows.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergVersionRows.java new file mode 100644 index 000000000..ba3b1f6f6 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergVersionRows.java @@ -0,0 +1,79 @@ +import java.util.ArrayList; +import java.util.List; +import java.util.UUID; +import org.apache.iceberg.DataFile; +import org.apache.iceberg.Schema; +import org.apache.iceberg.Table; +import org.apache.iceberg.catalog.TableIdentifier; +import org.apache.iceberg.data.GenericRecord; +import org.apache.iceberg.data.IcebergGenerics; +import org.apache.iceberg.data.Record; +import org.apache.iceberg.data.parquet.GenericParquetWriter; +import org.apache.iceberg.io.DataWriter; +import org.apache.iceberg.parquet.Parquet; +import org.apache.iceberg.rest.RESTCatalog; +import org.apache.iceberg.types.Types; + +public final class TestIcebergVersionRows { + public static void run(RESTCatalog catalog, boolean verifyOnly) throws Exception { + Schema schema = new Schema(Types.NestedField.required(1, "id", Types.LongType.get())); + for (int version = 1; version <= 3; version++) { + TableIdentifier name = TableIdentifier.of("analytics", "rows_v" + version); + Table table; + if (verifyOnly) { + table = catalog.loadTable(name); + } else { + table = catalog.buildTable(name, schema) + .withProperty("format-version", Integer.toString(version)).create(); + table.newAppend().appendFile(write(table, 10L)).commit(); + long first = table.currentSnapshot().snapshotId(); + table.newAppend().appendFile(write(table, 20L)).commit(); + rows(IcebergGenerics.read(table).useSnapshot(first), List.of(10L)); + rows(IcebergGenerics.read(table), List.of(10L, 20L)); + for (int upgrade = version + 1; upgrade <= 3; upgrade++) { + table.updateProperties().set("format-version", Integer.toString(upgrade)).commit(); + table = catalog.loadTable(name); + rows(IcebergGenerics.read(table), List.of(10L, 20L)); + rows(IcebergGenerics.read(table).useSnapshot(first), List.of(10L)); + } + table.expireSnapshots().expireSnapshotId(first).cleanExpiredFiles(false).commit(); + table.refresh(); + require(table.snapshot(first) == null, "logical expiry removes the old snapshot"); + table.updateProperties().set("expired-snapshot", Long.toString(first)).commit(); + } + long expired = Long.parseLong(table.properties().get("expired-snapshot")); + require(table.snapshot(expired) == null, "expired snapshot remains absent after reload"); + rows(IcebergGenerics.read(table), List.of(10L, 20L)); + } + } + + private static DataFile write(Table table, long value) throws Exception { + DataWriter writer = Parquet.writeData(table.io().newOutputFile( + table.location() + "/data/" + UUID.randomUUID() + ".parquet")) + .schema(table.schema()).withSpec(table.spec()) + .createWriterFunc(parquet -> GenericParquetWriter.create(table.schema(), parquet)).build(); + try (writer) { + GenericRecord row = GenericRecord.create(table.schema()); + row.setField("id", value); + writer.write(row); + } + return writer.toDataFile(); + } + + static void rows(IcebergGenerics.ScanBuilder scan, List expected) throws Exception { + List actual = new ArrayList<>(); + try (var rows = scan.build()) { + for (Record row : rows) { + actual.add((Long) row.getField("id")); + } + } + actual.sort(Long::compareTo); + require(actual.equals(expected), "visible rows: expected " + expected + ", got " + actual); + } + + private static void require(boolean condition, String message) { + if (!condition) { + throw new AssertionError(message); + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_journal.rs b/app/crowdb-access-server/tests/common/iceberg_journal.rs new file mode 100644 index 000000000..58689e007 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_journal.rs @@ -0,0 +1,104 @@ +use crowdb_access_iceberg::catalog::{CatalogContext, CatalogStore}; +use crowdb_access_iceberg::key::{NamespaceId, OperationId}; +use crowdb_access_iceberg::namespace::{ + NamespaceAction, NamespaceIdentifier, NamespaceJournal, NamespaceOperation, NamespacePhase, +}; +use crowdb_access_iceberg::operation::{ + PayloadStore, RequestIdentity, RetryAdmission, RetryLedger, RetryRecord, +}; + +use super::common::{now_ms, TestIcebergStack}; + +pub async fn verify_recovery(stack: &mut TestIcebergStack, context: CatalogContext) { + let store = stack.store().await; + let property_request = super::property::prepare(store.clone(), context).await; + let creation_request = super::creation::prepare(store.clone(), context).await; + let drop_request = super::dropping::prepare(store.clone(), context).await; + let identity = fresh_identity(store.as_ref()).await; + let body = vec![23; 70 * 1024]; + let input = PayloadStore::new(store.clone()) + .put(context.catalog, identity.operation, &body) + .await + .unwrap(); + let mut operation = NamespaceOperation { + context, + identity, + principal: "writer".into(), + action: NamespaceAction::Create, + identifier: NamespaceIdentifier::new(vec!["durable".into()]).unwrap(), + namespace: NamespaceId::random(), + parent: None, + phase: NamespacePhase::Prepared, + revision: 1, + input, + mutation: None, + scan_after: Vec::new(), + scan_generation: 0, + outcome: None, + }; + let journal = NamespaceJournal::new(store.clone()); + journal.begin(operation.clone()).await.unwrap(); + let previous = operation.clone(); + operation.phase = NamespacePhase::Reserved; + operation.revision += 1; + assert!(journal.advance(&previous, &operation).await.unwrap()); + let request = RetryRecord { + identity, + principal: "writer".into(), + route: "POST /namespaces".into(), + digest: operation.input.digest, + context, + retained_until_ms: 0, + status: 0, + body: Vec::new(), + }; + let ledger = RetryLedger::new(store); + ledger.begin(request.clone(), now_ms()).await.unwrap(); + ledger + .finish(request.clone(), 200, body.clone(), now_ms()) + .await + .unwrap(); + stack.chunk_kv.restart().await; + let recovered_store = stack.store().await; + super::property::verify(recovered_store.clone(), &property_request).await; + super::creation::verify(recovered_store.clone(), &creation_request).await; + super::dropping::verify(recovered_store.clone(), &drop_request).await; + let recovered_journal = NamespaceJournal::new(recovered_store.clone()); + assert_eq!( + recovered_journal + .load(context, identity.operation) + .await + .unwrap() + .unwrap(), + operation + ); + let RetryAdmission::Replay(result) = RetryLedger::new(recovered_store) + .begin(request, now_ms()) + .await + .unwrap() + else { + panic!("large response replay") + }; + assert_eq!(result.body, body); + assert_eq!(result.principal, "writer"); +} + +async fn fresh_identity(store: &dyn CatalogStore) -> RequestIdentity { + for _ in 0..32 { + let operation = OperationId::random(); + let key = crowdb_access_iceberg::operation::ledger_key( + crowdb_access_iceberg::key::SystemScope::RetryBinding, + operation, + ) + .unwrap() + .encode() + .unwrap(); + if store.get(&key).await.unwrap().is_none() { + return RequestIdentity { + operation, + issued_ms: now_ms(), + }; + } + } + panic!("no free retry slot for fixture"); +} diff --git a/app/crowdb-access-server/tests/common/iceberg_multipart.rs b/app/crowdb-access-server/tests/common/iceberg_multipart.rs new file mode 100644 index 000000000..ca149a34f --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_multipart.rs @@ -0,0 +1,128 @@ +use std::sync::Arc; + +use crowdb_access_iceberg::catalog::CatalogContext; +use crowdb_access_iceberg::file::{ + FileIdentity, FileReader, FileTreeWriter, MultipartLimits, MultipartPart, MultipartPhase, + MultipartRecovery, MultipartRepository, MultipartSelection, MultipartSession, NativeFileBlocks, + SelectedPart, TableLocation, +}; +use crowdb_access_iceberg::key::{FileId, OperationId}; + +use crate::common::TestIcebergStack; + +pub async fn verify_restart(stack: &mut TestIcebergStack, context: CatalogContext, table: TableLocation) { + let client = crate::chunks(stack).await; + let blocks = Arc::new(NativeFileBlocks::new(client.clone(), stack.store().await)); + let initial = MultipartSession { + context, + upload: OperationId::random(), + owner: FileIdentity { + table, + file: FileId::random(), + }, + location: table.file("multipart/native.bin").unwrap(), + principal: [1; 32], + revision: 1, + created_ms: 100, + expires_ms: 10_100, + limits: MultipartLimits { + max_parts: 10, + max_part_bytes: 100, + max_file_bytes: 1000, + max_staged_bytes: 1000, + ttl_ms: 10_000, + }, + phase: MultipartPhase::Open, + part_count: 0, + staged_bytes: 0, + completion: None, + published: None, + pending: None, + credit: None, + }; + let repository = MultipartRepository::new(stack.store().await); + repository.begin(&initial, 100).await.unwrap(); + let owner = FileIdentity { + table, + file: FileId::random(), + }; + let mut writer = FileTreeWriter::new(blocks.clone(), owner, 8).unwrap(); + writer.push(b"durable multipart bytes").await.unwrap(); + let part = MultipartPart { + upload: initial.upload, + number: 1, + revision: 1, + modified_ms: 101, + owner, + tree: writer.finish().await.unwrap(), + }; + assert!(repository.reserve_part(&initial, &part, 101).await.unwrap()); + let recovery = MultipartRecovery::new(stack.store().await, blocks.clone(), 7, 8).unwrap(); + let page = recovery.recover_page(context, None, 102).await.unwrap(); + assert!(page.failures.is_empty(), "{:?}", page.failures); + assert_eq!(page.progressed, 1); + let current = repository.load(context, initial.upload).await.unwrap().unwrap(); + assert!(current.pending.is_none()); + let selection = MultipartSelection::new(vec![SelectedPart { + number: 1, + revision: 1, + digest: part.tree.digest, + }]) + .unwrap(); + assert!(repository + .freeze_completion(¤t, &selection, 103) + .await + .unwrap()); + let page = recovery.recover_page(context, None, 104).await.unwrap(); + assert!(page.failures.is_empty(), "{:?}", page.failures); + assert_eq!(page.progressed, 1); + let current = repository.load(context, initial.upload).await.unwrap().unwrap(); + assert_eq!(current.completion.as_ref().unwrap().progress.completed_bytes, 7); + client.shutdown_small_writes().await.unwrap(); + drop(recovery); + drop(repository); + drop(blocks); + drop(client); + stack.chunk_kv.restart().await; + verify_resumed(stack, &initial).await; +} + +async fn verify_resumed(stack: &TestIcebergStack, initial: &MultipartSession) { + let client = crate::chunks(stack).await; + let blocks = Arc::new(NativeFileBlocks::new(client.clone(), stack.store().await)); + let repository = MultipartRepository::new(stack.store().await); + let recovered = repository + .load(initial.context, initial.upload) + .await + .unwrap() + .unwrap(); + assert_eq!(recovered.completion.as_ref().unwrap().progress.completed_bytes, 7); + let recovery = MultipartRecovery::new(stack.store().await, blocks.clone(), 7, 8).unwrap(); + for _ in 0..3 { + let page = recovery.recover_page(initial.context, None, 105).await.unwrap(); + assert!(page.failures.is_empty(), "{:?}", page.failures); + assert_eq!(page.progressed, 1); + } + let page = recovery.recover_page(initial.context, None, 105).await.unwrap(); + assert_eq!(page.awaiting_seal, vec![initial.upload]); + let current = repository + .load(initial.context, initial.upload) + .await + .unwrap() + .unwrap(); + let tree = repository + .assembled_tree(¤t, blocks.clone(), 8) + .await + .unwrap(); + let reader = FileReader::from_tree(blocks, initial.owner, tree, None, 4096).unwrap(); + assert_eq!(crate::read_all(reader).await, b"durable multipart bytes"); + assert!( + crowdb_access_iceberg::file::FileRepository::new(stack.store().await) + .load(initial.context, &initial.location) + .await + .unwrap() + .is_none() + ); + assert!(repository.abort(¤t).await.unwrap()); + client.shutdown_small_writes().await.unwrap(); +} diff --git a/app/crowdb-access-server/tests/common/iceberg_namespace.rs b/app/crowdb-access-server/tests/common/iceberg_namespace.rs new file mode 100644 index 000000000..d4987e72a --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_namespace.rs @@ -0,0 +1,111 @@ +use crowdb_access_iceberg::catalog::{CasOutcome, CatalogStore}; +use crowdb_access_iceberg::key::{CatalogId, CatalogScope, NamespaceId, OperationId}; +use crowdb_access_iceberg::namespace::{ + name_key, ChildScan, NamespaceMapping, NamespaceMappingState, NamespaceStore, +}; +use crowdb_access_iceberg::operation::mutation_identity; +use crowdb_access_iceberg::record::StorageRecord; + +use super::common::TestIcebergStack; + +pub async fn verify_name_index(stack: &TestIcebergStack, catalog: CatalogId) { + let store = stack.store().await; + let parent = Some(NamespaceId::random()); + let mut entries = Vec::new(); + for name in ["a", "b", "c"] { + let mapping = NamespaceMapping { + catalog, + parent, + name: name.into(), + namespace: NamespaceId::random(), + name_epoch: 1, + operation: OperationId::random(), + state: NamespaceMappingState::Reserved, + }; + let key = name_key(catalog, parent, name).unwrap().encode().unwrap(); + let bytes = StorageRecord::NamespaceMapping(mapping.clone()).encode().unwrap(); + assert!(matches!( + store + .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) + .await + .unwrap(), + CasOutcome::Applied(_) + )); + entries.push((key, bytes, mapping)); + } + let mut scan = ChildScan { + catalog, + parent, + scope: CatalogScope::NamespaceName, + limit: 1, + continuation: None, + }; + let mut keys = Vec::new(); + loop { + let page = store.scan_children(scan.clone()).await.unwrap(); + assert!(page.items.len() <= 1); + keys.extend(page.items.into_iter().map(|item| item.key)); + scan.continuation = page.continuation; + if scan.continuation.is_none() { + break; + } + assert!(keys.len() <= entries.len()); + } + assert_eq!( + keys, + entries.iter().map(|(key, _, _)| key.clone()).collect::>() + ); + assert!(store + .scan_children(ChildScan { + scope: CatalogScope::TableName, + ..scan + }) + .await + .unwrap() + .items + .is_empty()); + let (key, bytes, mapping) = entries.remove(0); + verify_conditional_recreation(store.as_ref(), key, bytes, mapping).await; +} + +async fn verify_conditional_recreation( + store: &dyn NamespaceStore, + key: Vec, + bytes: Vec, + mut mapping: NamespaceMapping, +) { + let mut mismatched = mapping.clone(); + mismatched.operation = OperationId::random(); + let wrong_bytes = StorageRecord::NamespaceMapping(mismatched).encode().unwrap(); + let wrong_identity = mutation_identity(&key, Some(&wrong_bytes), &[]); + assert!(matches!( + store + .delete_mapping(&key, &wrong_bytes, wrong_identity) + .await + .unwrap(), + CasOutcome::Conflict(Some(_)) + )); + let identity = mutation_identity(&key, Some(&bytes), &[]); + assert!(matches!( + store.delete_mapping(&key, &bytes, identity).await.unwrap(), + CasOutcome::Applied(_) + )); + assert!(store.get(&key).await.unwrap().is_none()); + mapping.namespace = NamespaceId::random(); + mapping.operation = OperationId::random(); + let replacement = StorageRecord::NamespaceMapping(mapping).encode().unwrap(); + assert!(matches!( + store + .compare_exchange( + &key, + None, + &replacement, + mutation_identity(&key, None, &replacement) + ) + .await + .unwrap(), + CasOutcome::Applied(_) + )); + store.delete_mapping(&key, &bytes, identity).await.unwrap(); + assert_eq!(store.get(&key).await.unwrap().unwrap().bytes, replacement); +} diff --git a/app/crowdb-access-server/tests/common/iceberg_namespace_client.py b/app/crowdb-access-server/tests/common/iceberg_namespace_client.py new file mode 100644 index 000000000..725a6993f --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_namespace_client.py @@ -0,0 +1,44 @@ +import sys +from concurrent.futures import ThreadPoolExecutor +from threading import Barrier + +from pyiceberg.catalog import load_catalog +from pyiceberg.exceptions import ServiceUnavailableError + + +def main(): + endpoint, mode, count = sys.argv[1:] + catalog = load_catalog("crowdb", type="rest", uri=endpoint, token="r" * 32) + if mode == "concurrency": + barrier = Barrier(5, timeout=10) + + def list_once(index): + reader = load_catalog(f"reader-{index}", type="rest", uri=endpoint, token="r" * 32) + barrier.wait() + try: + assert reader.list_namespaces() == [("analytics",)] + return 200 + except ServiceUnavailableError: + return 503 + + with ThreadPoolExecutor(max_workers=5) as executor: + results = list(executor.map(list_once, range(5))) + assert sorted(results) == [200, 200, 200, 200, 503] + elif mode == "overflow": + for attempt in range(5): + try: + catalog.list_namespaces() + except ServiceUnavailableError as error: + assert "ServiceUnavailableException" in str(error) + else: + raise AssertionError(f"exhausted complete list returned success on request {attempt}") + else: + namespaces = catalog.list_namespaces() + assert len(namespaces) == int(count) + assert len(set(namespaces)) == len(namespaces) + assert ("analytics",) in namespaces + print(f"PyIceberg namespace {mode} passed") + + +if __name__ == "__main__": + main() diff --git a/app/crowdb-access-server/tests/common/iceberg_process.rs b/app/crowdb-access-server/tests/common/iceberg_process.rs new file mode 100644 index 000000000..54951fdcb --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_process.rs @@ -0,0 +1,101 @@ +use std::net::{SocketAddr, TcpListener}; +use std::process::{Child, Command, Stdio}; +use std::time::Duration; + +pub struct TestIcebergProcess { + child: Child, + pub address: SocketAddr, +} + +impl TestIcebergProcess { + pub async fn start(seeds: &[String]) -> Self { + Self::start_with_gc(seeds, false).await + } + + pub async fn start_with_gc(seeds: &[String], gc_enabled: bool) -> Self { + Self::start_with_gc_settings(seeds, gc_enabled, &[]).await + } + + pub async fn start_with_gc_settings( + seeds: &[String], + gc_enabled: bool, + settings: &[(&str, &str)], + ) -> Self { + let reservation = TcpListener::bind("127.0.0.1:0").unwrap(); + let address = reservation.local_addr().unwrap(); + drop(reservation); + let mut launch = command(seeds); + launch + .env("CROWDB_ICEBERG_LISTEN", address.to_string()) + .env("CROWDB_ICEBERG_GC_ENABLED", if gc_enabled { "1" } else { "0" }) + .env("CROWDB_ICEBERG_GC_INTERVAL_MS", "100"); + for (name, value) in settings { + launch.env(name, value); + } + let child = launch + .arg("serve") + .stdout(Stdio::inherit()) + .stderr(Stdio::inherit()) + .spawn() + .unwrap(); + let mut process = Self { child, address }; + tokio::time::timeout(Duration::from_secs(30), async { + loop { + assert!( + process.child.try_wait().unwrap().is_none(), + "Iceberg listener exited" + ); + if tokio::net::TcpStream::connect(address).await.is_ok() { + break; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .unwrap(); + process + } + + pub fn check_official_client(&self) { + self.check_client(false); + } + + pub fn check_official_reads(&self) { + self.check_client(true); + } + + fn check_client(&self, read_only: bool) { + let python = std::env::var_os("CROWDB_ICEBERG_E2E_PYTHON") + .expect("run pixi run -e iceberg-e2e test-pyiceberg-e2e"); + let mut command = Command::new(python); + command + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_client.py" + )) + .arg(format!("http://{}", self.address)); + if read_only { + command.arg("--read-only"); + } + let status = command.status().unwrap(); + assert!(status.success(), "official Iceberg client contract failed"); + } +} + +pub fn command(seeds: &[String]) -> Command { + let mut command = Command::new(env!("CARGO_BIN_EXE_crowdb-iceberg")); + command + .env("CROWDB_MANAGEMENT_SEEDS", seeds.join(",")) + .env("CROWDB_ICEBERG_READ_TOKEN", "r".repeat(32)) + .env("CROWDB_ICEBERG_WRITE_TOKEN", "w".repeat(32)) + .env("CROWDB_ICEBERG_MANAGE_TOKEN", "m".repeat(32)) + .env("CROWDB_ICEBERG_CLEAR_TOKEN", "c".repeat(32)); + command +} + +impl Drop for TestIcebergProcess { + fn drop(&mut self) { + let _ = self.child.kill(); + let _ = self.child.wait(); + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_property.rs b/app/crowdb-access-server/tests/common/iceberg_property.rs new file mode 100644 index 000000000..db4b06154 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_property.rs @@ -0,0 +1,117 @@ +use std::collections::BTreeMap; +use std::sync::{atomic::AtomicU8, Arc}; + +use crowdb_access_iceberg::catalog::{CatalogContext, CatalogStore, RoutedCatalogStore}; +use crowdb_access_iceberg::key::{NamespaceId, OperationId}; +use crowdb_access_iceberg::namespace::{ + authority_key, name_key, NamespaceAuthority, NamespaceIdentifier, NamespaceJournal, NamespaceLifecycle, + NamespaceMapping, NamespaceMappingState, NamespacePhase, NamespaceProperties, NamespacePropertyRequest, + NamespaceRepository, PropertyChanges, +}; +use crowdb_access_iceberg::operation::{mutation_identity, PayloadStore, RequestIdentity}; +use crowdb_access_iceberg::record::StorageRecord; + +use super::{common::now_ms, fault::TestFaultStore}; + +pub async fn prepare(store: Arc, context: CatalogContext) -> NamespacePropertyRequest { + let request = NamespacePropertyRequest { + context, + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "writer".into(), + identifier: NamespaceIdentifier::new(vec!["property-recovery".into()]).unwrap(), + changes: PropertyChanges { + removals: Vec::new(), + updates: BTreeMap::from([("owner".into(), "survives-restart".into())]), + }, + }; + let authority = NamespaceAuthority { + catalog: context.catalog, + namespace: NamespaceId::random(), + parent: None, + identifier: request.identifier.clone(), + name_epoch: 1, + property_revision: 1, + admission_fence: 1, + mutation_revision: 1, + lifecycle: NamespaceLifecycle::Ready, + pending_operation: None, + properties: NamespaceProperties::default(), + }; + let mapping = NamespaceMapping { + catalog: context.catalog, + parent: None, + name: request.identifier.name().into(), + namespace: authority.namespace, + name_epoch: 1, + operation: OperationId::random(), + state: NamespaceMappingState::Published, + }; + for (key, record) in [ + ( + authority_key(context.catalog, authority.namespace), + StorageRecord::NamespaceAuthority(Box::new(authority)), + ), + ( + name_key(context.catalog, None, request.identifier.name()).unwrap(), + StorageRecord::NamespaceMapping(mapping), + ), + ] { + let key = key.encode().unwrap(); + let bytes = record.encode().unwrap(); + store + .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) + .await + .unwrap(); + } + let fault = Arc::new(TestFaultStore { + inner: store.clone(), + mode: AtomicU8::new(3), + }); + assert!(NamespaceRepository::new(fault) + .update_properties(&request) + .await + .is_err()); + assert_eq!( + NamespaceJournal::new(store) + .load(context, request.identity.operation) + .await + .unwrap() + .unwrap() + .phase, + NamespacePhase::Publishing + ); + request +} + +pub async fn verify(store: Arc, request: &NamespacePropertyRequest) { + let repository = NamespaceRepository::new(store.clone()); + let outcome = repository.update_properties(request).await.unwrap().unwrap(); + assert_eq!(outcome.status, 200); + let body = PayloadStore::new(store).get(&outcome.body).await.unwrap(); + assert_eq!( + serde_json::from_slice::(&body).unwrap(), + serde_json::json!({ + "removed": [], "updated": ["owner"], "missing": [], + }) + ); + assert_eq!( + repository.update_properties(request).await.unwrap(), + Some(outcome) + ); + let authority = repository + .load(request.context, &request.identifier) + .await + .unwrap() + .unwrap(); + assert_eq!(authority.property_revision, 2); + assert_eq!(authority.name_epoch, 1); + assert_eq!(authority.admission_fence, 1); + assert_eq!(authority.pending_operation, None); + assert_eq!( + authority.properties.entries().get("owner").unwrap(), + "survives-restart" + ); +} diff --git a/app/crowdb-access-server/tests/common/iceberg_response_loss.rs b/app/crowdb-access-server/tests/common/iceberg_response_loss.rs new file mode 100644 index 000000000..23c0302c4 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_response_loss.rs @@ -0,0 +1,105 @@ +use std::sync::{ + atomic::{AtomicU8, Ordering}, + Arc, +}; + +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +pub struct TestResponseLossProxy { + pub origin: String, + dropped: Arc, + task: tokio::task::JoinHandle<()>, +} + +impl TestResponseLossProxy { + pub async fn start(backend_origin: String, path: &'static str) -> Self { + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let origin = format!("http://{}", listener.local_addr().unwrap()); + let dropped = Arc::new(AtomicU8::new(0)); + let observed = dropped.clone(); + let target = format!("POST {path} "); + let task = tokio::spawn(async move { + loop { + let (client, _) = listener.accept().await.unwrap(); + let backend = backend_origin.clone(); + let target = target.clone(); + let dropped = dropped.clone(); + tokio::spawn(async move { + forward_or_lose(client, &backend, target.as_bytes(), dropped) + .await + .unwrap(); + }); + } + }); + Self { + origin, + dropped: observed, + task, + } + } + + pub fn assert_dropped(&self) { + assert_eq!( + self.dropped.load(Ordering::SeqCst), + 2, + "proxy did not drop a successful create response" + ); + } +} + +impl Drop for TestResponseLossProxy { + fn drop(&mut self) { + self.task.abort(); + } +} + +async fn forward_or_lose( + mut client: tokio::net::TcpStream, + backend_origin: &str, + target: &[u8], + dropped: Arc, +) -> std::io::Result<()> { + let mut header = Vec::new(); + while !header.windows(4).any(|window| window == b"\r\n\r\n") { + let mut buffer = [0_u8; 4096]; + let count = client.read(&mut buffer).await?; + if count == 0 || header.len() + count > 16 * 1024 { + return Err(std::io::Error::other("invalid proxy request header")); + } + header.extend_from_slice(&buffer[..count]); + } + let backend = backend_origin.trim_start_matches("http://"); + let mut upstream = tokio::net::TcpStream::connect(backend).await?; + upstream.write_all(&header).await?; + if header.starts_with(target) + && dropped + .compare_exchange(0, 1, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + { + let (mut client_read, client_write) = client.into_split(); + let (mut upstream_read, mut upstream_write) = upstream.into_split(); + let forwarding = tokio::spawn(async move { + let _ = tokio::io::copy(&mut client_read, &mut upstream_write).await; + }); + let mut response = [0_u8; 4096]; + let count = upstream_read.read(&mut response).await?; + forwarding.abort(); + drop(client_write); + if count == 0 || !response.starts_with(b"HTTP/1.1 200") { + return Err(std::io::Error::other( + "upstream did not publish the create response", + )); + } + dropped.store(2, Ordering::SeqCst); + return Ok(()); + } + if let Err(error) = tokio::io::copy_bidirectional(&mut client, &mut upstream).await { + if !matches!( + error.kind(), + std::io::ErrorKind::ConnectionReset | std::io::ErrorKind::BrokenPipe + ) { + return Err(error); + } + } + Ok(()) +} diff --git a/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.lock b/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.lock new file mode 100644 index 000000000..cb5de80ee --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.lock @@ -0,0 +1,3126 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "adler2" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" + +[[package]] +name = "aead" +version = "0.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d122413f284cf2d62fb1b7db97e02edb8cda96d769b16e443a4f6195e35662b0" +dependencies = [ + "crypto-common", + "generic-array", +] + +[[package]] +name = "aes" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b169f7a6d4742236a0a00c541b845991d0ac43e546831af1249753ab4c3aa3a0" +dependencies = [ + "cfg-if", + "cipher", + "cpufeatures", +] + +[[package]] +name = "aes-gcm" +version = "0.10.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "831010a0f742e1209b3bcea8fab6a8e149051ba6099432c8cb2cc117dec3ead1" +dependencies = [ + "aead", + "aes", + "cipher", + "ctr", + "ghash", + "subtle", +] + +[[package]] +name = "ahash" +version = "0.8.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a15f179cd60c4584b8a8c596927aadc462e27f2ca70c04e0071964a73ba7a75" +dependencies = [ + "cfg-if", + "const-random", + "getrandom 0.3.4", + "once_cell", + "version_check", + "zerocopy", +] + +[[package]] +name = "aho-corasick" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" +dependencies = [ + "memchr", +] + +[[package]] +name = "alloc-no-stdlib" +version = "2.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cc7bb162ec39d46ab1ca8c77bf72e890535becd1751bb45f64c597edb4c8c6b3" + +[[package]] +name = "alloc-stdlib" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0e76a019e91224d279006ff972f1e984179a6e9feb050adba6ce8274aef23195" +dependencies = [ + "alloc-no-stdlib", +] + +[[package]] +name = "android_system_properties" +version = "0.1.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" +dependencies = [ + "libc", +] + +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + +[[package]] +name = "apache-avro" +version = "0.21.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "36fa98bc79671c7981272d91a8753a928ff6a1cd8e4f20a44c45bd5d313840bf" +dependencies = [ + "bigdecimal", + "bon", + "crc32fast", + "digest", + "log", + "miniz_oxide 0.8.9", + "num-bigint", + "quad-rand", + "rand", + "regex-lite", + "serde", + "serde_bytes", + "serde_json", + "snap", + "strum", + "strum_macros", + "thiserror", + "uuid", + "zstd", +] + +[[package]] +name = "array-init" +version = "2.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3d62b7694a562cdf5a74227903507c56ab2cc8bdd1f781ed5cb4cf9c9f810bfc" + +[[package]] +name = "arrow-arith" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0a41203398f0eaa6f7ec8e62c0da742a21abf282c148fc157f6c35c90e29981a" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "chrono", + "num-traits", +] + +[[package]] +name = "arrow-array" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae33dad492b7df00a217563a7b0ef2874df68a0deea1b1a3acf628152f7f7a69" +dependencies = [ + "ahash", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "chrono", + "half", + "hashbrown 0.17.1", + "num-complex", + "num-integer", + "num-traits", +] + +[[package]] +name = "arrow-buffer" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9552f96391c005e6ab449fa941420935e7e062489b12b8b1b08879b2163f5b5" +dependencies = [ + "bytes", + "half", + "num-bigint", + "num-traits", +] + +[[package]] +name = "arrow-cast" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a8a327c9649f30d8406995f27642b68df354713cca3baaaf100f076f18d5f34" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-ord", + "arrow-schema", + "arrow-select", + "atoi", + "base64 0.22.1", + "chrono", + "half", + "lexical-core", + "num-traits", + "ryu", +] + +[[package]] +name = "arrow-data" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2b24852db04738907e06c04ea61e42fe7fda962a34513022dc0d0e754fb7976b" +dependencies = [ + "arrow-buffer", + "arrow-schema", + "half", + "num-integer", + "num-traits", +] + +[[package]] +name = "arrow-ipc" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29a908a11fcfb3fb2f6730f4ac15e367bc644e419155e96238f68cf3adde572b" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "arrow-select", + "flatbuffers", +] + +[[package]] +name = "arrow-ord" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63a083ec750f5c043f02946b4baf05fcdbb55f4560a3277055caca5cc99f3eb0" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "arrow-select", +] + +[[package]] +name = "arrow-schema" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "21ca356ad6425cecb6eb7b28e4f659f1ee7880fbb1a16127de7dd62901efee9e" + +[[package]] +name = "arrow-select" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c58da39eb3d8350ad4a549e5c2bc49284dac554016c69829310350f1731b0aad" +dependencies = [ + "ahash", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "num-traits", +] + +[[package]] +name = "arrow-string" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6789b388467525e3271326b6b4915666ecfdf5142aef09779445c954b67543c" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "arrow-select", + "memchr", + "num-traits", + "regex", + "regex-syntax", +] + +[[package]] +name = "as-any" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b0f477b951e452a0b6b4a10b53ccd569042d1d01729b519e02074a9c0958a063" + +[[package]] +name = "async-lock" +version = "3.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "290f7f2596bd5b78a9fec8088ccd89180d7f9f55b94b0576823bbbdc72ee8311" +dependencies = [ + "event-listener", + "event-listener-strategy", + "pin-project-lite", +] + +[[package]] +name = "async-trait" +version = "0.1.92" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "atoi" +version = "2.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f28d99ec8bfea296261ca1af174f24225171fea9664ba9003cbebee704810528" +dependencies = [ + "num-traits", +] + +[[package]] +name = "atomic-waker" +version = "1.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" + +[[package]] +name = "autocfg" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" + +[[package]] +name = "backon" +version = "1.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cffb0e931875b666fc4fcb20fee52e9bbd1ef836fd9e9e04ec21555f9f85f7ef" +dependencies = [ + "fastrand", + "gloo-timers", + "tokio", +] + +[[package]] +name = "base64" +version = "0.22.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" + +[[package]] +name = "base64" +version = "0.23.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" + +[[package]] +name = "bigdecimal" +version = "0.4.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4d6867f1565b3aad85681f1015055b087fcfd840d6aeee6eee7f2da317603695" +dependencies = [ + "autocfg", + "libm", + "num-bigint", + "num-integer", + "num-traits", + "serde", +] + +[[package]] +name = "bimap" +version = "0.6.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "230c5f1ca6a325a32553f8640d31ac9b49f2411e901e427570154868b46da4f7" + +[[package]] +name = "bitflags" +version = "1.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" + +[[package]] +name = "bitflags" +version = "2.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + +[[package]] +name = "bnum" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f781dba93de3a5ef6dc5b17c9958b208f6f3f021623b360fb605ea51ce443f10" +dependencies = [ + "serde", + "serde-big-array", +] + +[[package]] +name = "bon" +version = "3.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "60eafe0d77c3a2fc292c1d1346c3041b33c0a108085a2afabf672b70f69dbbc9" +dependencies = [ + "bon-macros", +] + +[[package]] +name = "bon-macros" +version = "3.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bd0f9631d8aaaee112c41985d675ef269e02acbd4f33122836af4f0c5f699ff6" +dependencies = [ + "darling 0.24.1", + "ident_case", + "prettyplease", + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "brotli" +version = "8.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5cc91aac060a7a1e25823bdccbfb6af1875b88f17c6daac97894eed8207166b3" +dependencies = [ + "alloc-no-stdlib", + "alloc-stdlib", + "brotli-decompressor", +] + +[[package]] +name = "brotli-decompressor" +version = "5.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a32acac15fe1967bc3986b2a6347dffc965602354ea6f450ad07e8bfd253583" +dependencies = [ + "alloc-no-stdlib", + "alloc-stdlib", +] + +[[package]] +name = "bs58" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf88ba1141d185c399bee5288d850d63b8369520c1eafc32a0430b5b6c287bf4" +dependencies = [ + "tinyvec", +] + +[[package]] +name = "bumpalo" +version = "3.20.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" + +[[package]] +name = "bytemuck" +version = "1.25.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" + +[[package]] +name = "byteorder" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" + +[[package]] +name = "bytes" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" + +[[package]] +name = "cc" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040" +dependencies = [ + "find-msvc-tools", + "jobserver", + "libc", + "shlex", +] + +[[package]] +name = "cfg-if" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" + +[[package]] +name = "chrono" +version = "0.4.45" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" +dependencies = [ + "iana-time-zone", + "js-sys", + "num-traits", + "serde", + "wasm-bindgen", + "windows-link", +] + +[[package]] +name = "cipher" +version = "0.4.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad" +dependencies = [ + "crypto-common", + "inout", +] + +[[package]] +name = "const-random" +version = "0.1.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "87e00182fe74b066627d63b85fd550ac2998d4b0bd86bfed477a0ae4c7c71359" +dependencies = [ + "const-random-macro", +] + +[[package]] +name = "const-random-macro" +version = "0.1.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9d839f2a20b0aee515dc581a6172f2321f96cab76c1a38a4c584a194955390e" +dependencies = [ + "getrandom 0.2.17", + "once_cell", + "tiny-keccak", +] + +[[package]] +name = "core-foundation-sys" +version = "0.8.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" + +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "crc32fast" +version = "1.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "crossbeam-channel" +version = "0.5.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "98b0cc327b5bc766e7fda9c9260cc0fa81b43a8e240440422dff70788e3f9ef1" +dependencies = [ + "crossbeam-utils", +] + +[[package]] +name = "crossbeam-epoch" +version = "0.9.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d" +dependencies = [ + "crossbeam-utils", +] + +[[package]] +name = "crossbeam-utils" +version = "0.8.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6" + +[[package]] +name = "crowdb-iceberg-rust-client-fixture" +version = "0.1.0-dev" +dependencies = [ + "iceberg", + "iceberg-catalog-rest", + "tokio", +] + +[[package]] +name = "crunchy" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5" + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "rand_core 0.6.4", + "typenum", +] + +[[package]] +name = "ctr" +version = "0.9.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0369ee1ad671834580515889b80f2ea915f23b8be8d0daa4bbaf2ac5c7590835" +dependencies = [ + "cipher", +] + +[[package]] +name = "darling" +version = "0.20.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc7f46116c46ff9ab3eb1597a45688b6715c6e628b5c133e288e709a29bcb4ee" +dependencies = [ + "darling_core 0.20.11", + "darling_macro 0.20.11", +] + +[[package]] +name = "darling" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed17f5901b6630b993ca003def43f2f8ef4014fc13b047b57aad617ff32bc2ec" +dependencies = [ + "darling_core 0.24.1", + "darling_macro 0.24.1", +] + +[[package]] +name = "darling_core" +version = "0.20.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0d00b9596d185e565c2207a0b01f8bd1a135483d02d9b7b0a54b11da8d53412e" +dependencies = [ + "fnv", + "ident_case", + "proc-macro2", + "quote", + "strsim", + "syn 2.0.119", +] + +[[package]] +name = "darling_core" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6837e2cf7485aaae18f86181d2f0e9a7ed297a025e220aeabf63fdebd3a2ddff" +dependencies = [ + "ident_case", + "proc-macro2", + "quote", + "strsim", + "syn 3.0.6", +] + +[[package]] +name = "darling_macro" +version = "0.20.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc34b93ccb385b40dc71c6fceac4b2ad23662c7eeb248cf10d529b7e055b6ead" +dependencies = [ + "darling_core 0.20.11", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "darling_macro" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2ac7135c3ef02b2f7833bbeb1be5ba7f966dcde8a87c6b87f65a778d71a02785" +dependencies = [ + "darling_core 0.24.1", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "defmt" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e2953bfe4f93bbd20cc71198842756f77d161884c99ebbabc41d80231ded88d1" +dependencies = [ + "bitflags 1.3.2", + "defmt-macros", +] + +[[package]] +name = "defmt-macros" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bad9c72e7ca2137e0dc3813245a0d282fd6daad32fd800af018306a9169b5fe8" +dependencies = [ + "defmt-parser", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "defmt-parser" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" +dependencies = [ + "thiserror", +] + +[[package]] +name = "deranged" +version = "0.5.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" +dependencies = [ + "serde_core", +] + +[[package]] +name = "derive_builder" +version = "0.20.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "507dfb09ea8b7fa618fcf76e953f4f5e192547945816d5358edffe39f6f94947" +dependencies = [ + "derive_builder_macro", +] + +[[package]] +name = "derive_builder_core" +version = "0.20.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2d5bcf7b024d6835cfb3d473887cd966994907effbe9227e8c8219824d06c4e8" +dependencies = [ + "darling 0.20.11", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "derive_builder_macro" +version = "0.20.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ab63b0e2bf4d5928aff72e83a7dace85d7bba5fe12dcc3c5a572d78caffd3f3c" +dependencies = [ + "derive_builder_core", + "syn 2.0.119", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer", + "crypto-common", +] + +[[package]] +name = "displaydoc" +version = "0.2.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "dissimilar" +version = "1.0.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aeda16ab4059c5fd2a83f2b9c9e9c981327b18aa8e3b313f7e6563799d4f093e" + +[[package]] +name = "dyn-clone" +version = "1.0.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555" + +[[package]] +name = "either" +version = "1.18.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" + +[[package]] +name = "equivalent" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" + +[[package]] +name = "erased-serde" +version = "0.4.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d2add8a07dd6a8d93ff627029c51de145e12686fbc36ecb298ac22e74cf02dec" +dependencies = [ + "serde", + "serde_core", + "typeid", +] + +[[package]] +name = "event-listener" +version = "5.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" +dependencies = [ + "parking", + "pin-project-lite", +] + +[[package]] +name = "event-listener-strategy" +version = "0.5.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8be9f3dfaaffdae2972880079a491a1a8bb7cbed0b8dd7a347f668b4150a3b93" +dependencies = [ + "event-listener", + "pin-project-lite", +] + +[[package]] +name = "expect-test" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63af43ff4431e848fb47472a920f14fa71c24de13255a5692e93d4e90302acb0" +dependencies = [ + "dissimilar", + "once_cell", +] + +[[package]] +name = "fastnum" +version = "0.7.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "020d1b59a944bc239d79903fbac2eda2365138b44890c27979562f6592059dcd" +dependencies = [ + "bnum", + "num-integer", + "num-traits", + "serde", +] + +[[package]] +name = "fastrand" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" + +[[package]] +name = "find-msvc-tools" +version = "0.1.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484" + +[[package]] +name = "flatbuffers" +version = "25.12.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "35f6839d7b3b98adde531effaf34f0c2badc6f4735d26fe74709d8e513a96ef3" +dependencies = [ + "bitflags 2.13.2", + "rustc_version", +] + +[[package]] +name = "flate2" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" +dependencies = [ + "crc32fast", + "miniz_oxide 0.9.1", + "zlib-rs", +] + +[[package]] +name = "fnv" +version = "1.0.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" + +[[package]] +name = "form_urlencoded" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" +dependencies = [ + "percent-encoding", +] + +[[package]] +name = "futures" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9a31d2a3fbaaeb2af2368bbdd904aa8e812d3c04a1ee10d3171f52d556e5d0a3" +dependencies = [ + "futures-channel", + "futures-core", + "futures-executor", + "futures-io", + "futures-sink", + "futures-task", + "futures-util", +] + +[[package]] +name = "futures-channel" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" +dependencies = [ + "futures-core", + "futures-sink", +] + +[[package]] +name = "futures-core" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" + +[[package]] +name = "futures-executor" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "031b47cf1a3c6cc8bc2fc76cd437f521619387907d469316e7c0bc278f1f5432" +dependencies = [ + "futures-core", + "futures-task", + "futures-util", +] + +[[package]] +name = "futures-io" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" + +[[package]] +name = "futures-macro" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "futures-sink" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" + +[[package]] +name = "futures-task" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" + +[[package]] +name = "futures-util" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" +dependencies = [ + "futures-channel", + "futures-core", + "futures-io", + "futures-macro", + "futures-sink", + "futures-task", + "memchr", + "pin-project-lite", + "slab", +] + +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + +[[package]] +name = "getrandom" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" +dependencies = [ + "cfg-if", + "js-sys", + "libc", + "wasi", + "wasm-bindgen", +] + +[[package]] +name = "getrandom" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" +dependencies = [ + "cfg-if", + "libc", + "r-efi 5.3.0", + "wasip2", +] + +[[package]] +name = "getrandom" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" +dependencies = [ + "cfg-if", + "libc", + "r-efi 6.0.0", +] + +[[package]] +name = "ghash" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0d8a4362ccb29cb0b265253fb0a2728f592895ee6854fd9bc13f2ffda266ff1" +dependencies = [ + "opaque-debug", + "polyval", +] + +[[package]] +name = "gloo-timers" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbb143cf96099802033e0d4f4963b19fd2e0b728bcf076cd9cf7f6634f092994" +dependencies = [ + "futures-channel", + "futures-core", + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "half" +version = "2.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ea2d84b969582b4b1864a92dc5d27cd2b77b622a8d79306834f1be5ba20d84b" +dependencies = [ + "cfg-if", + "crunchy", + "num-traits", + "zerocopy", +] + +[[package]] +name = "hashbrown" +version = "0.12.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8a9ee70c43aaf417c914396645a0fa852624801b24ebb7ae78fe8272889ac888" + +[[package]] +name = "hashbrown" +version = "0.17.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" + +[[package]] +name = "heck" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" + +[[package]] +name = "hex" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" + +[[package]] +name = "http" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" +dependencies = [ + "bytes", + "itoa", +] + +[[package]] +name = "http-body" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" +dependencies = [ + "bytes", + "http", +] + +[[package]] +name = "http-body-util" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" +dependencies = [ + "bytes", + "futures-core", + "http", + "http-body", + "pin-project-lite", +] + +[[package]] +name = "httparse" +version = "1.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" + +[[package]] +name = "hyper" +version = "1.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "27b501faa50e7a26c3d3560ca625132f4078a17771f4810baf70475ae48cbe43" +dependencies = [ + "atomic-waker", + "bytes", + "futures-channel", + "futures-core", + "http", + "http-body", + "httparse", + "itoa", + "pin-project-lite", + "smallvec", + "tokio", + "want", +] + +[[package]] +name = "hyper-util" +version = "0.1.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ddc03d96684f9226b8a787cdb71488417b53ab5ea8fdb1dac946cb9431cc8bff" +dependencies = [ + "base64 0.23.1", + "bytes", + "futures-channel", + "futures-util", + "http", + "http-body", + "httparse", + "hyper", + "ipnet", + "libc", + "percent-encoding", + "pin-project-lite", + "socket2", + "tokio", + "tower-service", + "tracing", +] + +[[package]] +name = "iana-time-zone" +version = "0.1.65" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" +dependencies = [ + "android_system_properties", + "core-foundation-sys", + "iana-time-zone-haiku", + "js-sys", + "log", + "wasm-bindgen", + "windows-core", +] + +[[package]] +name = "iana-time-zone-haiku" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" +dependencies = [ + "cc", +] + +[[package]] +name = "iceberg" +version = "0.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e4ed93c6be93e47ba2d8928b8cefe972932668940478f6be93eca6ade52c4e8a" +dependencies = [ + "aes-gcm", + "anyhow", + "apache-avro", + "array-init", + "arrow-arith", + "arrow-array", + "arrow-buffer", + "arrow-cast", + "arrow-ord", + "arrow-schema", + "arrow-select", + "arrow-string", + "as-any", + "async-trait", + "backon", + "base64 0.22.1", + "bimap", + "bytes", + "chrono", + "derive_builder", + "expect-test", + "fastnum", + "flate2", + "fnv", + "futures", + "itertools", + "moka", + "murmur3", + "once_cell", + "ordered-float 4.6.0", + "parquet", + "rand", + "reqwest", + "roaring", + "serde", + "serde_bytes", + "serde_derive", + "serde_json", + "serde_repr", + "serde_with", + "strum", + "tokio", + "tracing", + "typed-builder", + "typetag", + "url", + "uuid", + "zeroize", + "zstd", +] + +[[package]] +name = "iceberg-catalog-rest" +version = "0.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3e1b94fd46e309f0a728c32f9f7005a230937d4c8b039a85258074ec40c45614" +dependencies = [ + "async-trait", + "chrono", + "http", + "iceberg", + "itertools", + "reqwest", + "serde", + "serde_derive", + "serde_json", + "tokio", + "typed-builder", + "uuid", +] + +[[package]] +name = "icu_collections" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" +dependencies = [ + "displaydoc", + "potential_utf", + "utf8_iter", + "yoke", + "zerofrom", + "zerovec", +] + +[[package]] +name = "icu_locale_core" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" +dependencies = [ + "displaydoc", + "litemap", + "tinystr", + "writeable", + "zerovec", +] + +[[package]] +name = "icu_normalizer" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" +dependencies = [ + "icu_collections", + "icu_normalizer_data", + "icu_properties", + "icu_provider", + "smallvec", + "zerovec", +] + +[[package]] +name = "icu_normalizer_data" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" + +[[package]] +name = "icu_properties" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" +dependencies = [ + "displaydoc", + "icu_collections", + "icu_locale_core", + "icu_properties_data", + "icu_provider", + "zerotrie", + "zerovec", +] + +[[package]] +name = "icu_properties_data" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" + +[[package]] +name = "icu_provider" +version = "2.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" +dependencies = [ + "displaydoc", + "icu_locale_core", + "writeable", + "yoke", + "zerofrom", + "zerotrie", + "zerovec", +] + +[[package]] +name = "ident_case" +version = "1.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39" + +[[package]] +name = "idna" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3b0875f23caa03898994f6ddc501886a45c7d3d62d04d2d90788d47be1b1e4de" +dependencies = [ + "idna_adapter", + "smallvec", + "utf8_iter", +] + +[[package]] +name = "idna_adapter" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb68373c0d6620ef8105e855e7745e18b0d00d3bdb07fb532e434244cdb9a714" +dependencies = [ + "icu_normalizer", + "icu_properties", +] + +[[package]] +name = "indexmap" +version = "1.9.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bd070e393353796e801d209ad339e89596eb4c8d430d18ede6a1cced8fafbd99" +dependencies = [ + "autocfg", + "hashbrown 0.12.3", + "serde", +] + +[[package]] +name = "indexmap" +version = "2.14.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" +dependencies = [ + "equivalent", + "hashbrown 0.17.1", + "serde", + "serde_core", +] + +[[package]] +name = "inout" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "879f10e63c20629ecabbb64a8010319738c66a5cd0c29b02d63d272b03751d01" +dependencies = [ + "generic-array", +] + +[[package]] +name = "integer-encoding" +version = "3.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8bb03732005da905c88227371639bf1ad885cc712789c011c31c5fb3ab3ccf02" + +[[package]] +name = "inventory" +version = "0.3.24" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a4f0c30c76f2f4ccee3fe55a2435f691ca00c0e4bd87abe4f4a851b1d4dac39b" +dependencies = [ + "rustversion", +] + +[[package]] +name = "ipnet" +version = "2.12.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "791930b43c0d5973160d90a8f3894509f2b273430f5c5c73b668636d0287c5c0" + +[[package]] +name = "itertools" +version = "0.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "413ee7dfc52ee1a4949ceeb7dbc8a33f2d6c088194d9f922fb8318faf1f01186" +dependencies = [ + "either", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "jiff" +version = "0.2.37" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ab1baf72f08796de0260609515130699b890ac25f30e610ad894bc5856cafdb" +dependencies = [ + "defmt", + "jiff-core", + "jiff-static", + "jiff-tzdb-platform", + "log", + "portable-atomic", + "portable-atomic-util", + "serde_core", + "windows-link", +] + +[[package]] +name = "jiff-core" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5e52fe76043ccecc9005d2305ebaadf7d7fc0cc89ca6baa10a94d6bc68c7128c" +dependencies = [ + "defmt", + "log", +] + +[[package]] +name = "jiff-static" +version = "0.2.37" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "378268a1116ad67ae6228701118ac9f491d78fda38a40a1f1a9e1348de6f7212" +dependencies = [ + "jiff-core", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "jiff-tzdb" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "142bd39932ad231f10513df9ab62661fead8719872150b7ad02a2df79f4e141e" + +[[package]] +name = "jiff-tzdb-platform" +version = "0.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "875a5a69ac2bab1a891711cf5eccbec1ce0341ea805560dcd90b7a2e925132e8" +dependencies = [ + "jiff-tzdb", +] + +[[package]] +name = "jobserver" +version = "0.1.35" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3" +dependencies = [ + "getrandom 0.4.3", + "libc", +] + +[[package]] +name = "js-sys" +version = "0.3.106" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7883d941dae510fb2d978fc3fe018c71c9e2892fd38854de3e8b92c2e5ad9cc5" +dependencies = [ + "cfg-if", + "futures-util", + "wasm-bindgen", +] + +[[package]] +name = "lexical-core" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7d8d125a277f807e55a77304455eb7b1cb52f2b18c143b60e766c120bd64a594" +dependencies = [ + "lexical-parse-float", + "lexical-parse-integer", + "lexical-util", + "lexical-write-float", + "lexical-write-integer", +] + +[[package]] +name = "lexical-parse-float" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52a9f232fbd6f550bc0137dcb5f99ab674071ac2d690ac69704593cb4abbea56" +dependencies = [ + "lexical-parse-integer", + "lexical-util", +] + +[[package]] +name = "lexical-parse-integer" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9a7a039f8fb9c19c996cd7b2fcce303c1b2874fe1aca544edc85c4a5f8489b34" +dependencies = [ + "lexical-util", +] + +[[package]] +name = "lexical-util" +version = "1.0.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2604dd126bb14f13fb5d1bd6a66155079cb9fa655b37f875b3a742c705dbed17" + +[[package]] +name = "lexical-write-float" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "50c438c87c013188d415fbabbb1dceb44249ab81664efbd31b14ae55dabb6361" +dependencies = [ + "lexical-util", + "lexical-write-integer", +] + +[[package]] +name = "lexical-write-integer" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "409851a618475d2d5796377cad353802345cba92c867d9fbcde9cf4eac4e14df" +dependencies = [ + "lexical-util", +] + +[[package]] +name = "libc" +version = "0.2.189" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" + +[[package]] +name = "libm" +version = "0.2.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981" + +[[package]] +name = "litemap" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" + +[[package]] +name = "lock_api" +version = "0.4.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "224399e74b87b5f3557511d98dff8b14089b3dadafcab6bb93eab67d3aace965" +dependencies = [ + "scopeguard", +] + +[[package]] +name = "log" +version = "0.4.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" + +[[package]] +name = "lz4_flex" +version = "0.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7ef0d4ed8669f8f8826eb00dc878084aa8f253506c4fd5e8f58f5bce72ddb97e" +dependencies = [ + "twox-hash", +] + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "miniz_oxide" +version = "0.8.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fa76a2c86f704bdb222d66965fb3d63269ce38518b83cb0575fca855ebb6316" +dependencies = [ + "adler2", +] + +[[package]] +name = "miniz_oxide" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" +dependencies = [ + "adler2", + "simd-adler32", +] + +[[package]] +name = "mio" +version = "1.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4b18443e9c262bfe8fa82f51666e2642c53393f7e5c27b3e1aeab922cff5b9d8" +dependencies = [ + "libc", + "wasi", + "windows-sys 0.61.2", +] + +[[package]] +name = "moka" +version = "0.12.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4293f18e7567a1caf3c584855554377025c65e0aa445344d04171f5ad63d19b9" +dependencies = [ + "async-lock", + "crossbeam-channel", + "crossbeam-epoch", + "crossbeam-utils", + "equivalent", + "event-listener", + "futures-util", + "parking_lot", + "portable-atomic", + "smallvec", + "tagptr", + "uuid", +] + +[[package]] +name = "murmur3" +version = "0.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9252111cf132ba0929b6f8e030cac2a24b507f3a4d6db6fb2896f27b354c714b" + +[[package]] +name = "num-bigint" +version = "0.4.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c89e69e7e0f03bea5ef08013795c25018e101932225a656383bd384495ecc367" +dependencies = [ + "num-integer", + "num-traits", + "serde", +] + +[[package]] +name = "num-complex" +version = "0.4.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "73f88a1307638156682bada9d7604135552957b7818057dcef22705b4d509495" +dependencies = [ + "num-traits", +] + +[[package]] +name = "num-conv" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "521739c6d2bac4aa25192232afe6841231376b2b26d4d9fae5ecf8ca5772e441" + +[[package]] +name = "num-integer" +version = "0.1.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7ce2d95d4b3734dc35aa2f45e1aa22cd416814592a4f9d9205e11affd5b8e10b" +dependencies = [ + "num-traits", +] + +[[package]] +name = "num-traits" +version = "0.2.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" +dependencies = [ + "autocfg", + "libm", +] + +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "opaque-debug" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c08d65885ee38876c4f86fa503fb49d7b507c2b62552df7c70b2fce627e06381" + +[[package]] +name = "ordered-float" +version = "2.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "68f19d67e5a2795c94e73e0bb1cc1a7edeb2e28efd39e2e1c9b7a40c1108b11c" +dependencies = [ + "num-traits", +] + +[[package]] +name = "ordered-float" +version = "4.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7bb71e1b3fa6ca1c61f383464aaf2bb0e2f8e772a1f01d486832464de363b951" +dependencies = [ + "num-traits", +] + +[[package]] +name = "parking" +version = "2.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f38d5652c16fde515bb1ecef450ab0f6a219d619a7274976324d5e377f7dceba" + +[[package]] +name = "parking_lot" +version = "0.12.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "93857453250e3077bd71ff98b6a65ea6621a19bb0f559a85248955ac12c45a1a" +dependencies = [ + "lock_api", + "parking_lot_core", +] + +[[package]] +name = "parking_lot_core" +version = "0.9.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2621685985a2ebf1c516881c026032ac7deafcda1a2c9b7850dc81e3dfcb64c1" +dependencies = [ + "cfg-if", + "libc", + "redox_syscall", + "smallvec", + "windows-link", +] + +[[package]] +name = "parquet" +version = "58.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d298093b2dec60289dce0684c986d0f7679e9dd15771c2c65406e1aaf604a704" +dependencies = [ + "ahash", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-ipc", + "arrow-schema", + "arrow-select", + "base64 0.22.1", + "brotli", + "bytes", + "chrono", + "flate2", + "futures", + "half", + "hashbrown 0.17.1", + "lz4_flex", + "num-bigint", + "num-integer", + "num-traits", + "paste", + "ring", + "seq-macro", + "simdutf8", + "snap", + "thrift", + "tokio", + "twox-hash", + "zstd", +] + +[[package]] +name = "paste" +version = "1.0.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "57c0d7b74b563b49d38dae00a0c37d4d6de9b432382b2892f0574ddcae73fd0a" + +[[package]] +name = "percent-encoding" +version = "2.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" + +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + +[[package]] +name = "pkg-config" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" + +[[package]] +name = "polyval" +version = "0.6.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9d1fe60d06143b2430aa532c94cfe9e29783047f06c0d7fd359a9a51b729fa25" +dependencies = [ + "cfg-if", + "cpufeatures", + "opaque-debug", + "universal-hash", +] + +[[package]] +name = "portable-atomic" +version = "1.15.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" + +[[package]] +name = "portable-atomic-util" +version = "0.2.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "10ab3eb7f3becc3a1cbc4f2c6f20267996cfc1a6467a873763411b136a122715" +dependencies = [ + "portable-atomic", +] + +[[package]] +name = "potential_utf" +version = "0.1.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" +dependencies = [ + "zerovec", +] + +[[package]] +name = "powerfmt" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "439ee305def115ba05938db6eb1644ff94165c5ab5e9420d1c1bcedbba909391" + +[[package]] +name = "ppv-lite86" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +dependencies = [ + "zerocopy", +] + +[[package]] +name = "prettyplease" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2bfe0f4c752e450fc2faf62654f1c134747922825d5b04ca717b8874f41a40c0" +dependencies = [ + "proc-macro2", + "syn 3.0.6", +] + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quad-rand" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a651516ddc9168ebd67b24afd085a718be02f8858fe406591b013d101ce2f40" + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "r-efi" +version = "5.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" + +[[package]] +name = "r-efi" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" + +[[package]] +name = "rand" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" +dependencies = [ + "rand_chacha", + "rand_core 0.9.5", +] + +[[package]] +name = "rand_chacha" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" +dependencies = [ + "ppv-lite86", + "rand_core 0.9.5", +] + +[[package]] +name = "rand_core" +version = "0.6.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c" +dependencies = [ + "getrandom 0.2.17", +] + +[[package]] +name = "rand_core" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" +dependencies = [ + "getrandom 0.3.4", +] + +[[package]] +name = "redox_syscall" +version = "0.5.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" +dependencies = [ + "bitflags 2.13.2", +] + +[[package]] +name = "ref-cast" +version = "1.0.27" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7e440fb4e4b4147295338efb76001ab9e4efc0e5839df2c47fc5ac2381d365c3" +dependencies = [ + "ref-cast-impl", +] + +[[package]] +name = "ref-cast-impl" +version = "1.0.27" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92ecd8964f8453721699a1ed72037b0db49ce2f5a5138486ee89bed6f67cdf3a" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "regex" +version = "1.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" +dependencies = [ + "aho-corasick", + "memchr", + "regex-automata", + "regex-syntax", +] + +[[package]] +name = "regex-automata" +version = "0.4.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" +dependencies = [ + "aho-corasick", + "memchr", + "regex-syntax", +] + +[[package]] +name = "regex-lite" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cab834c73d247e67f4fae452806d17d3c7501756d98c8808d7c9c7aa7d18f973" + +[[package]] +name = "regex-syntax" +version = "0.8.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" + +[[package]] +name = "reqwest" +version = "0.12.28" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147" +dependencies = [ + "base64 0.22.1", + "bytes", + "futures-core", + "http", + "http-body", + "http-body-util", + "hyper", + "hyper-util", + "js-sys", + "log", + "percent-encoding", + "pin-project-lite", + "serde", + "serde_json", + "serde_urlencoded", + "sync_wrapper", + "tokio", + "tower", + "tower-http", + "tower-service", + "url", + "wasm-bindgen", + "wasm-bindgen-futures", + "web-sys", +] + +[[package]] +name = "ring" +version = "0.17.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7" +dependencies = [ + "cc", + "cfg-if", + "getrandom 0.2.17", + "libc", + "untrusted", + "windows-sys 0.52.0", +] + +[[package]] +name = "roaring" +version = "0.11.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "18bd8a37d17a58532776dcdf6041ce64929adca78e8489d5cacbafe99229d3e1" +dependencies = [ + "bytemuck", + "byteorder", +] + +[[package]] +name = "rustc_version" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cfcb3a22ef46e85b45de6ee7e79d063319ebb6594faafcf1c225ea92ab6e9b92" +dependencies = [ + "semver", +] + +[[package]] +name = "rustversion" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" + +[[package]] +name = "ryu" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" + +[[package]] +name = "schemars" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4cd191f9397d57d581cddd31014772520aa448f65ef991055d7f61582c65165f" +dependencies = [ + "dyn-clone", + "ref-cast", + "serde", + "serde_json", +] + +[[package]] +name = "schemars" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "687274d293b6cdc6e73e0fee520bf2049650090d7164f87672d212a3c530cf4a" +dependencies = [ + "dyn-clone", + "ref-cast", + "serde", + "serde_json", +] + +[[package]] +name = "scopeguard" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" + +[[package]] +name = "semver" +version = "1.0.28" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" + +[[package]] +name = "seq-macro" +version = "0.3.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1bc711410fbe7399f390ca1c3b60ad0f53f80e95c5eb935e52268a0e2cd49acc" + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", + "serde_derive", +] + +[[package]] +name = "serde-big-array" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "11fc7cc2c76d73e0f27ee52abbd64eec84d46f370c88371120433196934e4b7f" +dependencies = [ + "serde", +] + +[[package]] +name = "serde_bytes" +version = "0.11.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a5d440709e79d88e51ac01c4b72fc6cb7314017bb7da9eeff678aa94c10e3ea8" +dependencies = [ + "serde", + "serde_core", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "serde_repr" +version = "0.1.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8d3b1629de253c70a0508c3899572da79ca359fdab27c7920ff00406df418906" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "serde_urlencoded" +version = "0.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3491c14715ca2294c4d6a88f15e84739788c1d030eed8c110436aafdaa2f3fd" +dependencies = [ + "form_urlencoded", + "itoa", + "ryu", + "serde", +] + +[[package]] +name = "serde_with" +version = "3.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "df9adc193c780ef8f159aee8b61e2d5801aaa555e6eb0947fe45530ec506296f" +dependencies = [ + "base64 0.23.1", + "bs58", + "chrono", + "hex", + "indexmap 1.9.3", + "indexmap 2.14.2", + "jiff", + "schemars 0.9.0", + "schemars 1.2.2", + "serde_core", + "serde_json", + "serde_with_macros", + "time", +] + +[[package]] +name = "serde_with_macros" +version = "3.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3e17bbc68e28663bbbb90df47e058aa7eda4fb445b89fe70457bb94fbccf6e49" +dependencies = [ + "darling 0.24.1", + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "shlex" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" + +[[package]] +name = "simd-adler32" +version = "0.3.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" + +[[package]] +name = "simdutf8" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" + +[[package]] +name = "slab" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" + +[[package]] +name = "smallvec" +version = "1.16.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9395f0f0eee849a9b707b2f06bb92a6a422090e2123bb2ef8e87a0e61892a8e" + +[[package]] +name = "snap" +version = "1.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" + +[[package]] +name = "socket2" +version = "0.6.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + +[[package]] +name = "stable_deref_trait" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" + +[[package]] +name = "strsim" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" + +[[package]] +name = "strum" +version = "0.27.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "af23d6f6c1a224baef9d3f61e287d2761385a5b88fdab4eb4c6f11aeb54c4bcf" +dependencies = [ + "strum_macros", +] + +[[package]] +name = "strum_macros" +version = "0.27.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7695ce3845ea4b33927c055a39dc438a45b059f7c1b3d91d38d10355fb8cbca7" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "subtle" +version = "2.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" + +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "sync_wrapper" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0bf256ce5efdfa370213c1dabab5935a12e49f2c58d15e9eac2870d3b4f27263" +dependencies = [ + "futures-core", +] + +[[package]] +name = "synstructure" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "901704edd0dfe137f1987838ee4f259e4e063c31371bdb423f7ae38ec6f77f02" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "tagptr" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7b2093cf4c8eb1e67749a6762251bc9cd836b6fc171623bd0a9d324d37af2417" + +[[package]] +name = "thiserror" +version = "2.0.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09e52cb86a36cede5cb101bf8908837b3e4c6e5e59fe7fd85c23fb56200d189e" +dependencies = [ + "thiserror-impl", +] + +[[package]] +name = "thiserror-impl" +version = "2.0.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fe5197923287db20a58125f0bc85c062f7f2c892de97b18c356f9efb14b28524" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "thrift" +version = "0.17.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7e54bc85fc7faa8bc175c4bab5b92ba8d9a3ce893d0e9f42cc455c8ab16a9e09" +dependencies = [ + "byteorder", + "integer-encoding", + "ordered-float 2.10.1", +] + +[[package]] +name = "time" +version = "0.3.55" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" +dependencies = [ + "deranged", + "num-conv", + "powerfmt", + "serde_core", + "time-core", + "time-macros", +] + +[[package]] +name = "time-core" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9e1c906769ad99c88eaa54e728060edef082f8e358ff32030cb7c7d315e81109" + +[[package]] +name = "time-macros" +version = "0.2.32" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7e689342a48d2ea927c87ea50cabf8594854bf940e9310208848d680d668ed85" +dependencies = [ + "num-conv", + "time-core", +] + +[[package]] +name = "tiny-keccak" +version = "2.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2c9d3793400a45f954c52e73d068316d76b6f4e36977e3fcebb13a2721e80237" +dependencies = [ + "crunchy", +] + +[[package]] +name = "tinystr" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" +dependencies = [ + "displaydoc", + "zerovec", +] + +[[package]] +name = "tinyvec" +version = "1.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fd3ca314f692efd6c868f8408f53fe444634a845f96c028b97d35f6a1f79f0ee" + +[[package]] +name = "tokio" +version = "1.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" +dependencies = [ + "bytes", + "libc", + "mio", + "pin-project-lite", + "socket2", + "tokio-macros", + "windows-sys 0.61.2", +] + +[[package]] +name = "tokio-macros" +version = "2.7.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "tower" +version = "0.5.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4" +dependencies = [ + "futures-core", + "futures-util", + "pin-project-lite", + "sync_wrapper", + "tokio", + "tower-layer", + "tower-service", +] + +[[package]] +name = "tower-http" +version = "0.6.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840" +dependencies = [ + "bitflags 2.13.2", + "bytes", + "futures-util", + "http", + "http-body", + "pin-project-lite", + "tower", + "tower-layer", + "tower-service", + "url", +] + +[[package]] +name = "tower-layer" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "121c2a6cda46980bb0fcd1647ffaf6cd3fc79a013de288782836f6df9c48780e" + +[[package]] +name = "tower-service" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3" + +[[package]] +name = "tracing" +version = "0.1.44" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" +dependencies = [ + "pin-project-lite", + "tracing-attributes", + "tracing-core", +] + +[[package]] +name = "tracing-attributes" +version = "0.1.31" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "tracing-core" +version = "0.1.36" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" +dependencies = [ + "once_cell", +] + +[[package]] +name = "try-lock" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" + +[[package]] +name = "twox-hash" +version = "2.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" + +[[package]] +name = "typed-builder" +version = "0.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cd9d30e3a08026c78f246b173243cf07b3696d274debd26680773b6773c2afc7" +dependencies = [ + "typed-builder-macro", +] + +[[package]] +name = "typed-builder-macro" +version = "0.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3c36781cc0e46a83726d9879608e4cf6c2505237e263a8eb8c24502989cfdb28" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "typeid" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bc7d623258602320d5c55d1bc22793b57daff0ec7efc270ea7d55ce1d5f5471c" + +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + +[[package]] +name = "typetag" +version = "0.2.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c90e86058a30d42a1a928dfb4b49bb33c98c3a2b4909492e6b0881cd94798ec2" +dependencies = [ + "erased-serde", + "inventory", + "once_cell", + "serde", + "typetag-impl", +] + +[[package]] +name = "typetag-impl" +version = "0.2.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f153acc4e99a5f2a5aefa09fb078be54e26271b2813f6041200b224c098d8328" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "unicode-ident" +version = "1.0.26" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" + +[[package]] +name = "universal-hash" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc1de2c688dc15305988b563c3854064043356019f97a4b46276fe734c4f07ea" +dependencies = [ + "crypto-common", + "subtle", +] + +[[package]] +name = "untrusted" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" + +[[package]] +name = "url" +version = "2.5.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff67a8a4397373c3ef660812acab3268222035010ab8680ec4215f38ba3d0eed" +dependencies = [ + "form_urlencoded", + "idna", + "percent-encoding", + "serde", +] + +[[package]] +name = "utf8_iter" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" + +[[package]] +name = "uuid" +version = "1.26.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2ef6dac1e96601b4fb3acccccff2139741fcb757cb9a36089bf5be91cfb285ce" +dependencies = [ + "getrandom 0.4.3", + "js-sys", + "serde_core", + "wasm-bindgen", +] + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "want" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bfa7760aed19e106de2c7c0b581b509f2f25d3dacaf737cb82ac61bc6d760b0e" +dependencies = [ + "try-lock", +] + +[[package]] +name = "wasi" +version = "0.11.1+wasi-snapshot-preview1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" + +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen", +] + +[[package]] +name = "wasm-bindgen" +version = "0.2.129" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9bb54f33acc68fd454578d9820b0bde1a1a3d17aa17bb7b6595806d02886d409" +dependencies = [ + "cfg-if", + "once_cell", + "rustversion", + "wasm-bindgen-macro", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-futures" +version = "0.4.79" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3cbab34de2d982e9b48e18d216d04c4a6f641066ff19ffb699980f591ee3610e" +dependencies = [ + "js-sys", + "tokio", + "wasm-bindgen", +] + +[[package]] +name = "wasm-bindgen-macro" +version = "0.2.129" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2e29d0c35b16e224a7eeb5cd2d25e3e1968fbd65604117b44d3b789d00ee8535" +dependencies = [ + "quote", + "wasm-bindgen-macro-support", +] + +[[package]] +name = "wasm-bindgen-macro-support" +version = "0.2.129" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6f501a8bc3719dba86ef8ae4728879c08001bea749eb1333ac5b91e040e2a6b7" +dependencies = [ + "bumpalo", + "proc-macro2", + "quote", + "syn 3.0.6", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-shared" +version = "0.2.129" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "23f0c9c52aa7cd7d77769a4cfe2a9adb1b331f489a41d912ce14513d5ab995c6" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "web-sys" +version = "0.3.106" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "88261b9deccee56594c11a3460c462c41f58d148598fe70ad77070126a68aba4" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "windows-core" +version = "0.62.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" +dependencies = [ + "windows-implement", + "windows-interface", + "windows-link", + "windows-result", + "windows-strings", +] + +[[package]] +name = "windows-implement" +version = "0.60.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "windows-interface" +version = "0.59.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-result" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-strings" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-sys" +version = "0.52.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" +dependencies = [ + "windows-targets", +] + +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-targets" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" +dependencies = [ + "windows_aarch64_gnullvm", + "windows_aarch64_msvc", + "windows_i686_gnu", + "windows_i686_gnullvm", + "windows_i686_msvc", + "windows_x86_64_gnu", + "windows_x86_64_gnullvm", + "windows_x86_64_msvc", +] + +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" + +[[package]] +name = "windows_aarch64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" + +[[package]] +name = "windows_i686_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" + +[[package]] +name = "windows_i686_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" + +[[package]] +name = "windows_i686_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" + +[[package]] +name = "windows_x86_64_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" + +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" + +[[package]] +name = "windows_x86_64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" + +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" + +[[package]] +name = "writeable" +version = "0.6.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" + +[[package]] +name = "yoke" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" +dependencies = [ + "stable_deref_trait", + "yoke-derive", + "zerofrom", +] + +[[package]] +name = "yoke-derive" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "33811428bee40dbceb6d545e95754741d17a6aef9a4849f0fd62e2ba4f412a78" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", + "synstructure", +] + +[[package]] +name = "zerocopy" +version = "0.8.59" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6df92bf3d9227be3d53173901ddbffac2babc27ae50f397776ffd6dc33f800cb" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.59" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac4f328cf2f05d084e496c3e9c3f33ed0a183656a16e1fcec4d464d8373aec82" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zerofrom" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" +dependencies = [ + "zerofrom-derive", +] + +[[package]] +name = "zerofrom-derive" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f75b4683f6c7f45248d4d64056a24298c6281e0993356d7d1b4a1a962ef10d4a" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", + "synstructure", +] + +[[package]] +name = "zeroize" +version = "1.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" + +[[package]] +name = "zerotrie" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" +dependencies = [ + "displaydoc", + "yoke", + "zerofrom", +] + +[[package]] +name = "zerovec" +version = "0.11.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" +dependencies = [ + "yoke", + "zerofrom", + "zerovec-derive", +] + +[[package]] +name = "zerovec-derive" +version = "0.11.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "zlib-rs" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112" + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" + +[[package]] +name = "zstd" +version = "0.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a" +dependencies = [ + "zstd-safe", +] + +[[package]] +name = "zstd-safe" +version = "7.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882" +dependencies = [ + "zstd-sys", +] + +[[package]] +name = "zstd-sys" +version = "2.1.0+zstd.1.5.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0" +dependencies = [ + "cc", + "pkg-config", +] diff --git a/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.toml b/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.toml new file mode 100644 index 000000000..c5a24bab4 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.toml @@ -0,0 +1,12 @@ +[package] +name = "crowdb-iceberg-rust-client-fixture" +version = "0.1.0-dev" +edition = "2021" +publish = false + +[workspace] + +[dependencies] +iceberg = "=0.10.0" +iceberg-catalog-rest = "=0.10.0" +tokio = { version = "1", features = ["macros", "rt-multi-thread"] } diff --git a/app/crowdb-access-server/tests/common/iceberg_rust/src/main.rs b/app/crowdb-access-server/tests/common/iceberg_rust/src/main.rs new file mode 100644 index 000000000..96c6a3971 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_rust/src/main.rs @@ -0,0 +1,154 @@ +use std::collections::HashMap; +use std::env; +use std::sync::Arc; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +use iceberg::io::MemoryStorageFactory; +use iceberg::spec::{NestedField, PrimitiveType, Schema, Type}; +use iceberg::{Catalog, CatalogBuilder, NamespaceIdent, TableCreation, TableIdent}; +use iceberg_catalog_rest::RestCatalogBuilder; + +#[tokio::main] +async fn main() -> Result<(), Box> { + let origin = env::var("CROWDB_ICEBERG_RUST_ORIGIN")?; + let second_origin = env::var("CROWDB_ICEBERG_RUST_SECOND_ORIGIN")?; + let token = env::var("CROWDB_ICEBERG_RUST_TOKEN")?; + let namespace = NamespaceIdent::new(env::var("CROWDB_ICEBERG_RUST_NAMESPACE")?); + let catalog = RestCatalogBuilder::default() + .with_storage_factory(Arc::new(MemoryStorageFactory)) + .load( + "crowdb", + HashMap::from([("uri".to_owned(), origin), ("token".to_owned(), token.clone())]), + ) + .await?; + let second_catalog = RestCatalogBuilder::default() + .with_storage_factory(Arc::new(MemoryStorageFactory)) + .load( + "crowdb", + HashMap::from([("uri".to_owned(), second_origin), ("token".to_owned(), token)]), + ) + .await?; + + if let Ok(control) = env::var("CROWDB_ICEBERG_RUST_RETIRE_CONTROL") { + catalog.create_namespace(&namespace, HashMap::new()).await?; + let table = TableIdent::new(namespace.clone(), "rust_retired".to_owned()); + let schema = Schema::builder() + .with_fields(vec![NestedField::required( + 1, + "id", + Type::Primitive(PrimitiveType::Long), + ) + .into()]) + .build()?; + catalog + .create_table( + &namespace, + TableCreation::builder() + .name(table.name().to_owned()) + .schema(schema.clone()) + .build(), + ) + .await?; + second_catalog.load_table(&table).await?; + let mut control = tokio::net::TcpStream::connect(control).await?; + control.write_all(&[1]).await?; + control.read_exact(&mut [0]).await?; + assert!(catalog.load_table(&table).await.is_err()); + assert!(!second_catalog.namespace_exists(&namespace).await?); + second_catalog.create_namespace(&namespace, HashMap::new()).await?; + second_catalog + .create_table( + &namespace, + TableCreation::builder() + .name(table.name().to_owned()) + .schema(schema) + .build(), + ) + .await?; + catalog.load_table(&table).await?; + second_catalog.drop_table(&table).await?; + catalog.drop_namespace(&namespace).await?; + return Ok(()); + } + + if env::var_os("CROWDB_ICEBERG_RUST_VERIFY_EXISTING").is_some() { + let table = TableIdent::new(namespace.clone(), "rust_lost_reply".to_owned()); + assert!(second_catalog.namespace_exists(&namespace).await?); + assert!(catalog.table_exists(&table).await?); + second_catalog.load_table(&table).await?; + catalog.drop_table(&table).await?; + second_catalog.drop_namespace(&namespace).await?; + return Ok(()); + } + + if env::var_os("CROWDB_ICEBERG_RUST_RESPONSE_LOSS").is_some() { + second_catalog.create_namespace(&namespace, HashMap::new()).await?; + let table = TableIdent::new(namespace.clone(), "rust_lost_reply".to_owned()); + let schema = Schema::builder() + .with_fields(vec![NestedField::required( + 1, + "id", + Type::Primitive(PrimitiveType::Long), + ) + .into()]) + .build()?; + assert!(catalog + .create_table( + &namespace, + TableCreation::builder() + .name(table.name().to_owned()) + .schema(schema) + .build(), + ) + .await + .is_err()); + assert!(second_catalog.table_exists(&table).await?); + second_catalog.load_table(&table).await?; + if env::var_os("CROWDB_ICEBERG_RUST_KEEP_TABLE").is_none() { + second_catalog.drop_table(&table).await?; + second_catalog.drop_namespace(&namespace).await?; + } + return Ok(()); + } + + assert!(!catalog.namespace_exists(&namespace).await?); + catalog.create_namespace(&namespace, HashMap::new()).await?; + assert!(catalog.namespace_exists(&namespace).await?); + assert!(second_catalog.namespace_exists(&namespace).await?); + assert!(catalog.list_namespaces(None).await?.contains(&namespace)); + assert_eq!(catalog.get_namespace(&namespace).await?.name(), &namespace); + assert!(catalog.list_tables(&namespace).await?.is_empty()); + let table = TableIdent::new(namespace.clone(), "rust_table".to_owned()); + assert!(!catalog.table_exists(&table).await?); + let schema = Schema::builder() + .with_fields(vec![NestedField::required( + 1, + "id", + Type::Primitive(PrimitiveType::Long), + ) + .into()]) + .build()?; + catalog + .create_table( + &namespace, + TableCreation::builder() + .name(table.name().to_owned()) + .schema(schema) + .build(), + ) + .await?; + assert!(catalog.table_exists(&table).await?); + assert!(second_catalog.table_exists(&table).await?); + assert!(second_catalog.list_tables(&namespace).await?.contains(&table)); + second_catalog.load_table(&table).await?; + let renamed = TableIdent::new(namespace.clone(), "rust_renamed".to_owned()); + second_catalog.rename_table(&table, &renamed).await?; + assert!(!second_catalog.table_exists(&table).await?); + assert!(catalog.table_exists(&renamed).await?); + catalog.load_table(&renamed).await?; + catalog.drop_table(&renamed).await?; + assert!(!second_catalog.table_exists(&renamed).await?); + catalog.drop_namespace(&namespace).await?; + assert!(!second_catalog.namespace_exists(&namespace).await?); + Ok(()) +} diff --git a/app/crowdb-access-server/tests/common/iceberg_signed_chunks.rs b/app/crowdb-access-server/tests/common/iceberg_signed_chunks.rs new file mode 100644 index 000000000..6da75299f --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_signed_chunks.rs @@ -0,0 +1,149 @@ +use base64::{engine::general_purpose::STANDARD, Engine}; +use crowdb_access_s3::auth::{ + Credential, CredentialProvider, RawAuthRequest, SigV4Verifier, StreamingPayloadVerifier, +}; +use hmac::{Hmac, Mac}; +use hyper::{HeaderMap, Request}; +use sha2::{Digest, Sha256}; +use std::fmt::Write; + +pub struct TestAwsCredentials; + +impl CredentialProvider for TestAwsCredentials { + fn lookup(&self, access: &str) -> Option { + (access == "AKIAIOSFODNN7EXAMPLE").then(|| Credential { + secret_key: b"wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY".to_vec(), + session_token: None, + enabled: true, + }) + } +} + +pub fn signed_request() -> Request<()> { + Request::builder().method("PUT").uri("/examplebucket/chunkObject.txt") + .header("host", "s3.amazonaws.com") + .header("content-encoding", "aws-chunked") + .header("x-amz-content-sha256", "STREAMING-AWS4-HMAC-SHA256-PAYLOAD-TRAILER") + .header("x-amz-date", "20130524T000000Z") + .header("x-amz-decoded-content-length", "66560") + .header("x-amz-storage-class", "REDUCED_REDUNDANCY") + .header("x-amz-trailer", "x-amz-checksum-crc32c") + .header("authorization", "AWS4-HMAC-SHA256 Credential=AKIAIOSFODNN7EXAMPLE/20130524/us-east-1/s3/aws4_request, SignedHeaders=content-encoding;host;x-amz-content-sha256;x-amz-date;x-amz-decoded-content-length;x-amz-storage-class;x-amz-trailer, Signature=106e2a8a18243abcf37539882f36619c00e2dfc72633413f02d3b74544bfeb8e") + .body(()).unwrap() +} + +pub fn fixture() -> (HeaderMap, StreamingPayloadVerifier, Vec) { + let request = signed_request(); + let verifier = SigV4Verifier::new(TestAwsCredentials, "us-east-1".into(), 900); + let raw = RawAuthRequest::from_parts(request.method(), request.uri(), request.headers()); + assert!(verifier.verify(raw, 1_369_353_600).is_err()); + let streaming = verifier.verify_streaming(raw, 1_369_353_600).unwrap(); + let mut bytes = + b"10000;chunk-signature=b474d8862b1487a5145d686f57f013e54db672cee1c953b3010fb58501ef5aa2\r\n" + .to_vec(); + bytes.extend(vec![b'a'; 65536]); + bytes.extend_from_slice( + b"\r\n400;chunk-signature=1c1344b170168f8e65b41376b44b20fe354e373826ccbbe2c1d40a8cae51e5c7\r\n", + ); + bytes.extend(vec![b'a'; 1024]); + bytes.extend_from_slice(b"\r\n0;chunk-signature=2ca2aba2005185cf7159c6277faf83795951dd77a3a99e6e65d5c9f85863f992\r\nx-amz-checksum-crc32c:sOO8/Q==\r\nx-amz-trailer-signature:d81f82fc3505edab99d459891051a732e8730629a2e4a59689829ca17fe2e435\r\n\r\n"); + (request.into_parts().0.headers, streaming, bytes) +} + +pub fn other_fixture(unsigned: bool, empty: bool) -> (HeaderMap, StreamingPayloadVerifier, Vec) { + let mut request = signed_request(); + let payload = if empty { b"".as_slice() } else { b"abc".as_slice() }; + let mode = if unsigned { + "STREAMING-UNSIGNED-PAYLOAD-TRAILER" + } else { + "STREAMING-AWS4-HMAC-SHA256-PAYLOAD" + }; + request + .headers_mut() + .insert("x-amz-content-sha256", mode.parse().unwrap()); + request.headers_mut().insert( + "x-amz-decoded-content-length", + payload.len().to_string().parse().unwrap(), + ); + let mut names = "content-encoding;host;x-amz-content-sha256;x-amz-date;x-amz-decoded-content-length;x-amz-storage-class".to_owned(); + if unsigned { + names.push_str(";x-amz-trailer"); + request + .headers_mut() + .insert("x-amz-trailer", "x-amz-checksum-sha256".parse().unwrap()); + } else { + request.headers_mut().remove("x-amz-trailer"); + } + let mut headers = String::new(); + for name in names.split(';') { + writeln!(headers, "{name}:{}", request.headers()[name].to_str().unwrap()).unwrap(); + } + let canonical = format!("PUT\n/examplebucket/chunkObject.txt\n\n{headers}\n{names}\n{mode}"); + let scope = "20130524/us-east-1/s3/aws4_request"; + let date = "20130524T000000Z"; + let mut key = b"AWS4wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY".to_vec(); + for item in ["20130524", "us-east-1", "s3", "aws4_request"] { + key = mac(&key, item); + } + let seed = hex(&mac( + &key, + &format!( + "AWS4-HMAC-SHA256\n{date}\n{scope}\n{:x}", + Sha256::digest(canonical) + ), + )); + request.headers_mut().insert("authorization", format!("AWS4-HMAC-SHA256 Credential=AKIAIOSFODNN7EXAMPLE/{scope}, SignedHeaders={names}, Signature={seed}").parse().unwrap()); + let mut previous = seed; + let mut bytes = Vec::new(); + for data in std::iter::once(payload).chain((!payload.is_empty()).then_some(b"".as_slice())) { + if !bytes.is_empty() { + bytes.extend_from_slice(b"\r\n"); + } + if unsigned { + bytes.extend_from_slice(format!("{:x}\r\n", data.len()).as_bytes()); + } else { + let signature = hex(&mac( + &key, + &format!( + "AWS4-HMAC-SHA256-PAYLOAD\n{date}\n{scope}\n{previous}\n{:x}\n{:x}", + Sha256::digest([]), + Sha256::digest(data) + ), + )); + bytes.extend_from_slice(format!("{:x};chunk-signature={signature}\r\n", data.len()).as_bytes()); + previous = signature; + } + bytes.extend_from_slice(data); + } + if unsigned { + bytes.extend_from_slice( + format!( + "x-amz-checksum-sha256:{}\r\n", + STANDARD.encode(Sha256::digest(payload)) + ) + .as_bytes(), + ); + } + bytes.extend_from_slice(b"\r\n"); + let verifier = SigV4Verifier::new(TestAwsCredentials, "us-east-1".into(), 900) + .verify_streaming( + RawAuthRequest::from_parts(request.method(), request.uri(), request.headers()), + 1_369_353_600, + ) + .unwrap(); + (request.into_parts().0.headers, verifier, bytes) +} + +fn mac(key: &[u8], message: &str) -> Vec { + let mut signer = Hmac::::new_from_slice(key).unwrap(); + signer.update(message.as_bytes()); + signer.finalize().into_bytes().to_vec() +} + +fn hex(bytes: &[u8]) -> String { + let mut output = String::new(); + for byte in bytes { + write!(output, "{byte:02x}").unwrap(); + } + output +} diff --git a/app/crowdb-access-server/tests/common/iceberg_signed_file.rs b/app/crowdb-access-server/tests/common/iceberg_signed_file.rs new file mode 100644 index 000000000..a4bb32b5c --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_signed_file.rs @@ -0,0 +1,109 @@ +use super::common::now_ms; +use base64::engine::general_purpose::STANDARD; +use base64::Engine; +use hmac::{Hmac, Mac}; +use md5::Md5; +use reqwest::{Client, Method, Response}; +use sha2::{Digest, Sha256}; +use std::fmt::Write; + +fn hex(bytes: &[u8]) -> String { + let mut result = String::new(); + for byte in bytes { + write!(result, "{byte:02x}").unwrap(); + } + result +} + +fn mac(key: &[u8], input: &str) -> Vec { + let mut signer = Hmac::::new_from_slice(key).unwrap(); + signer.update(input.as_bytes()); + signer.finalize().into_bytes().to_vec() +} + +pub struct TestFileClient { + pub client: Client, + pub credentials: crowdb_access_iceberg::file::FileCredentials, + pub address: std::net::SocketAddr, +} + +impl TestFileClient { + pub async fn send(&self, method: Method, path: &str, query: &str, body: &[u8], md5: bool) -> Response { + self.send_range(method, path, query, body, md5, None).await + } + + pub async fn send_range( + &self, + method: Method, + path: &str, + query: &str, + body: &[u8], + md5: bool, + range: Option<&str>, + ) -> Response { + self.request(method, path, query, body, md5, range) + .send() + .await + .unwrap() + } + + pub fn request( + &self, + method: Method, + path: &str, + query: &str, + body: &[u8], + md5: bool, + range: Option<&str>, + ) -> reqwest::RequestBuilder { + let now = + chrono::DateTime::::from_timestamp_millis(i64::try_from(now_ms()).unwrap()).unwrap(); + let date = now.format("%Y%m%dT%H%M%SZ").to_string(); + let short = now.format("%Y%m%d").to_string(); + let hash = hex(&Sha256::digest(body)); + let host = self.address.to_string(); + let names = "host;x-amz-content-sha256;x-amz-date;x-amz-security-token"; + let canonical = format!( + "{}\n{path}\n{query}\nhost:{host}\nx-amz-content-sha256:{hash}\nx-amz-date:{date}\nx-amz-security-token:{}\n\n{names}\n{hash}", + method.as_str(), self.credentials.session_token() + ); + let date_key = mac( + format!("AWS4{}", self.credentials.secret_access_key()).as_bytes(), + &short, + ); + let region_key = mac(&date_key, "us-east-1"); + let service_key = mac(®ion_key, "s3"); + let signing_key = mac(&service_key, "aws4_request"); + let scope = format!("{short}/us-east-1/s3/aws4_request"); + let string_to_sign = format!( + "AWS4-HMAC-SHA256\n{date}\n{scope}\n{}", + hex(&Sha256::digest(canonical)) + ); + let authorization = format!( + "AWS4-HMAC-SHA256 Credential={}/{scope}, SignedHeaders={names}, Signature={}", + self.credentials.access_key_id(), + hex(&mac(&signing_key, &string_to_sign)) + ); + let url = if query.is_empty() { + format!("http://{host}{path}") + } else { + format!("http://{host}{path}?{query}") + }; + let mut request = self + .client + .request(method, url) + .header("host", host) + .header("x-amz-content-sha256", hash) + .header("x-amz-date", date) + .header("x-amz-security-token", self.credentials.session_token()) + .header("authorization", authorization) + .body(body.to_vec()); + if md5 { + request = request.header("content-md5", STANDARD.encode(Md5::digest(body))); + } + if let Some(range) = range { + request = request.header("range", range); + } + request + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_stack.rs b/app/crowdb-access-server/tests/common/iceberg_stack.rs new file mode 100644 index 000000000..3671dc9ce --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_stack.rs @@ -0,0 +1,223 @@ +use std::sync::Arc; +use std::time::{SystemTime, UNIX_EPOCH}; + +use crowdb_access_iceberg::catalog::RoutedCatalogStore; +use crowdb_chunk_kv_client::{ + ChunkKvClient, ChunkKvRpcTransport, ClientConfig, Group0ChunkKvRangeCatalogSource, +}; +use crowdb_diskio_client::{DiskId as DiskIoDiskId, TestWireDiskioClient}; +use crowdb_protocol::common::{DiskId, HwStatus, NodeValue, RackValue}; +use crowdb_protocol::diskdb::rpc::{DiskGroupValue, DiskType, DiskValue}; +use crowdb_rpc_ffi::RpcServer; +use crowdb_test_harness::chunk_kv::ChunkKvProcess; +use crowdb_test_harness::chunkdb::{ChunkdbPlacementMode, ChunkdbProcess, ChunkdbStartOptions}; +use crowdb_test_harness::cluster::KvCluster; +use crowdb_test_harness::diskdb::DiskdbProcess; +use crowdb_test_harness::diskio::{DiskArg, DiskioGroup0Identity, DiskioProcess, DiskioStartOpts}; + +pub struct TestIcebergStack { + pub chunk_kv: ChunkKvProcess, + _chunkdb: ChunkdbProcess, + _diskio: DiskioProcess, + _diskdb: DiskdbProcess, + _rpc: Arc, + pub cluster: KvCluster, +} + +pub fn now_ms() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_millis() + .try_into() + .unwrap() +} + +#[allow(dead_code)] +pub async fn activate(repository: &crowdb_access_iceberg::catalog::CatalogRepository) { + use crowdb_access_iceberg::{ + catalog::{Capabilities, ManagementPrivilege}, + key::OperationId, + operation::{ManagementAction, ManagementRequest, RequestIdentity}, + }; + let (root, authority) = repository.status().await.unwrap(); + let now = now_ms(); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now, + }, + principal: "manager".into(), + action: ManagementAction::Activate, + expected_epoch: root.context.activation_epoch, + display_name: authority.display_name, + confirmation: None, + capabilities: Some(Capabilities::from_bits(0x3fff).unwrap()), + }, + ManagementPrivilege::Manage, + now, + ) + .await + .unwrap(); +} + +impl TestIcebergStack { + pub async fn start() -> Self { + let mut cluster = KvCluster::start().await; + seed(&cluster).await; + let identity = DiskioGroup0Identity { + instance_id: 999, + rack_id: 1, + node_id: 10, + disk_group_id: 100, + }; + let path = cluster + .runtime_mut() + .service_dir("diskio", "iceberg") + .unwrap() + .join("data/disk.dat"); + let capacity = 16_384_u64 * 1024 * 1024; + std::fs::File::create(&path).unwrap().set_len(capacity).unwrap(); + let disk = DiskArg { + id_high: 0, + id_low: 1, + path: path.to_string_lossy().into_owned(), + zone_capacity: capacity.try_into().unwrap(), + }; + let seeds = cluster.mgmt_endpoints.clone(); + let started = now_ms(); + let diskdb = DiskdbProcess::start_for_instance_in(cluster.runtime_mut(), &seeds, 999, Some(16_384)); + diskdb.wait_for_ready().await; + diskdb + .wait_for_registry_ready(&cluster.make_service_registry_client(), 100, started) + .await; + let rpc = Arc::new(RpcServer::new(None)); + rpc.listen("127.0.0.1", 0).unwrap(); + rpc.start(); + let diskio = DiskioProcess::start_for_group_in( + cluster.runtime_mut(), + &DiskioStartOpts { + dummy_disk: "null", + kv_seeds: &seeds, + disks: &[disk], + fault_error_rate: 0.0, + fault_latency_ms: None, + no_o_direct: true, + }, + identity, + ); + let connection = rpc.connect("127.0.0.1", diskio.port).unwrap(); + let client = TestWireDiskioClient::new(); + client.attach(&connection); + diskio + .wait_for_disk(&client, &rpc, &connection, DiskIoDiskId::new(0, 1)) + .await; + cluster + .make_service_registry_client() + .heartbeat_diskio_at(999, &format!("127.0.0.1:{}", diskio.port), 1, 10, &[100], &[]) + .await + .unwrap(); + let chunkdb = ChunkdbProcess::start_with_options_in( + cluster.runtime_mut(), + &seeds, + ChunkdbStartOptions { + placement_mode: ChunkdbPlacementMode::UnsafeColocated, + repair_allow_unsafe_placement: true, + ..ChunkdbStartOptions::default() + }, + ); + chunkdb.wait_for_ready().await; + chunkdb + .wait_for_registry_ready(&cluster.make_service_registry_client()) + .await; + let mut chunk_kv = ChunkKvProcess::start_in(cluster.runtime_mut(), &seeds); + chunk_kv.wait_for_ready().await; + Self { + chunk_kv, + _chunkdb: chunkdb, + _diskio: diskio, + _diskdb: diskdb, + _rpc: rpc, + cluster, + } + } + + pub async fn store(&self) -> Arc { + let config = ClientConfig::default(); + let source = Arc::new(Group0ChunkKvRangeCatalogSource::from_shared(Arc::new( + crowdb_kv_client::CrowdbKvClient::new(crowdb_kv_client::ClientConfig::new( + self.cluster.mgmt_endpoints.clone(), + )), + ))); + let transport = Arc::new(ChunkKvRpcTransport::new(config.max_owner_connections, 1, 2)); + let client = Arc::new(ChunkKvClient::new(config, source, transport).unwrap()); + client.refresh_catalog().await.unwrap(); + Arc::new(RoutedCatalogStore::new(client)) + } +} + +async fn seed(cluster: &KvCluster) { + let hardware = cluster.make_hardware_client(); + hardware + .add_rack( + 1, + &RackValue { + status: HwStatus::Up as i32, + node_ids: vec![10], + }, + ) + .await + .unwrap(); + hardware + .add_node( + 1, + 10, + &NodeValue { + status: HwStatus::Up as i32, + last_used_dg_id: 100, + disk_group_ids: vec![100], + status_changed_at_ms: 0, + temp_failure_since_ms: None, + }, + ) + .await + .unwrap(); + let disk = DiskId { high: 0, low: 1 }; + hardware + .add_disk_group( + 1, + 10, + 100, + &DiskGroupValue { + status: HwStatus::Up as i32, + disk_ids: vec![disk], + }, + ) + .await + .unwrap(); + hardware + .add_disk( + 1, + 10, + 100, + &disk, + &DiskValue { + disk_type: DiskType::BlockSsd as i32, + capacity_units: 16_384, + zone_size_units: 16_384, + unit_size_bytes: 1024 * 1024, + zone_count: 1, + status: HwStatus::Up as i32, + device_path: String::new(), + }, + ) + .await + .unwrap(); + hardware + .set_owner(1, 10, 100, 999, now_ms() + 3_600_000) + .await + .unwrap(); + hardware.set_bind(1, 10, 100, 0, 1).await.unwrap(); +} diff --git a/app/crowdb-access-server/tests/common/iceberg_store.rs b/app/crowdb-access-server/tests/common/iceberg_store.rs new file mode 100644 index 000000000..5bc28cfd3 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_store.rs @@ -0,0 +1,215 @@ +use arc_swap::ArcSwap; +use async_trait::async_trait; +use crowdb_access_iceberg::catalog::{CasOutcome, CatalogStore, StoreError, StoredValue}; +use crowdb_protocol::chunk_kv::ClientRequestId; +use std::collections::BTreeMap; +use std::sync::atomic::{AtomicBool, AtomicU64, AtomicU8, Ordering}; +use std::sync::Arc; + +#[allow(dead_code)] +pub async fn activate(repository: &crowdb_access_iceberg::catalog::CatalogRepository) { + activate_bits(repository, 0x3fff).await; +} + +#[allow(dead_code)] +pub async fn activate_bits(repository: &crowdb_access_iceberg::catalog::CatalogRepository, bits: u16) { + use crowdb_access_iceberg::{ + catalog::{Capabilities, ManagementPrivilege}, + key::OperationId, + operation::{ManagementAction, ManagementRequest, RequestIdentity}, + }; + let (root, authority) = repository.status().await.unwrap(); + let now = u64::try_from( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap(); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now, + }, + principal: "manager".into(), + action: ManagementAction::Activate, + expected_epoch: root.context.activation_epoch, + display_name: authority.display_name, + confirmation: None, + capabilities: Some(Capabilities::from_bits(bits).unwrap()), + }, + ManagementPrivilege::Manage, + now, + ) + .await + .unwrap(); +} + +#[derive(Default)] +pub struct TestStore { + pub values: ArcSwap, StoredValue>>, + pub read_delay_ms: AtomicU64, + pub scan_delay_ms: AtomicU64, + pub scans: AtomicU64, + pub lose_reply_kind: AtomicU8, + pub pause_file_read: AtomicBool, + pub file_read_entered: tokio::sync::Notify, + pub file_read_release: tokio::sync::Notify, + pub pause_head_cas: AtomicBool, + pub head_cas_entered: tokio::sync::Notify, + pub head_cas_release: tokio::sync::Notify, +} + +#[async_trait] +impl crowdb_access_iceberg::namespace::NamespaceStore for TestStore { + async fn scan_children( + &self, + scan: crowdb_access_iceberg::namespace::ChildScan, + ) -> Result { + let request = scan.request()?; + self.scans.fetch_add(1, Ordering::SeqCst); + tokio::time::sleep(std::time::Duration::from_millis( + self.scan_delay_ms.load(Ordering::SeqCst), + )) + .await; + let snapshot = self.values.load_full(); + let mut candidates = snapshot.iter().filter(|(key, _)| { + *key >= request.start.as_ref().unwrap() + && *key < request.end.as_ref().unwrap() + && request + .continuation + .as_ref() + .map_or(true, |cursor| *key > &cursor.last_key) + }); + let items: Vec<_> = candidates + .by_ref() + .take(request.max_items) + .map(|(key, value)| crowdb_protocol::chunk_kv::RpcValue { + key: key.clone(), + value: value.bytes.clone(), + revision: value.revision, + }) + .collect(); + let continuation = candidates + .next() + .map(|_| crowdb_chunk_kv_client::MultiScanContinuation { + direction: request.direction, + original_start: request.start, + original_end: request.end, + last_key: items.last().unwrap().key.clone(), + catalog_generation: 1, + }); + Ok(crowdb_chunk_kv_client::MultiScanPage { + items, + continuation, + terminal_failure: None, + }) + } + + async fn delete_mapping( + &self, + key: &[u8], + expected: &[u8], + identity: ClientRequestId, + ) -> Result { + identity.validate().unwrap(); + loop { + let current = self.values.load_full(); + let previous = current.get(key); + if previous.map(|value| value.bytes.as_slice()) != Some(expected) { + return Ok(CasOutcome::Conflict(previous.cloned())); + } + let revision = previous.unwrap().revision + 1; + let mut next = (*current).clone(); + next.remove(key); + if Arc::ptr_eq(¤t, &self.values.compare_and_swap(¤t, Arc::new(next))) { + return Ok(CasOutcome::Applied(revision)); + } + } + } +} + +#[async_trait] +impl CatalogStore for TestStore { + async fn get(&self, key: &[u8]) -> Result, StoreError> { + let value = self.values.load().get(key).cloned(); + if matches!( + crowdb_access_iceberg::key::IcebergKey::decode(key), + Ok(crowdb_access_iceberg::key::IcebergKey::Catalog { + scope: crowdb_access_iceberg::key::CatalogScope::File, + .. + }) + ) && self.pause_file_read.swap(false, Ordering::SeqCst) + { + self.file_read_entered.notify_one(); + self.file_read_release.notified().await; + } + let delay = self.read_delay_ms.load(Ordering::SeqCst); + if delay != 0 { + tokio::time::sleep(std::time::Duration::from_millis(delay)).await; + } + Ok(value) + } + async fn compare_exchange( + &self, + key: &[u8], + expected: Option<&[u8]>, + value: &[u8], + identity: ClientRequestId, + ) -> Result { + identity.validate().unwrap(); + if matches!( + crowdb_access_iceberg::key::IcebergKey::decode(key), + Ok(crowdb_access_iceberg::key::IcebergKey::Catalog { + scope: crowdb_access_iceberg::key::CatalogScope::TableHead, + .. + }) + ) && self.pause_head_cas.swap(false, Ordering::SeqCst) + { + self.head_cas_entered.notify_one(); + self.head_cas_release.notified().await; + } + loop { + let current = self.values.load_full(); + let previous = current.get(key); + if previous.map(|value| value.bytes.as_slice()) != expected { + return Ok(CasOutcome::Conflict(previous.cloned())); + } + let revision = previous.map_or(1, |value| value.revision + 1); + let mut next = (*current).clone(); + next.insert( + key.to_vec(), + StoredValue { + bytes: value.to_vec(), + revision, + }, + ); + if Arc::ptr_eq(¤t, &self.values.compare_and_swap(¤t, Arc::new(next))) { + let mode = self.lose_reply_kind.load(Ordering::SeqCst); + let lose = mode == 1 + || match crowdb_access_iceberg::key::IcebergKey::decode(key) { + Ok(crowdb_access_iceberg::key::IcebergKey::Catalog { scope, .. }) => { + (mode == 2 + && scope == crowdb_access_iceberg::key::CatalogScope::NamespaceAuthority) + || (mode == 3 && scope == crowdb_access_iceberg::key::CatalogScope::Operation) + || (mode == 5 && scope == crowdb_access_iceberg::key::CatalogScope::TableHead) + || (mode == 4 + && scope + == crowdb_access_iceberg::key::CatalogScope::TableCommitOperation + && matches!(crowdb_access_iceberg::key::IcebergKey::decode(key).and_then(|key| + crowdb_access_iceberg::record::StorageRecord::decode(&key, value)), + Ok(crowdb_access_iceberg::record::StorageRecord::TableCommitOperation(operation)) + if operation.phase == crowdb_access_iceberg::commit::TableCommitPhase::Rejected)) + } + _ => false, + }; + if lose && self.lose_reply_kind.swap(0, Ordering::SeqCst) != 0 { + return Err(StoreError::Response); + } + return Ok(CasOutcome::Applied(revision)); + } + } + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_table_http.rs b/app/crowdb-access-server/tests/common/iceberg_table_http.rs new file mode 100644 index 000000000..148198600 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_table_http.rs @@ -0,0 +1,282 @@ +use std::{sync::Arc, time::Duration}; + +use crowdb_access_iceberg::{ + catalog::{CatalogContext, CatalogRepository, ClearBounds, ManagementPrivilege, StoredValue}, + file::{ContentFormat, FileContent, FileKind, FileRecord, FileRepository, TableLocation}, + key::{FileId, IcebergKey, NamespaceId, OperationId, TableId}, + namespace::{ + NamespaceCreateRequest, NamespaceCreator, NamespaceIdentifier, NamespaceProperties, + NamespaceRepository, + }, + operation::{ManagementAction, ManagementRequest, RequestIdentity}, + record::StorageRecord, + table::{head_key, name_key, TableHead, TableLifecycle, TableMapping, TableMappingState}, + wire::BearerAuthenticator, +}; +use crowdb_access_server::iceberg::{serve, IcebergHttpService}; +use serde_json::json; +use sha2::{Digest, Sha256}; + +use crate::{blocks::TestFileBlocks, common::TestStore}; + +pub struct TestTableHttp { + pub store: Arc, + pub service: Arc, + pub context: CatalogContext, + pub namespace: NamespaceId, + address: std::net::SocketAddr, + stop: tokio::sync::oneshot::Sender<()>, + server: tokio::task::JoinHandle<()>, +} + +impl TestTableHttp { + pub fn endpoint(&self) -> String { + format!("http://{}", self.address) + } + + pub async fn new() -> Self { + Self::start(false, false, 0x3fff).await + } + + pub async fn writable() -> Self { + Self::start(true, false, 0x3fff).await + } + + pub async fn vending() -> Self { + Self::start(true, true, 0x3fff).await + } + + pub async fn with_capabilities(bits: u16) -> Self { + Self::start(false, false, bits).await + } + + pub async fn writable_with_capabilities(bits: u16) -> Self { + Self::start(true, false, bits).await + } + + pub async fn vending_with_capabilities(bits: u16) -> Self { + Self::start(true, true, bits).await + } + + async fn start(writable: bool, vending: bool, bits: u16) -> Self { + let store = Arc::new(TestStore::default()); + let repository = Arc::new( + CatalogRepository::new( + store.clone(), + ClearBounds { + delegated_access_ms: if vending { 900_000 } else { 0 }, + ..ClearBounds::default() + }, + ) + .unwrap(), + ); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: 100, + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "catalog".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + 100, + ) + .await + .unwrap(); + crate::common::activate_bits(&repository, bits).await; + let context = repository.status().await.unwrap().0.context; + let identifier = NamespaceIdentifier::new(vec!["analytics".into()]).unwrap(); + NamespaceCreator::new(store.clone()) + .create(&NamespaceCreateRequest { + context, + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: 100, + }, + principal: "writer".into(), + identifier: identifier.clone(), + properties: NamespaceProperties::default(), + }) + .await + .unwrap(); + let namespace = NamespaceRepository::new(store.clone()) + .load(context, &identifier) + .await + .unwrap() + .unwrap() + .namespace; + let auth = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)) + .unwrap(); + let service = IcebergHttpService::new(repository, auth, Duration::from_secs(2)) + .with_namespaces(store.clone()) + .unwrap(); + let blocks = Arc::new(TestFileBlocks::default()); + let service = if writable { + service.with_tables(store.clone(), blocks).unwrap() + } else { + service.with_table_reads_for_tests(store.clone(), blocks).unwrap() + }; + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let service = Arc::new(if vending { + service + .with_table_credentials(store.clone(), format!("http://{address}")) + .unwrap() + } else { + service + }); + let (stop, stopped) = tokio::sync::oneshot::channel(); + let listener_service = service.clone(); + let server = tokio::spawn(async move { + serve(listener, listener_service, async { + let _ = stopped.await; + }) + .await + .unwrap(); + }); + Self { + store, + service, + context, + namespace, + address, + stop, + server, + } + } + + pub async fn install(&self, name: &str) -> (TableHead, Vec) { + let location = TableLocation { + catalog: self.context.catalog, + table: TableId::random(), + }; + let snapshots: Vec<_> = [(10,1),(20,2),(30,3)].into_iter().map(|(snapshot, sequence)| json!({ + "snapshot-id":snapshot,"sequence-number":sequence,"timestamp-ms":1000,"schema-id":0, + "summary":{"operation":"append"},"manifest-list":location.file(&format!("metadata/{snapshot}.avro")).unwrap().to_string() + })).collect(); + let metadata = json!({ + "format-version":3,"table-uuid":"12345678-1234-1234-1234-123456789abc", + "location":location.to_string(),"last-updated-ms":1000,"last-column-id":1, + "schemas":[{"type":"struct","schema-id":0,"fields":[{"id":1,"name":"id","type":"long","required":true}]}], + "current-schema-id":0,"partition-specs":[{"spec-id":0,"fields":[]}],"default-spec-id":0, + "last-partition-id":999,"sort-orders":[{"order-id":0,"fields":[]}],"default-sort-order-id":0, + "last-sequence-number":3,"next-row-id":0,"current-snapshot-id":20,"snapshots":snapshots, + "refs":{"main":{"type":"branch","snapshot-id":20},"tag":{"type":"tag","snapshot-id":30}} + }); + let mut text = serde_json::to_string_pretty(&metadata).unwrap(); + text.pop(); + text.push_str(",\"future-number\":123456789012345678901234567890}"); + let bytes = text.into_bytes(); + let head = TableHead { + catalog: self.context.catalog, + table: location.table, + namespace: self.namespace, + name: name.into(), + name_epoch: 1, + lifecycle: TableLifecycle::Ready, + generation: 1, + metadata_file: FileId::random(), + metadata_location: location.file("metadata/one.json").unwrap(), + metadata_digest: Sha256::digest(&bytes).into(), + format_version: 3, + table_uuid: Some("12345678-1234-1234-1234-123456789abc".parse().unwrap()), + operation_fence: 1, + pending_operation: None, + }; + FileRepository::new(self.store.clone()) + .publish( + self.context, + &FileRecord { + file: head.metadata_file, + location: head.metadata_location.clone(), + kind: FileKind::Metadata, + format: ContentFormat::Json, + length: bytes.len() as u64, + digest: head.metadata_digest, + content: FileContent::select_inline(FileKind::Metadata, &bytes).unwrap(), + hint: None, + }, + ) + .await + .unwrap(); + self.put( + &head_key(head.catalog, head.table), + &StorageRecord::TableHead(Box::new(head.clone())), + ); + self.put( + &name_key(head.catalog, head.namespace, name).unwrap(), + &StorageRecord::TableMapping(TableMapping { + catalog: head.catalog, + namespace: head.namespace, + name: name.into(), + table: head.table, + name_epoch: 1, + operation: OperationId::random(), + state: TableMappingState::Published, + }), + ); + (head, bytes) + } + + pub fn put(&self, key: &IcebergKey, record: &StorageRecord) { + let mut values = (**self.store.values.load()).clone(); + values.insert( + key.encode().unwrap(), + StoredValue { + bytes: record.encode().unwrap(), + revision: 1, + }, + ); + self.store.values.store(Arc::new(values)); + } + + pub async fn request( + &self, + method: reqwest::Method, + path: &str, + role: &str, + etag: Option<&str>, + ) -> reqwest::Response { + let client = reqwest::Client::builder() + .timeout(Duration::from_secs(3)) + .build() + .unwrap(); + let mut request = client + .request(method, format!("http://{}{path}", self.address)) + .bearer_auth(role.repeat(32)); + if let Some(etag) = etag { + request = request.header("If-None-Match", etag); + } + request.send().await.unwrap() + } + + pub async fn finish(self) { + self.stop.send(()).unwrap(); + self.server.await.unwrap(); + } + + pub async fn post( + &self, + path: &str, + role: &str, + key: Option<&str>, + body: &serde_json::Value, + ) -> reqwest::Response { + let mut request = reqwest::Client::new() + .post(format!("http://{}{path}", self.address)) + .bearer_auth(role.repeat(32)) + .header("content-type", "application/json") + .body(serde_json::to_vec(body).unwrap()); + if let Some(key) = key { + request = request.header("idempotency-key", key); + } + request.send().await.unwrap() + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_upload.rs b/app/crowdb-access-server/tests/common/iceberg_upload.rs new file mode 100644 index 000000000..f61d6f917 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_upload.rs @@ -0,0 +1,103 @@ +use std::collections::{BTreeMap, VecDeque}; +use std::pin::Pin; +use std::sync::{ + atomic::{AtomicBool, AtomicUsize, Ordering}, + Arc, +}; +use std::task::{Context, Poll}; + +use arc_swap::ArcSwap; +use async_trait::async_trait; +use crowdb_access_iceberg::file::{ChunkRoot, FileBlockStore, FileIdentity, FileIoError, TableLocation}; +use crowdb_access_iceberg::key::{CatalogId, FileId, TableId}; +use crowdb_protocol::common::ChunkId; +use hyper::body::{Body, Bytes, Frame}; +use sha2::{Digest, Sha256}; + +pub fn owner() -> FileIdentity { + FileIdentity { + table: TableLocation { + catalog: CatalogId::random(), + table: TableId::random(), + }, + file: FileId::random(), + } +} + +pub struct TestUploadBody { + pub frames: VecDeque, std::io::Error>>, + pub polls: Arc, +} + +impl TestUploadBody { + pub fn new(bytes: &[u8], frame_bytes: usize) -> Self { + Self { + frames: bytes + .chunks(frame_bytes) + .map(|bytes| Ok(Frame::data(Bytes::copy_from_slice(bytes)))) + .collect(), + polls: Arc::new(AtomicUsize::new(0)), + } + } +} + +impl Body for TestUploadBody { + type Data = Bytes; + type Error = std::io::Error; + + fn poll_frame( + mut self: Pin<&mut Self>, + _context: &mut Context<'_>, + ) -> Poll, Self::Error>>> { + self.polls.fetch_add(1, Ordering::SeqCst); + Poll::Ready(self.frames.pop_front()) + } +} + +#[derive(Default)] +pub struct TestUploadBlocks { + pub values: ArcSwap>>>, + pub writes: AtomicUsize, + pub max_input: AtomicUsize, + pub fail: AtomicBool, + pub pause: AtomicBool, + pub entered: tokio::sync::Notify, + pub release: tokio::sync::Notify, +} + +#[async_trait] +impl FileBlockStore for TestUploadBlocks { + async fn put(&self, _owner: FileIdentity, height: u8, bytes: &[u8]) -> Result { + self.max_input.fetch_max(bytes.len(), Ordering::SeqCst); + let index = self.writes.fetch_add(1, Ordering::SeqCst) as u64 + 1; + self.values.rcu(|values| { + let mut next = (**values).clone(); + next.insert(index, Arc::new(bytes.to_vec())); + next + }); + if self.pause.load(Ordering::SeqCst) { + self.entered.notify_one(); + self.release.notified().await; + } + if self.fail.load(Ordering::SeqCst) { + return Err(FileIoError::Bounds); + } + Ok(ChunkRoot { + chunk: ChunkId { high: 1, low: index }, + offset: 0, + physical_length: bytes.len() as u64 + 64, + logical_offset: 0, + logical_length: bytes.len() as u64, + height, + digest: Sha256::digest(bytes).into(), + }) + } + + async fn read(&self, root: &ChunkRoot) -> Result, FileIoError> { + self.values + .load() + .get(&root.chunk.low) + .map(|bytes| bytes.as_ref().clone()) + .ok_or(FileIoError::Bounds) + } +} diff --git a/app/crowdb-access-server/tests/iceberg_auth_test.rs b/app/crowdb-access-server/tests/iceberg_auth_test.rs new file mode 100644 index 000000000..ec8abc8a9 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_auth_test.rs @@ -0,0 +1,35 @@ +use std::process::Command; + +#[test] +fn writer_configuration_fails_before_backend_connection() { + for writer in [ + None, + Some("short".into()), + Some("r".repeat(32)), + Some("m".repeat(32)), + Some("c".repeat(32)), + ] { + let mut command = Command::new(env!("CARGO_BIN_EXE_crowdb-iceberg")); + command + .env("CROWDB_MANAGEMENT_SEEDS", "127.0.0.1:1") + .env("CROWDB_ICEBERG_READ_TOKEN", "r".repeat(32)) + .env("CROWDB_ICEBERG_MANAGE_TOKEN", "m".repeat(32)) + .env("CROWDB_ICEBERG_CLEAR_TOKEN", "c".repeat(32)) + .env_remove("CROWDB_ICEBERG_WRITE_TOKEN") + .arg("serve"); + if let Some(token) = &writer { + command.env("CROWDB_ICEBERG_WRITE_TOKEN", token); + } + let output = command.output().unwrap(); + assert!(!output.status.success()); + let error = String::from_utf8_lossy(&output.stderr); + if writer.is_none() { + assert!( + error.contains("CROWDB_ICEBERG_WRITE_TOKEN must be set"), + "{error}" + ); + } else { + assert!(error.contains("invalid or oversized text field"), "{error}"); + } + } +} diff --git a/app/crowdb-access-server/tests/iceberg_commit_crash_test.rs b/app/crowdb-access-server/tests/iceberg_commit_crash_test.rs new file mode 100644 index 000000000..42bee93d7 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_commit_crash_test.rs @@ -0,0 +1,204 @@ +#![cfg(feature = "iceberg-e2e")] + +#[path = "common/iceberg_commit_case.rs"] +mod case; +#[path = "common/iceberg_commit_child.rs"] +mod child; +#[path = "common/iceberg_stack.rs"] +#[allow(dead_code)] +mod common; +#[path = "common/iceberg_commit_fault.rs"] +mod fault; +#[path = "common/iceberg_commit_loser.rs"] +mod loser; +#[path = "common/iceberg_process.rs"] +#[allow(dead_code)] +mod process; + +use std::collections::BTreeSet; +use std::path::Path; + +use crowdb_access_iceberg::{ + catalog::{CatalogContext, CatalogRepository, ClearBounds, ManagementPrivilege}, + key::OperationId, + namespace::{NamespaceIdentifier, NamespaceRepository}, + operation::{ManagementAction, ManagementRequest, RequestIdentity}, + table::TableRepository, +}; +use serde_json::{json, Value}; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "test-only child listener invoked by native crash matrix"] +async fn native_fault_listener_child() { + child::run().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires native storage processes and kills listener subprocesses at every durable boundary"] +async fn native_table_publication_recovers_before_and_after_every_durable_write() { + let mut stack = common::TestIcebergStack::start().await; + let context = initialize(&stack).await; + let directory = stack + .cluster + .runtime_mut() + .service_dir("iceberg", "commit-faults") + .unwrap(); + let setup = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + case::success( + &format!("http://{}", setup.address), + "/v1/namespaces", + &json!({"namespace":["analytics"]}), + ) + .await; + drop(setup); + for kind in ["create", "stage", "publish-stage", "update"] { + let count = baseline(&stack, &directory, kind).await; + let mut labels = BTreeSet::new(); + let mut head_offset = None; + for offset in 1..=count { + for after in [false, true] { + let name = format!("{kind}-{offset}-{after}"); + let setup = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let case = + case::TestCommitCase::prepare(&format!("http://{}", setup.address), kind, name).await; + drop(setup); + let marker = directory.join(format!("{kind}-{offset}-{after}.json")); + let mut child = + child::TestCommitChild::start(&stack.cluster.mgmt_endpoints, marker, offset, after).await; + let endpoint = format!("http://{}", child.address); + let path = case.path.clone(); + let identity = case.identity.clone(); + let body = case.body.clone(); + let mut request = + tokio::spawn(async move { case::post(&endpoint, &path, &identity, &body).await }); + let boundary = tokio::select! { + boundary = child.paused() => boundary, + result = &mut request => { + let response = result.unwrap().unwrap(); + panic!("{kind} boundary {offset}/{count} returned early: {} {}", response.status(), response.text().await.unwrap()); + } + }; + println!( + "kill {kind} boundary {offset}/{count} after={after}: {}", + boundary["label"] + ); + labels.insert(boundary["label"].as_str().unwrap().to_owned()); + if kind == "update" && boundary["label"] == "head-2" { + head_offset.get_or_insert(offset); + } + drop(child); + assert!( + request.await.unwrap().is_err(), + "killed request must lose its response" + ); + let recovery = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + case.replay(&format!("http://{}", recovery.address)).await; + verify_generation(&stack, context, &case).await; + drop(recovery); + } + } + assert!(labels + .iter() + .any(|label| label.starts_with("create-") || label.starts_with("commit-"))); + if kind != "stage" { + assert!( + labels.contains("file-block"), + "candidate bytes must use native chunks: {labels:?}" + ); + assert!(labels.iter().any(|label| label.starts_with("head-"))); + } + if let Some(offset) = head_offset { + loser::verify(&stack, context, &directory, offset).await; + } + } +} + +async fn baseline(stack: &common::TestIcebergStack, directory: &Path, kind: &str) -> usize { + let setup = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let case = case::TestCommitCase::prepare( + &format!("http://{}", setup.address), + kind, + format!("baseline-{kind}"), + ) + .await; + drop(setup); + let child = child::TestCommitChild::start( + &stack.cluster.mgmt_endpoints, + directory.join(format!("baseline-{kind}.json")), + usize::MAX, + false, + ) + .await; + let response = case::post( + &format!("http://{}", child.address), + &case.path, + &case.identity, + &case.body, + ) + .await + .unwrap(); + let status = response.status(); + let body = response.text().await.unwrap(); + assert_eq!(status, 200, "{body}"); + let marker: Value = serde_json::from_slice(&std::fs::read(&child.marker).unwrap()).unwrap(); + usize::try_from(marker["index"].as_u64().unwrap()).unwrap() +} + +async fn verify_generation( + stack: &common::TestIcebergStack, + context: CatalogContext, + case: &case::TestCommitCase, +) { + let store = stack.store().await; + let parent = NamespaceRepository::new(store.clone()) + .load( + context, + &NamespaceIdentifier::new(vec!["analytics".into()]).unwrap(), + ) + .await + .unwrap() + .unwrap(); + let table = TableRepository::new(store) + .select(context, parent.namespace, &case.name) + .await + .unwrap(); + if case.staged { + assert!(table.is_none()); + } else { + assert_eq!(table.unwrap().head.generation, case.generation); + } +} + +async fn initialize(stack: &common::TestIcebergStack) -> CatalogContext { + let repository = CatalogRepository::new( + stack.store().await, + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + ) + .unwrap(); + let now = common::now_ms(); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now, + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "commit-crashes".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + now, + ) + .await + .unwrap(); + common::activate(&repository).await; + repository.status().await.unwrap().0.context +} diff --git a/app/crowdb-access-server/tests/iceberg_commit_sdk_test.rs b/app/crowdb-access-server/tests/iceberg_commit_sdk_test.rs new file mode 100644 index 000000000..566ca76ce --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_commit_sdk_test.rs @@ -0,0 +1,148 @@ +#![cfg(feature = "iceberg-e2e")] + +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +use crowdb_access_iceberg::{ + commit::{TableCommitJournal, TableCommitPhase}, + key::{IcebergKey, OperationId}, + record::StorageRecord, +}; +use serde_json::{json, Value}; +use std::{sync::atomic::Ordering, time::Duration}; + +const TABLE: &str = "/v1/namespaces/analytics/tables/events"; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires Maven and pinned Apache Iceberg Java dependencies"] +async fn official_commit_loses_real_head_cas_without_rebase_and_replays_conflict() { + let fixture = fixture::TestTableHttp::writable().await; + let initial = value( + fixture + .post( + "/v1/namespaces/analytics/tables", + "w", + None, + &json!({"name":"events", "schema":{"type":"struct", "schema-id":0, + "fields":[{"id":1,"name":"id","type":"long","required":true}]}}), + ) + .await, + ) + .await; + let identity = request_key(); + let operation: OperationId = identity.parse().unwrap(); + fixture.store.pause_head_cas.store(true, Ordering::SeqCst); + let endpoint = fixture.endpoint(); + let client = tokio::task::spawn_blocking(move || run_sdk(&endpoint, &identity)); + tokio::time::timeout(Duration::from_secs(60), fixture.store.head_cas_entered.notified()) + .await + .expect("SDK did not reach head publication"); + let journal = TableCommitJournal::new(fixture.store.clone()); + let paused = journal.load(fixture.context, operation).await.unwrap().unwrap(); + assert_eq!(paused.phase, TableCommitPhase::Publishing); + assert_eq!( + paused.before.metadata_location.to_string(), + initial["metadata-location"] + ); + let candidate = paused.candidate.as_ref().unwrap(); + assert_eq!(candidate.generation, paused.before.generation + 1); + let winner = value( + fixture + .post( + TABLE, + "w", + None, + &json!({"requirements":[], "updates":[ + {"action":"set-properties", "updates":{"winner-only":"visible"}}]}), + ) + .await, + ) + .await; + fixture.store.head_cas_release.notify_one(); + assert!( + client.await.unwrap().success(), + "official SDK CAS conflict acceptance failed" + ); + let rejected = journal.load(fixture.context, operation).await.unwrap().unwrap(); + assert_eq!(rejected.phase, TableCommitPhase::Rejected); + assert_eq!(rejected.before, paused.before); + assert_eq!(rejected.candidate, paused.candidate); + assert_eq!(rejected.outcome.as_ref().unwrap().status, 409); + let commits: Vec<_> = fixture + .store + .values + .load() + .iter() + .filter_map(|(key, stored)| { + let key = IcebergKey::decode(key).unwrap(); + match StorageRecord::decode(&key, &stored.bytes).unwrap() { + StorageRecord::TableCommitOperation(commit) => Some(commit), + _ => None, + } + }) + .collect(); + assert_eq!(commits.len(), 2, "replay and changed input create no new commit"); + let completed = commits + .iter() + .find(|commit| commit.phase == TableCommitPhase::Complete) + .unwrap(); + assert_eq!(completed.before, paused.before); + assert_eq!( + completed.candidate.as_ref().unwrap().generation, + candidate.generation + ); + assert_ne!( + candidate.metadata_location.to_string(), + winner["metadata-location"] + ); + let selected = value(fixture.request(reqwest::Method::GET, TABLE, "r", None).await).await; + assert_eq!(selected, winner); + let listed = value( + fixture + .request(reqwest::Method::GET, "/v1/namespaces/analytics/tables", "r", None) + .await, + ) + .await; + assert_eq!( + listed["identifiers"], + json!([{"namespace":["analytics"],"name":"events"}]) + ); + fixture.finish().await; +} + +fn request_key() -> String { + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(); + format!("{:08x}-{:04x}-7000-8000-000000000001", now >> 16, now & 0xffff) +} + +fn run_sdk(endpoint: &str, identity: &str) -> std::process::ExitStatus { + let maven = std::env::var_os("CROWDB_ICEBERG_E2E_MVN").unwrap_or_else(|| "mvn".into()); + std::process::Command::new("timeout") + .arg("60") + .arg(maven) + .args(["-o", "--batch-mode", "--no-transfer-progress", "-f"]) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_java/pom.xml" + )) + .args(["compile", "exec:java", "-Dexec.mainClass=TestIcebergCommitRace"]) + .arg(format!("-Dexec.args={endpoint} {identity}")) + .status() + .unwrap() +} + +async fn value(response: reqwest::Response) -> Value { + let status = response.status(); + let text = response.text().await.unwrap(); + assert_eq!(status.as_u16(), 200, "{text}"); + serde_json::from_str(&text).unwrap() +} diff --git a/app/crowdb-access-server/tests/iceberg_file_admission_test.rs b/app/crowdb-access-server/tests/iceberg_file_admission_test.rs new file mode 100644 index 000000000..5b8da7acc --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_admission_test.rs @@ -0,0 +1,287 @@ +#[path = "common/iceberg_file_blocks.rs"] +mod blocks; +#[path = "common/iceberg_upload.rs"] +mod upload; + +use std::sync::{atomic::Ordering, Arc}; + +use crowdb_access_iceberg::catalog::CatalogContext; +use crowdb_access_iceberg::file::{ + ByteRange, FileGrant, FileIdentity, FileOperation, FileOperations, MultipartAdmissionLimits, + MultipartAdmissionRecord, MultipartCredit, MultipartLimits, MultipartPhase, MultipartSession, + TableLocation, +}; +use crowdb_access_iceberg::key::{CatalogId, FileId, OperationId, TableId}; +use crowdb_access_server::iceberg::{ + FileAdmissionError, FileRequest, FileResponseBudget, FileServiceLimits, FileTransferAdmission, + FileUploadBudget, +}; +use http_body_util::BodyExt; +use hyper::Method; + +fn grant(operations: &[FileOperation]) -> FileGrant { + FileGrant { + context: CatalogContext { + catalog: CatalogId::random(), + activation_epoch: 1, + }, + table: TableId::random(), + principal: [5; 32], + nonce: OperationId::random(), + issued_ms: 100, + expires_ms: 200, + operations: FileOperations::new(operations).unwrap(), + max_request_bytes: 5, + max_file_bytes: 20, + } +} + +fn location(grant: &FileGrant) -> crowdb_access_iceberg::file::FileLocation { + TableLocation { + catalog: grant.context.catalog, + table: grant.table, + } + .file("data/a.parquet") + .unwrap() +} + +fn request(grant: &FileGrant, method: &Method, suffix: &str) -> FileRequest { + let uri: hyper::Uri = format!( + "/{}/{}{}", + location(grant).table().bucket(), + location(grant).object_key(), + suffix + ) + .parse() + .unwrap(); + FileRequest::parse(method, &uri).unwrap() +} + +fn service() -> FileServiceLimits { + FileServiceLimits { + max_request_bytes: 7, + max_file_bytes: 12, + max_part_bytes: 4, + max_staged_bytes: 30, + } +} + +fn session(grant: &FileGrant) -> MultipartSession { + let table = location(grant).table(); + MultipartSession { + context: grant.context, + upload: OperationId::random(), + owner: FileIdentity { + table, + file: FileId::random(), + }, + location: location(grant), + principal: grant.principal, + revision: 1, + created_ms: 100, + expires_ms: 200, + limits: MultipartLimits { + max_parts: 10, + max_part_bytes: 4, + max_file_bytes: 12, + max_staged_bytes: 20, + ttl_ms: 100, + }, + phase: MultipartPhase::Open, + part_count: 0, + staged_bytes: 0, + completion: None, + published: None, + pending: None, + credit: None, + } +} + +fn policy(grant: &FileGrant) -> MultipartAdmissionRecord { + MultipartAdmissionRecord { + context: grant.context, + policy: OperationId::random(), + revision: 1, + limits: MultipartAdmissionLimits { + max_sessions: 2, + max_reserved_bytes: 25, + }, + sessions: 0, + reserved_bytes: 0, + pending: None, + } +} + +#[test] +fn authorization_intersects_grant_operation_scope_and_service_limits() { + let grant = grant(&[FileOperation::Get, FileOperation::Put]); + let get_request = request(&grant, &Method::GET, ""); + let admitted = FileTransferAdmission::authorize(&grant, &get_request, service(), None, 100).unwrap(); + assert!(admitted.check_bytes(5, 12).is_ok()); + assert!(matches!( + admitted.check_bytes(6, 12), + Err(FileAdmissionError::Bounds) + )); + assert!(matches!( + admitted.check_bytes(5, 13), + Err(FileAdmissionError::Bounds) + )); + let mut wrong = get_request.clone(); + wrong.location = TableLocation { + catalog: grant.context.catalog, + table: TableId::random(), + } + .file("data/a.parquet") + .unwrap(); + assert!(FileTransferAdmission::authorize(&grant, &wrong, service(), None, 100).is_err()); + wrong = get_request; + wrong.operation = FileOperation::CreateMultipart; + assert!(FileTransferAdmission::authorize(&grant, &wrong, service(), None, 100).is_err()); + assert!(FileTransferAdmission::authorize( + &grant, + &request(&grant, &Method::GET, ""), + service(), + None, + 200 + ) + .is_err()); + let mut invalid = service(); + invalid.max_part_bytes = 0; + assert!( + FileTransferAdmission::authorize(&grant, &request(&grant, &Method::GET, ""), invalid, None, 100) + .is_err() + ); +} + +#[tokio::test] +async fn authorized_upload_and_range_reads_enforce_declared_and_actual_bytes() { + let owner = upload::owner(); + let mut grant = grant(&[FileOperation::Put, FileOperation::Get]); + grant.context.catalog = owner.table.catalog; + grant.table = owner.table.table; + let put = + FileTransferAdmission::authorize(&grant, &request(&grant, &Method::PUT, ""), service(), None, 101) + .unwrap(); + let store = Arc::new(upload::TestUploadBlocks::default()); + let budget = FileUploadBudget::new(1).unwrap(); + let too_large = upload::TestUploadBody::new(b"123456", 2); + let polls = too_large.polls.clone(); + assert!(put + .receive(&budget, too_large, store.clone(), owner, Some(6), None) + .await + .is_err()); + assert_eq!(polls.load(Ordering::SeqCst), 0); + let too_large = upload::TestUploadBody::new(b"123456", 2); + assert!(put + .receive(&budget, too_large, store.clone(), owner, None, None) + .await + .is_err()); + let tree = put + .receive( + &budget, + upload::TestUploadBody::new(b"12345", 2), + store, + owner, + Some(5), + None, + ) + .await + .unwrap(); + assert_eq!(tree.length, 5); + assert_eq!(budget.active(), 0); + + let read_store = Arc::new(blocks::TestFileBlocks { + bytes: vec![9; 12], + ..Default::default() + }); + let mut record = read_store.record(); + record.location = location(&grant); + let get = + FileTransferAdmission::authorize(&grant, &request(&grant, &Method::GET, ""), service(), None, 101) + .unwrap(); + let responses = FileResponseBudget::new(1).unwrap(); + assert!(get + .read_body(&responses, read_store.clone(), record.clone(), None) + .is_err()); + let range = Some(ByteRange { start: 0, end: 5 }); + let body = get + .read_body(&responses, read_store.clone(), record.clone(), range) + .unwrap(); + assert_eq!(body.collect().await.unwrap().to_bytes().len(), 5); + assert_eq!(responses.active(), 0); + let mut larger = record; + larger.length = 13; + assert!(matches!( + get.read_body(&responses, read_store, larger, range), + Err(FileAdmissionError::Bounds) + )); +} + +#[test] +fn multipart_session_and_global_credit_intersections_fail_closed() { + let grant = grant(&[ + FileOperation::CreateMultipart, + FileOperation::UploadPart, + FileOperation::ListParts, + ]); + let create = request(&grant, &Method::POST, "?uploads"); + let admitted = FileTransferAdmission::authorize(&grant, &create, service(), None, 101).unwrap(); + let mut session = session(&grant); + let mut policy = policy(&grant); + assert!(admitted.check_create(&session, &policy).is_ok()); + policy.sessions = policy.limits.max_sessions; + policy.reserved_bytes = policy.limits.max_reserved_bytes; + assert!(matches!( + admitted.check_create(&session, &policy), + Err(FileAdmissionError::Bounds) + )); + policy.sessions = 1; + policy.reserved_bytes = 10; + assert!(matches!( + admitted.check_create(&session, &policy), + Err(FileAdmissionError::Bounds) + )); + policy.reserved_bytes = 0; + policy.sessions = 0; + session.limits.max_part_bytes = 5; + assert!(matches!( + admitted.check_create(&session, &policy), + Err(FileAdmissionError::Bounds) + )); + session.limits.max_part_bytes = 4; + session.limits.max_file_bytes = 13; + assert!(matches!( + admitted.check_create(&session, &policy), + Err(FileAdmissionError::Bounds) + )); + session.limits.max_file_bytes = 12; + session.limits.max_staged_bytes = 31; + assert!(matches!( + admitted.check_create(&session, &policy), + Err(FileAdmissionError::Bounds) + )); + session.limits.max_staged_bytes = 20; + + let part_request = request( + &grant, + &Method::PUT, + &format!("?uploadId={}&partNumber=1", session.upload), + ); + assert!(FileTransferAdmission::authorize(&grant, &part_request, service(), Some(&session), 101).is_err()); + session.credit = Some(MultipartCredit { + policy: policy.policy, + sequence: 2, + released: false, + }); + let part = + FileTransferAdmission::authorize(&grant, &part_request, service(), Some(&session), 101).unwrap(); + assert!(part.check_bytes(4, 12).is_ok()); + assert!(matches!(part.check_bytes(5, 12), Err(FileAdmissionError::Bounds))); + let list = request(&grant, &Method::GET, &format!("?uploadId={}", session.upload)); + assert!(FileTransferAdmission::authorize(&grant, &list, service(), Some(&session), 101).is_ok()); + session.principal[0] ^= 1; + assert!(FileTransferAdmission::authorize(&grant, &part_request, service(), Some(&session), 101).is_err()); + session.principal = grant.principal; + session.credit.as_mut().unwrap().released = true; + assert!(FileTransferAdmission::authorize(&grant, &part_request, service(), Some(&session), 101).is_err()); +} diff --git a/app/crowdb-access-server/tests/iceberg_file_auth_test.rs b/app/crowdb-access-server/tests/iceberg_file_auth_test.rs new file mode 100644 index 000000000..9e7138530 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_auth_test.rs @@ -0,0 +1,182 @@ +use crowdb_access_iceberg::catalog::CatalogContext; +use crowdb_access_iceberg::file::{ + FileCredentials, FileGrant, FileGrantIssuer, FileOperation, FileOperations, +}; +use crowdb_access_iceberg::key::{CatalogId, OperationId, TableId}; +use crowdb_access_s3::auth::RawAuthRequest; +use crowdb_access_server::iceberg::authenticate_file_request; +use hmac::{Hmac, Mac}; +use hyper::{header::HeaderValue, Request}; +use sha2::{Digest, Sha256}; +use std::fmt::Write as _; + +const NOW: u64 = 1_704_067_200_000; +const DATE: &str = "20240101T000000Z"; +const HASH: &str = "UNSIGNED-PAYLOAD"; + +fn credentials() -> (FileGrantIssuer, FileCredentials) { + let issuer = FileGrantIssuer::new([42; 32], 60_000).unwrap(); + let credentials = issuer + .issue(FileGrant { + context: CatalogContext { + catalog: CatalogId::random(), + activation_epoch: 1, + }, + table: TableId::random(), + principal: [7; 32], + nonce: OperationId::random(), + issued_ms: NOW, + expires_ms: NOW + 60_000, + operations: FileOperations::new(&[FileOperation::Get]).unwrap(), + max_request_bytes: 1024, + max_file_bytes: 4096, + }) + .unwrap(); + (issuer, credentials) +} + +fn mac(key: &[u8], value: &str) -> Vec { + let mut signer = Hmac::::new_from_slice(key).unwrap(); + signer.update(value.as_bytes()); + signer.finalize().into_bytes().to_vec() +} + +fn hex(bytes: &[u8]) -> String { + let mut result = String::with_capacity(bytes.len() * 2); + for byte in bytes { + write!(&mut result, "{byte:02x}").unwrap(); + } + result +} + +fn signature(credentials: &FileCredentials, canonical: &str) -> String { + let date = mac( + format!("AWS4{}", credentials.secret_access_key()).as_bytes(), + "20240101", + ); + let region = mac(&date, "us-east-1"); + let service = mac(®ion, "s3"); + let key = mac(&service, "aws4_request"); + hex(&mac( + &key, + &format!( + "AWS4-HMAC-SHA256\n{DATE}\n20240101/us-east-1/s3/aws4_request\n{}", + hex(&Sha256::digest(canonical)) + ), + )) +} + +fn signed(credentials: &FileCredentials, presigned: bool) -> Request<()> { + let mut request = Request::builder() + .method("GET") + .uri("/bucket/key") + .header("host", "localhost"); + if presigned { + let query = format!("X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential={}%2F20240101%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Date={DATE}&X-Amz-Expires=60&X-Amz-Security-Token={}&X-Amz-SignedHeaders=host", credentials.access_key_id(), credentials.session_token()); + let canonical = format!("GET\n/bucket/key\n{query}\nhost:localhost\n\nhost\n{HASH}"); + request = request.uri(format!( + "/bucket/key?{query}&X-Amz-Signature={}", + signature(credentials, &canonical) + )); + } else { + let names = "host;x-amz-content-sha256;x-amz-date;x-amz-security-token"; + let canonical = format!("GET\n/bucket/key\n\nhost:localhost\nx-amz-content-sha256:{HASH}\nx-amz-date:{DATE}\nx-amz-security-token:{}\n\n{names}\n{HASH}", credentials.session_token()); + request = request.header("x-amz-content-sha256", HASH).header("x-amz-date", DATE) + .header("x-amz-security-token", credentials.session_token()) + .header("authorization", format!("AWS4-HMAC-SHA256 Credential={}/20240101/us-east-1/s3/aws4_request, SignedHeaders={names}, Signature={}", credentials.access_key_id(), signature(credentials, &canonical))); + } + request.body(()).unwrap() +} + +fn verify(issuer: &FileGrantIssuer, context: CatalogContext, request: &Request<()>, now: u64) -> bool { + authenticate_file_request( + issuer, + context, + RawAuthRequest::from_parts(request.method(), request.uri(), request.headers()), + "us-east-1", + now, + ) + .is_ok() +} + +#[test] +fn native_file_auth_verifies_header_and_presigned_requests_with_exact_grant_expiry() { + let (issuer, credentials) = credentials(); + let context = credentials.grant().context; + for presigned in [false, true] { + let request = signed(&credentials, presigned); + let grant = authenticate_file_request( + &issuer, + context, + RawAuthRequest::from_parts(request.method(), request.uri(), request.headers()), + "us-east-1", + NOW, + ) + .unwrap(); + assert_eq!(&grant, credentials.grant()); + assert!(!verify(&issuer, context, &request, NOW - 1)); + assert!(!verify(&issuer, context, &request, NOW + 60_000)); + assert!(!verify( + &issuer, + CatalogContext { + activation_epoch: 2, + ..context + }, + &request, + NOW + )); + assert!(!verify( + &FileGrantIssuer::new([43; 32], 60_000).unwrap(), + context, + &request, + NOW + )); + } +} + +#[test] +fn native_file_tokens_are_not_bearer_credentials_and_signatures_bind_request_bytes() { + let (issuer, credentials) = credentials(); + let context = credentials.grant().context; + let mut request = signed(&credentials, false); + *request.uri_mut() = "/bucket/other".parse().unwrap(); + assert!(!verify(&issuer, context, &request, NOW)); + let mut request = signed(&credentials, false); + request.headers_mut().remove("authorization"); + assert!(!verify(&issuer, context, &request, NOW)); + let mut request = signed(&credentials, false); + *request.method_mut() = hyper::Method::DELETE; + assert!(!verify(&issuer, context, &request, NOW)); +} + +#[test] +fn ambiguous_and_oversized_authentication_fails_before_signature_work() { + let (issuer, credentials) = credentials(); + let context = credentials.grant().context; + for name in [ + "authorization", + "host", + "x-amz-date", + "x-amz-content-sha256", + "x-amz-security-token", + ] { + let mut request = signed(&credentials, false); + let value = request.headers()[name].clone(); + request.headers_mut().append(name, value); + assert!(!verify(&issuer, context, &request, NOW)); + } + let mut request = signed(&credentials, false); + request + .headers_mut() + .insert("extra", HeaderValue::from_str(&"x".repeat(16384)).unwrap()); + assert!(!verify(&issuer, context, &request, NOW)); + for suffix in [ + "&X-Amz-Expires=60", + "&%58-Amz-Expires=60", + "&X-Amz-Security-Token=wrong", + ] { + let mut request = signed(&credentials, true); + *request.uri_mut() = format!("{}{suffix}", request.uri()).parse().unwrap(); + assert!(!verify(&issuer, context, &request, NOW)); + } +} diff --git a/app/crowdb-access-server/tests/iceberg_file_body_test.rs b/app/crowdb-access-server/tests/iceberg_file_body_test.rs new file mode 100644 index 000000000..b79295922 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_body_test.rs @@ -0,0 +1,95 @@ +#[path = "common/iceberg_file_blocks.rs"] +mod blocks; + +use blocks::TestFileBlocks; +use crowdb_access_iceberg::file::ByteRange; +use crowdb_access_server::iceberg::{FileBodyError, FileResponseBudget}; +use http_body_util::BodyExt; +use hyper::body::Body; +use std::sync::{atomic::Ordering, Arc}; + +#[tokio::test] +async fn file_http_body_pulls_bounded_frames_and_releases_credit_on_completion() { + let store = Arc::new(TestFileBlocks { + bytes: vec![17; 70_000], + ..Default::default() + }); + let budget = FileResponseBudget::new(1).unwrap(); + let record = store.record(); + let mut body = budget.body(store.clone(), record.clone(), None).unwrap(); + assert_eq!(budget.active(), 1); + assert!(matches!( + budget.body(store.clone(), record, None), + Err(FileBodyError::Busy) + )); + assert_eq!(store.reads.load(Ordering::SeqCst), 0); + let mut output = Vec::new(); + while let Some(frame) = body.frame().await { + let bytes = frame.unwrap().into_data().unwrap(); + assert!(bytes.len() <= 16 * 1024); + output.extend_from_slice(&bytes); + assert_eq!(body.size_hint().exact(), Some(70_000 - output.len() as u64)); + assert_eq!(store.reads.load(Ordering::SeqCst), 1); + } + assert_eq!(output, store.bytes); + assert!(body.is_end_stream()); + assert_eq!(budget.active(), 0); +} + +#[tokio::test] +async fn file_http_body_handles_ranges_empty_files_invalid_records_and_read_errors() { + let store = Arc::new(TestFileBlocks { + bytes: vec![5; 100], + ..Default::default() + }); + let budget = FileResponseBudget::new(1).unwrap(); + let body = budget + .body( + store.clone(), + store.record(), + Some(ByteRange { start: 10, end: 40 }), + ) + .unwrap(); + assert_eq!(body.collect().await.unwrap().to_bytes().len(), 30); + assert_eq!(budget.active(), 0); + assert!(budget + .body( + store.clone(), + store.record(), + Some(ByteRange { start: 101, end: 100 }) + ) + .is_err()); + assert_eq!(budget.active(), 0); + store.fail.store(true, Ordering::SeqCst); + let mut body = budget.body(store.clone(), store.record(), None).unwrap(); + assert!(body.frame().await.unwrap().is_err()); + assert!(body.frame().await.is_none()); + assert_eq!(budget.active(), 0); + let store = Arc::new(TestFileBlocks::default()); + let body = budget.body(store.clone(), store.record(), None).unwrap(); + assert!(body.is_end_stream()); + assert_eq!(budget.active(), 0); +} + +#[tokio::test] +async fn dropping_pending_file_http_body_cancels_reads_and_releases_admission() { + let store = Arc::new(TestFileBlocks { + bytes: vec![1; 100], + ..Default::default() + }); + store.pause.store(true, Ordering::SeqCst); + let budget = FileResponseBudget::new(1).unwrap(); + let mut body = budget.body(store.clone(), store.record(), None).unwrap(); + tokio::time::timeout(std::time::Duration::from_secs(1), async { + tokio::select! { + _ = body.frame() => panic!("paused read unexpectedly returned"), + () = store.entered.notified() => {} + } + }) + .await + .unwrap(); + assert_eq!(budget.active(), 1); + drop(body); + assert_eq!(budget.active(), 0); + assert_eq!(store.reads.load(Ordering::SeqCst), 1); +} diff --git a/app/crowdb-access-server/tests/iceberg_file_complete_test.rs b/app/crowdb-access-server/tests/iceberg_file_complete_test.rs new file mode 100644 index 000000000..2b59de7e0 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_complete_test.rs @@ -0,0 +1,156 @@ +use std::future::pending; +use std::sync::{ + atomic::{AtomicBool, Ordering}, + Arc, +}; +use std::time::Duration; + +use crowdb_access_server::iceberg::{active_io_for_tests, FileCompleteBody, FileS3ErrorCode}; +use http_body_util::BodyExt; +use hyper::body::Body; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +const PREFIX: &[u8] = b""; + +#[tokio::test(start_paused = true)] +async fn completion_streams_heartbeats_then_one_parseable_document() { + let (sender, receiver) = tokio::sync::oneshot::channel(); + let mut body = FileCompleteBody::new( + async move { receiver.await.unwrap() }, + "object", + Duration::from_secs(10), + Duration::from_secs(300), + ) + .unwrap(); + assert_eq!(body.size_hint().exact(), None); + let mut output = body.frame().await.unwrap().unwrap().into_data().unwrap().to_vec(); + assert_eq!(output, PREFIX); + for _ in 0..3 { + output.extend_from_slice(&body.frame().await.unwrap().unwrap().into_data().unwrap()); + } + assert_eq!(&output[PREFIX.len()..], b"\n\n\n"); + sender + .send(Ok([ + PREFIX, + b"etag", + ] + .concat())) + .unwrap(); + output.extend_from_slice(&body.collect().await.unwrap().to_bytes()); + let mut reader = quick_xml::Reader::from_reader(output.as_slice()); + let mut declarations = 0; + loop { + match reader.read_event().unwrap() { + quick_xml::events::Event::Decl(_) => declarations += 1, + quick_xml::events::Event::Eof => break, + _ => {} + } + } + assert_eq!(declarations, 1); + assert!(output.ends_with(b"")); +} + +#[tokio::test(start_paused = true)] +async fn late_failure_and_work_deadline_return_error_xml() { + let mut failure = FileCompleteBody::new( + async { Err(FileS3ErrorCode::InvalidRequest) }, + "object<&>", + Duration::from_secs(10), + Duration::from_secs(300), + ) + .unwrap(); + assert_eq!( + failure.frame().await.unwrap().unwrap().into_data().unwrap(), + PREFIX + ); + let error = failure.frame().await.unwrap().unwrap().into_data().unwrap(); + let error = std::str::from_utf8(&error).unwrap(); + assert!(error.starts_with("")); + assert!(error.contains("InvalidRequest")); + assert!(error.contains("object<&>")); + assert!(failure.is_end_stream()); + assert_eq!(failure.size_hint().exact(), Some(0)); + assert!(failure.frame().await.is_none()); + + let body = FileCompleteBody::new( + pending(), + "object", + Duration::from_secs(10), + Duration::from_secs(30), + ) + .unwrap(); + let start = tokio::time::Instant::now(); + let output = body.collect().await.unwrap().to_bytes(); + assert_eq!(start.elapsed(), Duration::from_secs(30)); + assert!(std::str::from_utf8(&output) + .unwrap() + .contains("SlowDown")); +} + +struct TestCancellation(Arc); + +impl Drop for TestCancellation { + fn drop(&mut self) { + self.0.store(true, Ordering::SeqCst); + } +} + +#[tokio::test(start_paused = true)] +async fn disconnect_drops_pending_completion_without_detached_work() { + let dropped = Arc::new(AtomicBool::new(false)); + let guard = TestCancellation(dropped.clone()); + let mut body = FileCompleteBody::new( + async move { + let _guard = guard; + pending().await + }, + "object", + Duration::from_secs(10), + Duration::from_secs(300), + ) + .unwrap(); + body.frame().await.unwrap().unwrap(); + body.frame().await.unwrap().unwrap(); + assert!(!dropped.load(Ordering::SeqCst)); + drop(body); + assert!(dropped.load(Ordering::SeqCst)); +} + +#[tokio::test(start_paused = true)] +async fn connection_activity_extends_idle_deadline() { + let (stream, mut peer) = tokio::io::duplex(64); + let (mut stream, expired) = active_io_for_tests(stream, Duration::from_secs(30)); + tokio::pin!(expired); + for _ in 0..4 { + tokio::select! { + () = &mut expired => panic!("active connection expired"), + () = tokio::time::sleep(Duration::from_secs(20)) => {} + } + stream.write_all(b" ").await.unwrap(); + assert_eq!(peer.read_u8().await.unwrap(), b' '); + peer.write_all(b"x").await.unwrap(); + assert_eq!(stream.read_u8().await.unwrap(), b'x'); + } + let start = tokio::time::Instant::now(); + expired.await; + assert_eq!(start.elapsed(), Duration::from_secs(30)); +} + +#[tokio::test(start_paused = true)] +async fn active_response_transmission_survives_prior_absolute_lifetime() { + let (stream, mut peer) = tokio::io::duplex(64); + let (mut stream, expired) = active_io_for_tests(stream, Duration::from_secs(30)); + tokio::pin!(expired); + let start = tokio::time::Instant::now(); + for _ in 0..4 { + tokio::select! { + () = &mut expired => panic!("connection expired before its deadline"), + () = tokio::time::sleep(Duration::from_secs(20)) => {} + } + stream.write_all(b" ").await.unwrap(); + assert_eq!(peer.read_u8().await.unwrap(), b' '); + } + assert_eq!(start.elapsed(), Duration::from_secs(80)); + expired.await; + assert_eq!(start.elapsed(), Duration::from_secs(110)); +} diff --git a/app/crowdb-access-server/tests/iceberg_file_encoding_test.rs b/app/crowdb-access-server/tests/iceberg_file_encoding_test.rs new file mode 100644 index 000000000..20411ebc0 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_encoding_test.rs @@ -0,0 +1,210 @@ +#[path = "common/iceberg_signed_chunks.rs"] +mod signed; + +use std::collections::VecDeque; +use std::convert::Infallible; +use std::pin::Pin; +use std::task::{Context, Poll}; + +use crowdb_access_s3::auth::{RawAuthRequest, SigV4Verifier}; +use crowdb_access_server::iceberg::{FileEncodingError, FileUploadBody}; +use http_body_util::{BodyExt, Full}; +use hyper::body::{Body, Bytes, Frame}; +use hyper::{header::HeaderValue, HeaderMap}; + +struct TestFrames(VecDeque); + +#[tokio::test] +async fn content_md5_checks_decoded_bytes_before_successful_eof() { + use base64::{engine::general_purpose::STANDARD, Engine}; + use md5::{Digest, Md5}; + + for streaming in [false, true] { + for valid in [false, true] { + let (mut headers, verifier, wire) = signed::fixture(); + let decoded = vec![b'a'; 66560]; + let digest = if valid { + Md5::digest(&decoded) + } else { + Md5::digest(b"wrong") + }; + if !streaming { + headers = HeaderMap::new(); + } + headers.insert("content-md5", STANDARD.encode(digest).parse().unwrap()); + let bytes = if streaming { wire } else { decoded }; + let input = TestFrames(bytes.chunks(997).map(Bytes::copy_from_slice).collect()); + let body = FileUploadBody::new(input, &headers, streaming.then_some(verifier), 100_000).unwrap(); + let result = body.collect().await; + if valid { + assert_eq!(result.unwrap().to_bytes(), vec![b'a'; 66560]); + } else { + assert!(matches!(result, Err(FileEncodingError::Checksum))); + } + } + } +} + +#[test] +fn content_md5_rejects_malformed_and_duplicate_headers() { + for value in ["invalid", "YQ=="] { + let mut headers = HeaderMap::new(); + headers.insert("content-md5", value.parse().unwrap()); + assert!(FileUploadBody::new(Full::new(Bytes::new()), &headers, None, 100).is_err()); + } + let mut headers = HeaderMap::new(); + headers.append( + "content-md5", + HeaderValue::from_static("1B2M2Y8AsgTpgAmY7PhCfg=="), + ); + headers.append( + "content-md5", + HeaderValue::from_static("1B2M2Y8AsgTpgAmY7PhCfg=="), + ); + assert!(FileUploadBody::new(Full::new(Bytes::new()), &headers, None, 100).is_err()); +} + +impl Body for TestFrames { + type Data = Bytes; + type Error = Infallible; + fn poll_frame( + mut self: Pin<&mut Self>, + _: &mut Context<'_>, + ) -> Poll, Self::Error>>> { + Poll::Ready(self.0.pop_front().map(|bytes| Ok(Frame::data(bytes)))) + } +} + +#[tokio::test] +async fn aws_published_signed_trailer_vector_survives_arbitrary_http_boundaries() { + for width in [1, 7, 16 * 1024, 100_000] { + let (headers, verifier, bytes) = signed::fixture(); + let input = TestFrames(bytes.chunks(width).map(Bytes::copy_from_slice).collect()); + let mut body = FileUploadBody::new(input, &headers, Some(verifier), 100_000).unwrap(); + assert_eq!(body.decoded_length(), Some(66560)); + let mut output = Vec::new(); + while let Some(frame) = body.frame().await { + let bytes = frame.unwrap().into_data().unwrap(); + assert!(bytes.len() <= 64 * 1024); + output.extend_from_slice(&bytes); + } + assert!(body.is_end_stream()); + assert_eq!(output, vec![b'a'; 66560]); + } +} + +#[tokio::test] +async fn corrupt_chunks_checksums_signatures_suffixes_and_truncation_fail_closed() { + for variant in 0..7 { + let (headers, verifier, mut bytes) = signed::fixture(); + match variant { + 0 => bytes[100] ^= 1, + 1 => bytes[25] = b'0', + 2 => { + let offset = bytes.windows(8).position(|bytes| bytes == b"sOO8/Q==").unwrap(); + bytes[offset] = b't'; + } + 3 => { + let offset = bytes + .windows(b"x-amz-trailer-signature:".len()) + .position(|bytes| bytes == b"x-amz-trailer-signature:") + .unwrap(); + bytes[offset + b"x-amz-trailer-signature:".len()] = b'0'; + } + 4 => bytes.extend_from_slice(b"extra"), + 5 => { + bytes.truncate(bytes.len() - 2); + } + 6 => { + bytes.truncate(100); + } + _ => unreachable!(), + } + let mut body = + FileUploadBody::new(Full::new(Bytes::from(bytes)), &headers, Some(verifier), 100_000).unwrap(); + loop { + match body.frame().await { + Some(Ok(_)) => {} + Some(Err(_)) => break, + None => panic!("corrupt variant {variant} succeeded"), + } + } + assert!(body.is_end_stream()); + assert!(body.frame().await.is_none()); + } +} + +#[tokio::test] +async fn encoded_byte_budget_and_plain_checksum_headers_are_enforced() { + let (headers, verifier, bytes) = signed::fixture(); + let body = FileUploadBody::new(Full::new(Bytes::from(bytes)), &headers, Some(verifier), 66560).unwrap(); + assert!(matches!(body.collect().await, Err(FileEncodingError::Length))); + for (name, value) in [("crc32", "y/Q5Jg=="), ("crc32c", "4waSgw==")] { + let mut headers = HeaderMap::new(); + headers.insert( + format!("x-amz-checksum-{name}") + .parse::() + .unwrap(), + HeaderValue::from_static(value), + ); + let body = + FileUploadBody::new(Full::new(Bytes::from_static(b"123456789")), &headers, None, 100).unwrap(); + assert_eq!(body.collect().await.unwrap().to_bytes(), b"123456789".as_slice()); + let body = + FileUploadBody::new(Full::new(Bytes::from_static(b"123456788")), &headers, None, 100).unwrap(); + assert!(matches!(body.collect().await, Err(FileEncodingError::Checksum))); + } +} + +#[test] +fn altered_streaming_seed_and_duplicate_framing_headers_are_rejected() { + let verifier = SigV4Verifier::new(signed::TestAwsCredentials, "us-east-1".into(), 900); + for field in [ + "x-amz-decoded-content-length", + "x-amz-trailer", + "content-encoding", + ] { + let mut request = signed::signed_request(); + request + .headers_mut() + .insert(field, HeaderValue::from_static("changed")); + assert!(verifier + .verify_streaming( + RawAuthRequest::from_parts(request.method(), request.uri(), request.headers()), + 1_369_353_600 + ) + .is_err()); + let mut request = signed::signed_request(); + let value = request.headers()[field].clone(); + request.headers_mut().append(field, value); + assert!(verifier + .verify_streaming( + RawAuthRequest::from_parts(request.method(), request.uri(), request.headers()), + 1_369_353_600 + ) + .is_err()); + } + let (mut headers, verifier, _) = signed::fixture(); + headers.append("x-amz-decoded-content-length", HeaderValue::from_static("66560")); + assert!(FileUploadBody::new(Full::new(Bytes::new()), &headers, Some(verifier), 100_000).is_err()); +} + +#[tokio::test] +async fn signed_without_trailers_and_unsigned_trailers_require_complete_framing() { + for unsigned in [false, true] { + for empty in [false, true] { + let (headers, verifier, bytes) = signed::other_fixture(unsigned, empty); + let input = TestFrames(bytes.chunks(1).map(Bytes::copy_from_slice).collect()); + let body = FileUploadBody::new(input, &headers, Some(verifier), 1000).unwrap(); + assert_eq!( + body.collect().await.unwrap().to_bytes().as_ref(), + if empty { b"".as_slice() } else { b"abc".as_slice() } + ); + let (headers, verifier, mut bytes) = signed::other_fixture(unsigned, empty); + bytes.truncate(bytes.len() - 2); + let body = + FileUploadBody::new(Full::new(Bytes::from(bytes)), &headers, Some(verifier), 1000).unwrap(); + assert!(body.collect().await.is_err()); + } + } +} diff --git a/app/crowdb-access-server/tests/iceberg_file_http_test.rs b/app/crowdb-access-server/tests/iceberg_file_http_test.rs new file mode 100644 index 000000000..36e79e384 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_http_test.rs @@ -0,0 +1,362 @@ +#[path = "common/iceberg_stack.rs"] +#[allow(dead_code)] +mod common; +#[path = "common/iceberg_process.rs"] +#[allow(dead_code)] +mod process; + +#[path = "common/iceberg_commit_child.rs"] +mod child; +#[path = "common/iceberg_commit_fault.rs"] +mod fault; +#[path = "common/iceberg_file_lifecycle.rs"] +mod lifecycle; +#[path = "common/iceberg_file_recovery.rs"] +mod recovery; + +use common::{now_ms, TestIcebergStack}; +use crowdb_access_iceberg::catalog::{CatalogRepository, ClearBounds, ManagementPrivilege}; +use crowdb_access_iceberg::file::{ + FileGrant, FileGrantIssuer, FileKind, FileOperation, FileOperations, FileRepository, TableLocation, +}; +use crowdb_access_iceberg::key::OperationId; +use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; +use crowdb_access_iceberg::wire::BearerAuthenticator; +use reqwest::{Client, Method}; + +#[path = "common/iceberg_signed_file.rs"] +mod signed; +use signed::TestFileClient; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "test-only child listener invoked by native file crash matrix"] +async fn native_fault_listener_child() { + child::run().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires native storage and kills listener subprocesses at durable FileIO boundaries"] +async fn native_file_publication_recovers_across_listeners_at_every_durable_write() { + recovery::run().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires native storage and checks real-time credential expiry and lifecycle fencing"] +async fn native_file_credentials_refresh_expire_and_follow_lifecycle_fences() { + lifecycle::run().await; +} + +async fn setup() -> ( + TestIcebergStack, + process::TestIcebergProcess, + TestFileClient, + TableLocation, +) { + setup_with_bounds(ClearBounds { + delegated_access_ms: 900_000, + ..ClearBounds::default() + }) + .await +} + +async fn setup_with_bounds( + bounds: ClearBounds, +) -> ( + TestIcebergStack, + process::TestIcebergProcess, + TestFileClient, + TableLocation, +) { + let stack = TestIcebergStack::start().await; + let repository = CatalogRepository::new(stack.store().await, bounds).unwrap(); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "file-http".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + now_ms(), + ) + .await + .unwrap(); + common::activate(&repository).await; + let context = repository.status().await.unwrap().0.context; + let process = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let client = Client::new(); + let endpoint = format!("http://{}", process.address); + let namespace = client + .post(format!("{endpoint}/v1/namespaces")) + .bearer_auth("w".repeat(32)) + .json(&serde_json::json!({"namespace": ["analytics"]})) + .send() + .await + .unwrap(); + assert_eq!(namespace.status(), 200, "{}", namespace.text().await.unwrap()); + let draft = client + .post(format!("{endpoint}/v1/namespaces/analytics/tables")) + .bearer_auth("w".repeat(32)) + .json(&serde_json::json!({"name": "files", "stage-create": true, + "schema": {"type": "struct", "fields": []}})) + .send() + .await + .unwrap(); + assert_eq!(draft.status(), 200, "{}", draft.text().await.unwrap()); + let draft: serde_json::Value = draft.json().await.unwrap(); + let table: TableLocation = format!("{}/", draft["metadata"]["location"].as_str().unwrap()) + .parse() + .unwrap(); + let authenticator = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let issuer = FileGrantIssuer::new(authenticator.namespace_token_key(), 15 * 60 * 1000).unwrap(); + let started = now_ms(); + let credentials = issuer + .issue(FileGrant { + context, + table: table.table, + principal: [7; 32], + nonce: OperationId::random(), + issued_ms: started - 1_000, + expires_ms: started + 10 * 60 * 1000, + operations: FileOperations::new(&[ + FileOperation::Head, + FileOperation::Get, + FileOperation::Put, + FileOperation::CreateMultipart, + FileOperation::UploadPart, + FileOperation::ListParts, + FileOperation::CompleteMultipart, + FileOperation::AbortMultipart, + ]) + .unwrap(), + max_request_bytes: 16 * 1024 * 1024, + max_file_bytes: 64 * 1024 * 1024, + }) + .unwrap(); + let client = TestFileClient { + client, + credentials, + address: process.address, + }; + (stack, process, client, table) +} + +fn path(table: TableLocation, key: &str) -> String { + format!("/{}/{}", table.bucket(), table.file(key).unwrap().object_key()) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn signed_standard_put_get_and_multipart_publish_unbound_files() { + let (stack, _process, client, table) = setup().await; + let parquet = b"PAR1datafoot\x04\0\0\0PAR1"; + let object = path(table, "data/a.parquet"); + let put = client.send(Method::PUT, &object, "", parquet, true).await; + assert_eq!(put.status(), 200, "{}", put.text().await.unwrap()); + let head = client.send(Method::HEAD, &object, "", b"", false).await; + assert_eq!(head.status(), 200); + assert_eq!(head.headers()["content-length"], parquet.len().to_string()); + let get = client.send(Method::GET, &object, "", b"", false).await; + assert_eq!(get.status(), 200); + assert_eq!(get.bytes().await.unwrap().as_ref(), parquet); + let range = client + .send_range(Method::GET, &object, "", b"", false, Some("bytes=4-7")) + .await; + assert_eq!(range.status(), 206); + assert_eq!(range.headers()["content-range"], "bytes 4-7/20"); + assert_eq!(range.bytes().await.unwrap().as_ref(), b"data"); + let repository = FileRepository::new(stack.store().await); + let record = repository + .load( + client.credentials.grant().context, + &table.file("data/a.parquet").unwrap(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(record.kind, FileKind::Unbound); + assert!(record.bind_kind(FileKind::EqualityDelete).is_ok()); + let conflict = client + .send(Method::PUT, &object, "", b"PAR1difffoot\x04\0\0\0PAR1", false) + .await; + assert_eq!(conflict.status(), 409); + + let metadata = path(table, "metadata/b.json"); + let create = client.send(Method::POST, &metadata, "uploads=", b"", false).await; + assert_eq!(create.status(), 200); + let xml = create.text().await.unwrap(); + let upload = xml + .split_once("") + .unwrap() + .1 + .split_once("") + .unwrap() + .0; + let mut document = b"{\"answer\":\"".to_vec(); + document.extend(std::iter::repeat_n(b'x', 5 * 1024 * 1024)); + document.extend_from_slice(b"\"}"); + let first = &document[..5 * 1024 * 1024]; + let second = &document[5 * 1024 * 1024..]; + let mut etags = Vec::new(); + for (number, bytes) in [(1, first), (2, second)] { + let query = format!("partNumber={number}&uploadId={upload}"); + let response = client.send(Method::PUT, &metadata, &query, bytes, true).await; + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + etags.push(response.headers()["etag"].to_str().unwrap().to_owned()); + } + let listed = client + .send(Method::GET, &metadata, &format!("uploadId={upload}"), b"", false) + .await; + assert_eq!(listed.status(), 200); + assert!(listed + .text() + .await + .unwrap() + .contains("2")); + let complete_xml = format!("{}1{}2", etags[0], etags[1]); + let complete = client + .send( + Method::POST, + &metadata, + &format!("uploadId={upload}"), + complete_xml.as_bytes(), + false, + ) + .await; + assert_eq!(complete.status(), 200, "{}", complete.text().await.unwrap()); + assert!(complete + .text() + .await + .unwrap() + .ends_with("")); + let replay = client + .send( + Method::POST, + &metadata, + &format!("uploadId={upload}"), + complete_xml.as_bytes(), + false, + ) + .await; + assert_eq!(replay.status(), 200); + assert!(replay + .text() + .await + .unwrap() + .ends_with("")); + let get = client.send(Method::GET, &metadata, "", b"", false).await; + assert_eq!(get.status(), 200); + assert_eq!(get.bytes().await.unwrap().as_ref(), document); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires Maven and the pinned Apache Iceberg Java dependencies"] +async fn official_java_s3_fileio_uploads_and_reads_native_files() { + use base64::Engine as _; + use crowdb_access_iceberg::wire::LoadCredentialsResponse; + use std::io::Write as _; + use std::process::{Command, Stdio}; + + let (_stack, _process, client, table) = setup().await; + let response = serde_json::to_vec(&LoadCredentialsResponse::from(client.credentials)).unwrap(); + let configuration = format!( + "endpoint=http://{}\ncredentials={}\nlocation={}\n", + client.address, + base64::engine::general_purpose::STANDARD.encode(response), + table + .file("placeholder") + .unwrap() + .to_string() + .trim_end_matches("placeholder"), + ); + let status = tokio::task::spawn_blocking(move || { + let maven = std::env::var_os("CROWDB_ICEBERG_E2E_MVN").unwrap_or_else(|| "mvn".into()); + let mut child = Command::new("timeout") + .arg("600") + .arg(maven) + .args(["--batch-mode", "--no-transfer-progress", "-f"]) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_java/pom.xml" + )) + .args(["compile", "exec:java"]) + .stdin(Stdio::piped()) + .spawn() + .unwrap(); + child + .stdin + .take() + .unwrap() + .write_all(configuration.as_bytes()) + .unwrap(); + child.wait().unwrap() + }) + .await + .unwrap(); + assert!(status.success(), "official Apache Iceberg S3FileIO failed"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires native storage services, Maven and pinned Apache Iceberg dependencies"] +async fn official_java_catalog_commits_native_parquet_snapshots_and_staged_tables() { + let (stack, process, _, _) = setup_with_bounds(ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }) + .await; + let endpoint = format!("http://{}", process.address); + run_catalog_sdk(endpoint, "data").await; + drop(process); + let restarted = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + run_catalog_sdk(format!("http://{}", restarted.address), "verify").await; +} + +async fn run_catalog_sdk(endpoint: String, mode: &'static str) { + run_sdk(endpoint, "TestIcebergCatalogWrites", mode).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires native storage services, Maven and pinned Apache Iceberg dependencies"] +async fn official_java_identical_s3_uploads_validate_selected_data_and_delete_uses() { + let (_stack, process, _, _) = setup_with_bounds(ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }) + .await; + let endpoint = format!("http://{}", process.address); + run_sdk(endpoint, "TestIcebergSelectedFiles", "").await; +} + +async fn run_sdk(endpoint: String, class: &'static str, mode: &'static str) { + let status = tokio::task::spawn_blocking(move || { + let maven = std::env::var_os("CROWDB_ICEBERG_E2E_MVN").unwrap_or_else(|| "mvn".into()); + std::process::Command::new("timeout") + .arg("600") + .arg(maven) + .args(["-o", "--batch-mode", "--no-transfer-progress", "-f"]) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_java/pom.xml" + )) + .args(["compile", "exec:java"]) + .arg(format!("-Dexec.mainClass={class}")) + .arg(format!("-Dexec.args={endpoint} {mode}")) + .status() + .unwrap() + }) + .await + .unwrap(); + assert!( + status.success(), + "official native catalog and Parquet acceptance failed" + ); +} diff --git a/app/crowdb-access-server/tests/iceberg_file_request_test.rs b/app/crowdb-access-server/tests/iceberg_file_request_test.rs new file mode 100644 index 000000000..69be8646a --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_request_test.rs @@ -0,0 +1,135 @@ +use crowdb_access_iceberg::file::{FileOperation, TableLocation}; +use crowdb_access_iceberg::key::{CatalogId, TableId}; +use crowdb_access_server::iceberg::{FileRequest, FileRequestError, MultipartRequest}; +use hyper::Method; + +fn table() -> TableLocation { + TableLocation { + catalog: CatalogId::random(), + table: TableId::random(), + } +} + +fn path(table: TableLocation, suffix: &str) -> hyper::Uri { + format!("/{}{suffix}", table.to_string().trim_start_matches("s3://")) + .parse() + .unwrap() +} + +#[test] +fn native_object_routes_decode_once_and_preserve_plus_percent_and_repeated_slashes() { + let table = table(); + for (wire, key) in [ + ("a+b", "a+b"), + ("a%252Fb", "a%2Fb"), + ("a//b", "a//b"), + ("%E5%86%B0/a%20b", "冰/a b"), + ] { + for (method, operation) in [ + (Method::GET, FileOperation::Get), + (Method::HEAD, FileOperation::Head), + (Method::PUT, FileOperation::Put), + ] { + let request = FileRequest::parse(&method, &path(table, wire)).unwrap(); + assert_eq!(request.location, table.file(key).unwrap()); + assert_eq!(request.operation, operation); + assert!(request.multipart.is_none()); + } + } +} + +#[test] +fn native_routes_expose_exact_multipart_operations_but_never_file_delete() { + let table = table(); + let create = FileRequest::parse(&Method::POST, &path(table, "file?uploads")).unwrap(); + assert_eq!(create.multipart, Some(MultipartRequest::Create)); + let upload = FileRequest::parse(&Method::PUT, &path(table, "file?partNumber=10000&uploadId=id")).unwrap(); + assert_eq!( + upload.multipart, + Some(MultipartRequest::Upload { + upload_id: "id".into(), + part_number: 10000 + }) + ); + let list = FileRequest::parse( + &Method::GET, + &path(table, "file?uploadId=id&max-parts=3&part-number-marker=4"), + ) + .unwrap(); + assert_eq!( + list.multipart, + Some(MultipartRequest::List { + upload_id: "id".into(), + marker: 4, + max_parts: 3 + }) + ); + assert_eq!( + FileRequest::parse(&Method::POST, &path(table, "file?uploadId=id")) + .unwrap() + .operation, + FileOperation::CompleteMultipart + ); + assert_eq!( + FileRequest::parse(&Method::DELETE, &path(table, "file?uploadId=id")) + .unwrap() + .operation, + FileOperation::AbortMultipart + ); + assert_eq!( + FileRequest::parse(&Method::DELETE, &path(table, "file")), + Err(FileRequestError::Unsupported) + ); +} + +#[test] +fn native_routes_reject_path_escape_duplicate_parameters_and_general_s3_operations() { + let table = table(); + for suffix in [ + "", + "%2e%2e/escape", + "a/%2e/b", + "%2Fescape", + "a%5Cb", + "a%00b", + "%ff", + "%", + "%2G", + "file?tagging", + "file?uploads&uploadId=id", + "file?partNumber=1", + "file?uploadId=", + "file?uploadId=id&uploadId=id", + "file?uploadId=id&%75ploadId=id", + "file?uploadId=id&max-parts=1001", + "file?uploadId=id&part-number-marker=-1", + ] { + assert!( + FileRequest::parse(&Method::GET, &path(table, suffix)).is_err(), + "{suffix}" + ); + } + for part in ["0", "10001", "+1", "-1", "65536", ""] { + assert!(FileRequest::parse( + &Method::PUT, + &path(table, &format!("file?uploadId=id&partNumber={part}")) + ) + .is_err()); + } + assert!(FileRequest::parse(&Method::GET, &"/ordinary-bucket/file".parse().unwrap()).is_err()); + assert!(FileRequest::parse(&Method::GET, &"/".parse().unwrap()).is_err()); +} + +#[test] +fn routing_ignores_only_known_presign_fields_and_bounds_total_input() { + let table = table(); + let query = "file?X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential=x&X-Amz-Date=x&X-Amz-Expires=1&X-Amz-Security-Token=x&X-Amz-SignedHeaders=host&X-Amz-Signature=x"; + assert_eq!( + FileRequest::parse(&Method::GET, &path(table, query)) + .unwrap() + .operation, + FileOperation::Get + ); + assert!(FileRequest::parse(&Method::GET, &path(table, "file?X-Amz-Unknown=x")).is_err()); + assert!(FileRequest::parse(&Method::GET, &path(table, &format!("file?{}", "x".repeat(8192)))).is_err()); +} diff --git a/app/crowdb-access-server/tests/iceberg_file_response_test.rs b/app/crowdb-access-server/tests/iceberg_file_response_test.rs new file mode 100644 index 000000000..d50c9fe03 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_response_test.rs @@ -0,0 +1,208 @@ +#[path = "common/iceberg_file_blocks.rs"] +mod blocks; + +use crowdb_access_iceberg::catalog::CatalogContext; +use crowdb_access_iceberg::file::{ + AssemblyProgress, ContentFormat, FileContent, FileIdentity, FileKind, FileTree, FileWriterCheckpoint, + MultipartCompletion, MultipartLimits, MultipartPart, MultipartPartPage, MultipartPhase, MultipartSession, + TableLocation, +}; +use crowdb_access_iceberg::key::{CatalogId, FileId, OperationId, TableId}; +use crowdb_access_iceberg::operation::PayloadReference; +use crowdb_access_server::iceberg::{FileS3ErrorCode, MultipartResponses}; +use hyper::StatusCode; +use sha2::{Digest, Sha256}; + +fn session() -> MultipartSession { + let table = TableLocation { + catalog: CatalogId::random(), + table: TableId::random(), + }; + MultipartSession { + context: CatalogContext { + catalog: table.catalog, + activation_epoch: 1, + }, + upload: OperationId::random(), + owner: FileIdentity { + table, + file: FileId::random(), + }, + location: table.file("metadata/a&.json").unwrap(), + principal: [1; 32], + revision: 1, + created_ms: 1_704_067_200_000, + expires_ms: 1_704_067_300_000, + limits: MultipartLimits { + max_parts: 10, + max_part_bytes: 1024, + max_file_bytes: 1024, + max_staged_bytes: 2048, + ttl_ms: 100_000, + }, + phase: MultipartPhase::Open, + part_count: 0, + staged_bytes: 0, + completion: None, + published: None, + pending: None, + credit: None, + } +} + +fn part(session: &MultipartSession, number: u16, modified_ms: u64) -> MultipartPart { + MultipartPart { + upload: session.upload, + number, + revision: 1, + modified_ms, + owner: FileIdentity { + file: FileId::random(), + ..session.owner + }, + tree: FileTree { + root: None, + length: 0, + digest: Sha256::digest([]).into(), + }, + } +} + +#[test] +fn create_list_and_abort_emit_s3_shaped_xml_and_stable_page_markers() { + let session = session(); + let create = MultipartResponses::create(&session).unwrap(); + assert_eq!(create.status(), StatusCode::OK); + let body = String::from_utf8(create.into_body()).unwrap(); + assert!(body.contains("t/")); + assert!(body.contains("a&<b>.json")); + assert!(body.contains(&format!("{}", session.upload))); + let first = part(&session, 1, session.created_ms); + let third = part(&session, 3, session.created_ms + 1234); + let response = MultipartResponses::upload_part(&third).unwrap(); + let tag = response.headers().get("etag").unwrap().to_str().unwrap(); + assert_eq!(tag.len(), 66); + let page = MultipartPartPage { + parts: vec![first, third], + next_marker: Some(3), + }; + let list = MultipartResponses::list_parts(&session, &page, 0, 1000).unwrap(); + assert_eq!(list.headers().get("content-type").unwrap(), "application/xml"); + let body = String::from_utf8(list.into_body()).unwrap(); + assert!(body.contains("3")); + assert!(body.contains("1000")); + assert!(body.contains("true")); + assert!(body.contains("2024-01-01T00:00:01.234Z")); + assert_eq!(body.matches("").count(), 2); + assert!(body.contains(&format!("{}", tag.replace('"', """)))); + let next = MultipartPartPage { + parts: vec![], + next_marker: None, + }; + let body = String::from_utf8( + MultipartResponses::list_parts(&session, &next, 3, 1000) + .unwrap() + .into_body(), + ) + .unwrap(); + assert!(body.contains("false")); + assert!(!body.contains("NextPartNumberMarker")); + assert_eq!(MultipartResponses::abort().status(), StatusCode::NO_CONTENT); +} + +#[test] +fn malformed_pages_timestamps_and_escape_input_fail_before_success() { + let session = session(); + let mut page = MultipartPartPage { + parts: vec![part(&session, 2, session.created_ms)], + next_marker: Some(1), + }; + assert!(MultipartResponses::list_parts(&session, &page, 0, 1000).is_err()); + page.next_marker = Some(2); + assert!(MultipartResponses::list_parts(&session, &page, 2, 1000).is_err()); + page.parts[0].modified_ms = session.expires_ms; + assert!(MultipartResponses::list_parts(&session, &page, 0, 1000).is_err()); + page.parts[0].modified_ms = u64::MAX; + assert!(MultipartResponses::list_parts(&session, &page, 0, 1000).is_err()); + let response = MultipartResponses::error(FileS3ErrorCode::InvalidPart, "/x&", "req-1").unwrap(); + assert_eq!(response.status(), StatusCode::BAD_REQUEST); + let body = String::from_utf8(response.into_body()).unwrap(); + assert!(body.contains("InvalidPart")); + assert!(body.contains("/x&<y>")); + assert!(MultipartResponses::error(FileS3ErrorCode::InternalError, &"x".repeat(2049), "req").is_err()); + for (code, status) in [ + (FileS3ErrorCode::AccessDenied, StatusCode::FORBIDDEN), + (FileS3ErrorCode::NoSuchUpload, StatusCode::NOT_FOUND), + (FileS3ErrorCode::InvalidPart, StatusCode::BAD_REQUEST), + (FileS3ErrorCode::EntityTooLarge, StatusCode::PAYLOAD_TOO_LARGE), + (FileS3ErrorCode::InvalidRequest, StatusCode::BAD_REQUEST), + (FileS3ErrorCode::InternalError, StatusCode::INTERNAL_SERVER_ERROR), + ] { + assert_eq!( + MultipartResponses::error(code, "/file", "id").unwrap().status(), + status + ); + } +} + +#[test] +fn complete_uses_only_a_published_matching_record() { + let mut session = session(); + let store = blocks::TestFileBlocks { + bytes: b"bytes".to_vec(), + ..Default::default() + }; + let mut record = store.record(); + record.file = session.owner.file; + record.location = session.location.clone(); + record.kind = FileKind::Metadata; + record.format = ContentFormat::Json; + let candidate = FileTree { + root: match &record.content { + FileContent::Chunks { root } => root.clone(), + FileContent::Inline { .. } => None, + }, + length: record.length, + digest: record.digest, + }; + let selection = PayloadReference { + catalog: session.context.catalog, + operation: session.upload, + digest: [3; 32], + length: 8, + }; + session.part_count = 1; + session.staged_bytes = 5; + session.phase = MultipartPhase::Published; + session.published = Some(session.owner.file); + session.completion = Some(MultipartCompletion { + selection: selection.clone(), + selected_parts: 1, + progress: AssemblyProgress { + selection: selection.digest, + next_part: 1, + part_offset: 0, + completed_bytes: 5, + writer: Some(FileWriterCheckpoint { + root: candidate.root.clone().unwrap(), + }), + active: None, + part_digest: None, + }, + candidate: Some(candidate.clone()), + publication: Some(selection), + }); + let complete = MultipartResponses::complete(&session, &record, "https://storage.example/a&b").unwrap(); + let body = String::from_utf8(complete.into_body()).unwrap(); + assert!(body.contains(""")); + assert!(body.contains("https://storage.example/a&b")); + let mut wrong = record.clone(); + wrong.file = FileId::random(); + assert!(MultipartResponses::complete(&session, &wrong, "https://storage.example/a").is_err()); + assert!(MultipartResponses::complete(&session, &record, "s3://wrong").is_err()); + session.phase = MultipartPhase::Publishing; + session.published = None; + assert!(MultipartResponses::complete(&session, &record, "https://storage.example/a").is_err()); +} diff --git a/app/crowdb-access-server/tests/iceberg_file_selection_test.rs b/app/crowdb-access-server/tests/iceberg_file_selection_test.rs new file mode 100644 index 000000000..bb0eea7c2 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_selection_test.rs @@ -0,0 +1,74 @@ +use crowdb_access_server::iceberg::CompleteSelection; + +#[test] +fn complete_xml_accepts_only_ordered_sha256_parts() { + let first = "01".repeat(32); + let second = "ab".repeat(32); + let xml = format!("1\"{first}\"10000\"{second}\""); + let selection = CompleteSelection::parse(xml.as_bytes()).unwrap(); + assert_eq!(selection.parts().len(), 2); + assert_eq!(selection.parts()[0].number, 1); + assert_eq!(selection.parts()[0].digest, [1; 32]); + assert_eq!(selection.parts()[1].number, 10_000); + assert_eq!(selection.parts()[1].digest, [0xab; 32]); + let sdk_xml = format!("\"{first}\"1"); + assert_eq!( + CompleteSelection::parse(sdk_xml.as_bytes()).unwrap().parts()[0].digest, + [1; 32] + ); + for quote in [""", """, """] { + let escaped = sdk_xml.replace(&format!("\"{first}\""), &format!("{quote}{first}{quote}")); + assert_eq!( + CompleteSelection::parse(escaped.as_bytes()).unwrap().parts()[0].digest, + [1; 32] + ); + } +} + +#[test] +fn complete_xml_rejects_ambiguous_or_unbounded_inputs() { + let etag = format!("\"{}\"", "01".repeat(32)); + let part = + |number: &str, tag: &str| format!("{number}{tag}"); + for body in [ + String::new(), + "".into(), + format!( + "{}{}", + part("2", &etag), + part("1", &etag) + ), + format!( + "{}{}", + part("1", &etag), + part("1", &etag) + ), + format!( + "{}", + part("0", &etag) + ), + format!( + "{}", + part("1", "bad") + ), + format!( + "{}", + part("1", &etag) + ), + format!( + "{}", + part("1", &etag) + ), + format!( + "{}", + part("1", &etag) + ), + "x".repeat(2 * 1024 * 1024 + 1), + format!( + "{}", + part("1", &format!("&unknown;{}&unknown;", "01".repeat(32))) + ), + ] { + assert!(CompleteSelection::parse(body.as_bytes()).is_err(), "{body:.100}"); + } +} diff --git a/app/crowdb-access-server/tests/iceberg_file_storage_test.rs b/app/crowdb-access-server/tests/iceberg_file_storage_test.rs new file mode 100644 index 000000000..02893266f --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_storage_test.rs @@ -0,0 +1,185 @@ +#[path = "common/iceberg_stack.rs"] +mod common; +#[path = "common/iceberg_multipart.rs"] +mod multipart; +#[path = "common/iceberg_file_worker.rs"] +mod worker; + +use std::sync::Arc; + +use common::TestIcebergStack; +use crowdb_access_iceberg::catalog::{CatalogContext, CatalogRepository, ClearBounds, ManagementPrivilege}; +use crowdb_access_iceberg::file::{ + ByteRange, ContentFormat, FileContent, FileIdentity, FileKind, FileReader, FileRecord, FileRepository, + FileTreeWriter, NativeFileBlocks, TableLocation, +}; +use crowdb_access_iceberg::key::{FileId, OperationId, TableId}; +use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; +use crowdb_chunk_client::{ChunkIoClient, ChunkIoClientConfig, SmallWritePolicy}; + +async fn chunks(stack: &TestIcebergStack) -> ChunkIoClient { + ChunkIoClient::connect(ChunkIoClientConfig { + management_seeds: stack.cluster.mgmt_endpoints.clone(), + diskio_connections_per_endpoint: 2, + diskio_rpc_workers: 1, + small_write: SmallWritePolicy { + min_pipelines: 1, + max_pipelines: 1, + memory_budget: 8 * 1024 * 1024, + mirror_copies: 1, + conversion_enabled: false, + ..SmallWritePolicy::default() + }, + }) + .await + .unwrap() +} + +async fn assert_checkpoint_intent( + stack: &TestIcebergStack, + owner: FileIdentity, + checkpoint: &crowdb_access_iceberg::file::FileWriterCheckpoint, +) { + let intents = crowdb_access_iceberg::gc::GcStore::scan_gc( + stack.store().await.as_ref(), + crowdb_access_iceberg::gc::GcScan { + catalog: owner.table.catalog, + scope: Some(crowdb_access_iceberg::key::CatalogScope::FileWriteIntent), + prefix: owner.table.table.as_bytes().to_vec(), + after: Vec::new(), + items: 32, + bytes: 64 * 1024, + }, + ) + .await + .unwrap(); + assert_eq!(intents.items.len(), 3); + assert!(intents.items.iter().any(|item| { + let key = crowdb_access_iceberg::key::IcebergKey::decode(&item.key).unwrap(); + let crowdb_access_iceberg::record::StorageRecord::FileWriteIntent(intent) = + crowdb_access_iceberg::record::StorageRecord::decode(&key, &item.value).unwrap() + else { + panic!() + }; + assert_eq!(intent.owner, owner); + intent.root == checkpoint.root + })); +} + +async fn seed_root(stack: &TestIcebergStack) -> CatalogContext { + let repository = CatalogRepository::new(stack.store().await, ClearBounds::default()).unwrap(); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: 100, + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "native-files".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + 100, + ) + .await + .unwrap(); + common::activate(&repository).await; + repository.status().await.unwrap().0.context +} + +async fn read_all(mut reader: FileReader) -> Vec { + let mut bytes = Vec::new(); + while let Some(frame) = reader.next().await.unwrap() { + assert!(frame.len() <= 4096); + bytes.extend_from_slice(&frame); + } + bytes +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn native_file_tree_publication_and_ranges_survive_catalog_storage_restart() { + let _ = tracing_subscriber::fmt() + .with_env_filter(tracing_subscriber::EnvFilter::from_default_env()) + .try_init(); + let mut stack = TestIcebergStack::start().await; + let context = seed_root(&stack).await; + let owner = FileIdentity { + table: TableLocation { + catalog: context.catalog, + table: TableId::random(), + }, + file: FileId::random(), + }; + let client = chunks(&stack).await; + let blocks = Arc::new(NativeFileBlocks::new(client.clone(), stack.store().await)); + let mut writer = FileTreeWriter::new(blocks.clone(), owner, 16 * 1024).unwrap(); + let bytes: Vec = (0..50_000) + .map(|index| u8::try_from(index % 251).unwrap()) + .collect(); + for piece in bytes[..25_000].chunks(3000) { + writer.push(piece).await.unwrap(); + } + let checkpoint = writer.checkpoint().await.unwrap(); + assert_checkpoint_intent(&stack, owner, &checkpoint).await; + drop(writer); + client.shutdown_small_writes().await.unwrap(); + drop(blocks); + drop(client); + let client = chunks(&stack).await; + let blocks = Arc::new(NativeFileBlocks::new(client.clone(), stack.store().await)); + let mut writer = FileTreeWriter::restore(blocks.clone(), owner, 16 * 1024, &checkpoint) + .await + .unwrap(); + for piece in bytes[25_000..].chunks(3000) { + writer.push(piece).await.unwrap(); + } + let tree = writer.finish().await.unwrap(); + assert_eq!(tree.root.as_ref().unwrap().height, 1); + let candidate = FileRecord { + file: owner.file, + location: owner.table.file("data/native.bin").unwrap(), + kind: FileKind::Data, + format: ContentFormat::Parquet, + length: tree.length, + digest: tree.digest, + content: FileContent::Chunks { root: tree.root }, + hint: None, + }; + let repository = FileRepository::new(stack.store().await); + assert_eq!(repository.publish(context, &candidate).await.unwrap(), candidate); + let reader = FileReader::new(blocks.clone(), candidate.clone(), None, 4096).unwrap(); + assert_eq!(read_all(reader).await, bytes); + client.shutdown_small_writes().await.unwrap(); + drop(repository); + drop(blocks); + drop(client); + stack.chunk_kv.restart().await; + let repository = FileRepository::new(stack.store().await); + let recovered = repository + .load(context, &candidate.location) + .await + .unwrap() + .unwrap(); + assert_eq!(recovered, candidate); + let client = chunks(&stack).await; + let reader = FileReader::new( + Arc::new(NativeFileBlocks::new(client.clone(), stack.store().await)), + recovered, + Some(ByteRange { + start: 16_380, + end: 33_000, + }), + 4096, + ) + .unwrap(); + assert_eq!(read_all(reader).await, bytes[16_380..33_000]); + assert_eq!(repository.publish(context, &candidate).await.unwrap(), candidate); + client.shutdown_small_writes().await.unwrap(); + drop(client); + Box::pin(multipart::verify_restart(&mut stack, context, owner.table)).await; + worker::verify(&stack, context, owner.table).await; +} diff --git a/app/crowdb-access-server/tests/iceberg_file_upload_test.rs b/app/crowdb-access-server/tests/iceberg_file_upload_test.rs new file mode 100644 index 000000000..46c1d3ffc --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_file_upload_test.rs @@ -0,0 +1,203 @@ +#[path = "common/iceberg_upload.rs"] +mod common; + +use crowdb_access_iceberg::file::{FileReader, NATIVE_FILE_BLOCK_BYTES}; +use crowdb_access_server::iceberg::{FileUploadBudget, FileUploadConstraints, FileUploadError}; +use hyper::body::{Bytes, Frame}; +use sha2::{Digest, Sha256}; +use std::sync::{atomic::Ordering, Arc}; + +fn constraints(max_bytes: u64) -> FileUploadConstraints { + FileUploadConstraints { + max_bytes, + content_length: None, + sha256: None, + } +} + +#[tokio::test] +async fn native_upload_pulls_bounded_frames_and_verifies_exact_bytes_before_returning_tree() { + let bytes = vec![19; 1024 * 1024 + 23]; + let store = Arc::new(common::TestUploadBlocks::default()); + let budget = FileUploadBudget::new(1).unwrap(); + let identity = common::owner(); + let tree = budget + .receive( + common::TestUploadBody::new(&bytes, 16 * 1024), + store.clone(), + identity, + FileUploadConstraints { + max_bytes: bytes.len() as u64, + content_length: Some(bytes.len() as u64), + sha256: Some(Sha256::digest(&bytes).into()), + }, + ) + .await + .unwrap(); + assert_eq!(budget.active(), 0); + assert_eq!(tree.length, bytes.len() as u64); + assert!(store.max_input.load(Ordering::SeqCst) <= NATIVE_FILE_BLOCK_BYTES); + let mut reader = FileReader::from_tree(store, identity, tree, None, 16 * 1024).unwrap(); + let mut actual = Vec::new(); + while let Some(frame) = reader.next().await.unwrap() { + actual.extend_from_slice(&frame); + } + assert_eq!(actual, bytes); +} + +#[tokio::test] +async fn upload_byte_length_and_digest_failures_never_return_a_tree() { + let budget = FileUploadBudget::new(1).unwrap(); + let store = Arc::new(common::TestUploadBlocks::default()); + let identity = common::owner(); + for maximum in [0, u64::MAX] { + let body = common::TestUploadBody::new(b"bytes", 5); + let polls = body.polls.clone(); + assert!(matches!( + budget + .receive(body, store.clone(), identity, constraints(maximum)) + .await, + Err(FileUploadError::Bounds) + )); + assert_eq!(polls.load(Ordering::SeqCst), 0); + } + for declared in [4, 6] { + assert!(matches!( + budget + .receive( + common::TestUploadBody::new(b"bytes", 2), + store.clone(), + identity, + FileUploadConstraints { + content_length: Some(declared), + ..constraints(10) + } + ) + .await, + Err(FileUploadError::Length) + )); + } + assert!(matches!( + budget + .receive( + common::TestUploadBody::new(b"bytes", 2), + store.clone(), + identity, + constraints(4) + ) + .await, + Err(FileUploadError::Bounds) + )); + assert!(matches!( + budget + .receive( + common::TestUploadBody::new(b"bytes", 2), + store.clone(), + identity, + FileUploadConstraints { + sha256: Some([0; 32]), + ..constraints(10) + } + ) + .await, + Err(FileUploadError::Digest) + )); + let large_frame = vec![1; 65_537]; + let tree = budget + .receive( + common::TestUploadBody::new(&large_frame, large_frame.len()), + store.clone(), + identity, + constraints(100_000), + ) + .await + .unwrap(); + assert_eq!(tree.length, large_frame.len() as u64); + assert!(store.max_input.load(Ordering::SeqCst) <= NATIVE_FILE_BLOCK_BYTES); + assert_eq!(budget.active(), 0); +} + +#[tokio::test] +async fn upload_transport_and_storage_errors_retain_orphans_but_empty_uploads_succeed() { + let budget = FileUploadBudget::new(1).unwrap(); + let store = Arc::new(common::TestUploadBlocks::default()); + let identity = common::owner(); + let mut body = common::TestUploadBody::new(b"", 1); + body.frames + .push_back(Err(std::io::Error::other("test body failure"))); + assert!(matches!( + budget + .receive(body, store.clone(), identity, constraints(10)) + .await, + Err(FileUploadError::Body) + )); + let mut body = common::TestUploadBody::new(b"", 1); + body.frames + .push_back(Ok(Frame::trailers(hyper::HeaderMap::new()))); + assert!(matches!( + budget + .receive(body, store.clone(), identity, constraints(10)) + .await, + Err(FileUploadError::Trailers) + )); + store.fail.store(true, Ordering::SeqCst); + assert!(matches!( + budget + .receive( + common::TestUploadBody::new(b"bytes", 2), + store.clone(), + identity, + constraints(10) + ) + .await, + Err(FileUploadError::Storage(_)) + )); + assert_eq!(budget.active(), 0); + assert!(!store.values.load().is_empty()); + store.fail.store(false, Ordering::SeqCst); + let empty = budget + .receive( + common::TestUploadBody::new(b"", 1), + store, + identity, + FileUploadConstraints { + content_length: Some(0), + sha256: Some(Sha256::digest([]).into()), + ..constraints(1) + }, + ) + .await + .unwrap(); + assert_eq!(empty.length, 0); + assert!(empty.root.is_none()); +} + +#[tokio::test] +async fn pending_storage_applies_backpressure_and_cancellation_releases_only_memory_credit() { + let budget = FileUploadBudget::new(1).unwrap(); + let store = Arc::new(common::TestUploadBlocks::default()); + store.pause.store(true, Ordering::SeqCst); + let body = common::TestUploadBody::new(&vec![7; 512 * 1024], 64 * 1024); + let polls = body.polls.clone(); + let mut upload = Box::pin(budget.receive(body, store.clone(), common::owner(), constraints(1024 * 1024))); + tokio::select! { + result = &mut upload => panic!("upload completed before storage release: {result:?}"), + () = store.entered.notified() => {} + } + assert_eq!(budget.active(), 1); + assert_eq!(polls.load(Ordering::SeqCst), 1); + let mut other = common::TestUploadBody::new(b"", 1); + other.frames.push_back(Ok(Frame::data(Bytes::new()))); + let other_polls = other.polls.clone(); + assert!(matches!( + budget + .receive(other, store.clone(), common::owner(), constraints(10)) + .await, + Err(FileUploadError::Busy) + )); + assert_eq!(other_polls.load(Ordering::SeqCst), 0); + drop(upload); + assert_eq!(budget.active(), 0); + assert_eq!(polls.load(Ordering::SeqCst), 1); + assert_eq!(store.values.load().len(), 1); +} diff --git a/app/crowdb-access-server/tests/iceberg_full_stack_test.rs b/app/crowdb-access-server/tests/iceberg_full_stack_test.rs new file mode 100644 index 000000000..dcb0f750c --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_full_stack_test.rs @@ -0,0 +1,319 @@ +#[path = "common/iceberg_background.rs"] +mod background; +#[path = "common/iceberg_stack.rs"] +mod common; +#[path = "common/iceberg_creation.rs"] +mod creation; +#[path = "common/iceberg_drop.rs"] +mod dropping; +#[path = "common/iceberg_fault.rs"] +mod fault; +#[path = "common/iceberg_journal.rs"] +mod journal; +#[path = "common/iceberg_namespace.rs"] +mod namespace; +#[path = "common/iceberg_process.rs"] +mod process; +#[path = "common/iceberg_property.rs"] +mod property; + +use std::sync::Arc; +use std::time::Duration; + +use common::{now_ms, TestIcebergStack}; +use crowdb_access_iceberg::catalog::{ + CatalogAuthority, CatalogError, CatalogRepository, ClearBounds, ManagementPrivilege, +}; +use crowdb_access_iceberg::key::OperationId; +use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; + +fn request( + action: ManagementAction, + name: &str, + previous: Option<(u64, &CatalogAuthority)>, +) -> ManagementRequest { + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::from_bytes( + &(u128::from(previous.map_or(0, |(epoch, _)| epoch)) * 8 + u128::from(action as u8) + 1) + .to_be_bytes(), + ) + .unwrap(), + issued_ms: now_ms(), + }, + principal: "clearer".into(), + action, + display_name: name.into(), + expected_epoch: previous.map_or(0, |(epoch, _)| epoch), + confirmation: previous + .filter(|_| action == ManagementAction::Clear) + .map(|(_, authority)| authority.catalog), + capabilities: None, + } +} + +async fn execute(repository: &CatalogRepository, request: ManagementRequest) -> CatalogAuthority { + tokio::time::timeout(Duration::from_secs(30), async { + loop { + match repository + .execute(request.clone(), ManagementPrivilege::Clear, now_ms()) + .await + { + Ok(authority) => return authority, + Err(CatalogError::Busy) => tokio::time::sleep(Duration::from_millis(20)).await, + Err(error) => panic!("management failed: {error:?}"), + } + } + }) + .await + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn catalog_recovery_survives_real_chunk_kv_restart() { + let mut stack = TestIcebergStack::start().await; + let bounds = ClearBounds { + request_ms: 500, + root_lease_ms: 0, + delegated_access_ms: 0, + clock_skew_ms: 10, + }; + let repository = Arc::new(CatalogRepository::new(stack.store().await, bounds).unwrap()); + let initialize = request(ManagementAction::Initialize, "original", None); + let original = execute(&repository, initialize.clone()).await; + common::activate(&repository).await; + let frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let second_frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + frontend.check_official_reads(); + second_frontend.check_official_reads(); + let denied = process::command(&stack.cluster.mgmt_endpoints) + .env("CROWDB_ICEBERG_TOKEN", "r".repeat(32)) + .arg("status") + .output() + .unwrap(); + assert!(!denied.status.success()); + assert!(String::from_utf8_lossy(&denied.stderr).contains("management privilege")); + for action in ["status", "initialize", "rename", "clear"] { + let denied = process::command(&stack.cluster.mgmt_endpoints) + .env("CROWDB_ICEBERG_TOKEN", "w".repeat(32)) + .arg(action) + .output() + .unwrap(); + assert!(!denied.status.success()); + assert!(String::from_utf8_lossy(&denied.stderr).contains("management privilege")); + } + let rename = request(ManagementAction::Rename, "renamed", Some((1, &original))); + let renamed = execute(&repository, rename).await; + assert_eq!(renamed.catalog, original.catalog); + let clear = request(ManagementAction::Clear, "replacement", Some((1, &renamed))); + assert!(matches!( + repository + .execute(clear.clone(), ManagementPrivilege::Clear, now_ms()) + .await, + Err(CatalogError::Busy) + )); + drop(repository); + stack.chunk_kv.restart().await; + let repository = CatalogRepository::new(stack.store().await, bounds).unwrap(); + let (recovering, _) = repository.status().await.unwrap(); + if let crowdb_access_iceberg::catalog::RootState::Published(transition) = recovering.state { + let remaining = transition.complete_after_ms.saturating_sub(now_ms()); + tokio::time::sleep(Duration::from_millis(remaining)).await; + } + repository.recover(now_ms()).await.unwrap(); + let replacement = execute(&repository, clear.clone()).await; + assert_ne!(replacement.catalog, original.catalog); + let second = request(ManagementAction::Clear, "second", Some((2, &replacement))); + let latest = execute(&repository, second).await; + assert_ne!(latest.catalog, replacement.catalog); + assert_eq!(execute(&repository, clear).await, replacement); + assert_eq!(execute(&repository, initialize).await, original); + assert_eq!(repository.status().await.unwrap().0.context.activation_epoch, 3); + common::activate(&repository).await; + verify_retry_scan(&stack, &repository).await; + drop(frontend); + drop(second_frontend); + namespace::verify_name_index(&stack, latest.catalog).await; + let frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let second_frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + background::verify(stack.store().await, repository.status().await.unwrap().0.context).await; + frontend.check_official_reads(); + second_frontend.check_official_reads(); + drop(frontend); + drop(second_frontend); + journal::verify_recovery(&mut stack, repository.status().await.unwrap().0.context).await; + verify_interrupted_clear(&stack, &repository).await; + common::activate(&repository).await; + let frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let second_frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + frontend.check_official_reads(); + second_frontend.check_official_reads(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn namespace_functional_crud_survives_native_storage_and_listener_restart() { + let mut stack = TestIcebergStack::start().await; + let repository = CatalogRepository::new( + stack.store().await, + ClearBounds { + request_ms: 300_000, + ..ClearBounds::default() + }, + ) + .unwrap(); + execute( + &repository, + request(ManagementAction::Initialize, "functional", None), + ) + .await; + common::activate(&repository).await; + let frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let second_frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + frontend.check_official_client(); + second_frontend.check_official_client(); + let client = reqwest::Client::builder() + .timeout(Duration::from_secs(5)) + .build() + .unwrap(); + let created = client + .post(format!("http://{}/v1/namespaces", frontend.address)) + .bearer_auth("w".repeat(32)) + .header("content-type", "application/json") + .body(r#"{"namespace":["persisted"],"properties":{"owner":"before-restart"}}"#) + .send() + .await + .unwrap(); + assert_eq!(created.status(), 200, "{}", created.text().await.unwrap()); + verify_retained_namespace(&client, &second_frontend).await; + drop(frontend); + drop(second_frontend); + stack.chunk_kv.restart().await; + let frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let second_frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + for listener in [&frontend, &second_frontend] { + verify_retained_namespace(&client, listener).await; + listener.check_official_client(); + } +} + +async fn verify_retained_namespace(client: &reqwest::Client, listener: &process::TestIcebergProcess) { + let response = client + .get(format!("http://{}/v1/namespaces/persisted", listener.address)) + .bearer_auth("r".repeat(32)) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 200); + let body: serde_json::Value = serde_json::from_slice(&response.bytes().await.unwrap()).unwrap(); + assert_eq!(body["namespace"], serde_json::json!(["persisted"])); + assert_eq!(body["properties"], serde_json::json!({"owner":"before-restart"})); +} + +async fn verify_interrupted_clear(stack: &TestIcebergStack, repository: &CatalogRepository) { + for mode in [1, 2] { + let (root, authority) = repository.status().await.unwrap(); + let clear = request( + ManagementAction::Clear, + "recovered", + Some((root.context.activation_epoch, &authority)), + ); + let faulty = CatalogRepository::new( + Arc::new(fault::TestFaultStore { + inner: stack.store().await, + mode: std::sync::atomic::AtomicU8::new(mode), + }), + authority.admission_bounds, + ) + .unwrap(); + assert!(matches!( + faulty + .execute(clear.clone(), ManagementPrivilege::Clear, now_ms()) + .await, + Err(CatalogError::Store(_)) + )); + drop(faulty); + let recovered = execute(repository, clear.clone()).await; + assert_ne!(recovered.catalog, authority.catalog); + assert_eq!( + repository.status().await.unwrap().0.context.activation_epoch, + root.context.activation_epoch + 1 + ); + assert_eq!(execute(repository, clear).await, recovered); + } +} + +async fn verify_retry_scan(stack: &TestIcebergStack, repository: &CatalogRepository) { + use crowdb_access_iceberg::key::IcebergKey; + use crowdb_access_iceberg::operation::{RetryAdmission, RetryLedger, RetryRecord}; + use crowdb_chunk_kv_client::MultiScanRequest; + use crowdb_protocol::chunk_kv::ScanDirection; + let store = stack.store().await; + let ledger = RetryLedger::new(store.clone()); + let context = repository.status().await.unwrap().0.context; + for _ in 0..3 { + let request = RetryRecord { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "reader".into(), + route: "test mutation".into(), + digest: [1; 32], + context, + retained_until_ms: 0, + status: 0, + body: Vec::new(), + }; + assert!(matches!( + ledger.begin(request.clone(), now_ms()).await.unwrap(), + RetryAdmission::New(_) + )); + assert!(!ledger + .finish(request.clone(), 503, Vec::new(), now_ms()) + .await + .unwrap()); + assert!(matches!( + ledger.begin(request.clone(), now_ms()).await.unwrap(), + RetryAdmission::Resume(_) + )); + ledger + .finish(request.clone(), 409, b"conflict".to_vec(), now_ms()) + .await + .unwrap(); + let RetryAdmission::Replay(result) = RetryLedger::new(stack.store().await) + .begin(request, now_ms()) + .await + .unwrap() + else { + panic!("durable result replay") + }; + assert_eq!(result.status, 409); + assert_eq!(result.body, b"conflict"); + } + let range = IcebergKey::catalog_range(context.catalog); + let mut scan = MultiScanRequest { + start: Some(range.start.clone()), + end: Some(range.end.clone()), + direction: ScanDirection::Forward, + max_items: 1, + max_bytes: 64 * 1024, + continuation: None, + }; + let mut count = 0; + loop { + let page = store.scan(scan.clone()).await.unwrap(); + assert!(page.items.len() <= 1); + for item in &page.items { + assert!(range.contains(&item.key)); + } + count += page.items.len(); + scan.continuation = page.continuation; + if scan.continuation.is_none() { + break; + } + assert!(count <= 4); + } + assert_eq!(count, 4); + scan.start = None; + assert!(store.scan(scan).await.is_err()); +} diff --git a/app/crowdb-access-server/tests/iceberg_gc_budget_test.rs b/app/crowdb-access-server/tests/iceberg_gc_budget_test.rs new file mode 100644 index 000000000..77220d091 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_gc_budget_test.rs @@ -0,0 +1,135 @@ +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; + +use async_trait::async_trait; +use crowdb_access_iceberg::file::{ChunkRoot, FileBlockStore, FileIdentity, FileIoError}; +use crowdb_access_iceberg::{ + catalog::{CasOutcome, CatalogStore, StoreError, StoredValue}, + gc::{GcScan, GcStore, GcSystemScan}, + record::MAX_RECORD_BYTES, +}; +use crowdb_access_server::iceberg::{BudgetedGcBlocks, BudgetedGcStore, GcIoBudget}; +use crowdb_chunk_client::ReclaimOutcome; +use crowdb_chunk_kv_client::MultiScanPage; +use crowdb_protocol::chunk_kv::ClientRequestId; +use crowdb_protocol::common::ChunkId; + +#[derive(Default)] +struct TestBlocks { + reads: AtomicUsize, + deletes: AtomicUsize, +} + +#[derive(Default)] +struct TestStore { + gets: AtomicUsize, +} + +#[async_trait] +impl CatalogStore for TestStore { + async fn get(&self, _key: &[u8]) -> Result, StoreError> { + self.gets.fetch_add(1, Ordering::Relaxed); + Ok(None) + } + + async fn compare_exchange( + &self, + _key: &[u8], + _expected: Option<&[u8]>, + _value: &[u8], + _identity: ClientRequestId, + ) -> Result { + Ok(CasOutcome::Conflict(None)) + } +} + +#[async_trait] +impl GcStore for TestStore { + async fn scan_gc(&self, _request: GcScan) -> Result { + Ok(MultiScanPage { + items: Vec::new(), + continuation: None, + terminal_failure: None, + }) + } + + async fn scan_gc_system(&self, _request: GcSystemScan) -> Result { + Ok(MultiScanPage { + items: Vec::new(), + continuation: None, + terminal_failure: None, + }) + } + + async fn delete_gc_record( + &self, + _key: &[u8], + _expected: &[u8], + _identity: ClientRequestId, + ) -> Result { + Ok(CasOutcome::Conflict(None)) + } +} + +#[async_trait] +impl FileBlockStore for TestBlocks { + async fn put(&self, _owner: FileIdentity, _height: u8, _bytes: &[u8]) -> Result { + Err(FileIoError::Bounds) + } + + async fn read(&self, root: &ChunkRoot) -> Result, FileIoError> { + self.reads.fetch_add(1, Ordering::Relaxed); + Ok(vec![7; usize::try_from(root.logical_length).unwrap()]) + } + + async fn reclaim(&self, _root: &ChunkRoot) -> Result { + self.deletes.fetch_add(1, Ordering::Relaxed); + Ok(ReclaimOutcome::Reclaimed) + } +} + +#[tokio::test] +async fn chunk_io_budget_rejects_work_before_dispatch_and_resets_per_step() { + let budget = Arc::new(GcIoBudget::for_tests(1024, 8, 24, 2)); + let inner = Arc::new(TestBlocks::default()); + let blocks = BudgetedGcBlocks::new(inner.clone(), budget.clone()); + let root = ChunkRoot { + chunk: ChunkId { high: 1, low: 1 }, + offset: 0, + physical_length: 16, + logical_offset: 0, + logical_length: 16, + height: 0, + digest: [7; 32], + }; + assert_eq!(blocks.read(&root).await.unwrap().len(), 16); + assert!(matches!(blocks.read(&root).await, Err(FileIoError::Bounds))); + assert_eq!(inner.reads.load(Ordering::Relaxed), 1); + assert!(matches!(blocks.reclaim(&root).await, Err(FileIoError::Bounds))); + assert_eq!(inner.deletes.load(Ordering::Relaxed), 0); + budget.reset(); + assert_eq!(blocks.reclaim(&root).await.unwrap(), ReclaimOutcome::Reclaimed); + assert_eq!(inner.deletes.load(Ordering::Relaxed), 1); +} + +#[tokio::test] +async fn kv_budget_rejects_work_before_dispatch_and_resets_per_step() { + let budget = Arc::new(GcIoBudget::for_tests( + u64::try_from(MAX_RECORD_BYTES).unwrap() + 1, + 1, + 24, + 2, + )); + let inner = Arc::new(TestStore::default()); + let store = BudgetedGcStore::new(inner.clone(), budget.clone()); + assert!(store.get(b"x").await.unwrap().is_none()); + assert!(matches!(store.get(b"x").await, Err(StoreError::Budget))); + assert_eq!(inner.gets.load(Ordering::Relaxed), 1); + assert!(inner.get(b"x").await.unwrap().is_none()); + assert_eq!(inner.gets.load(Ordering::Relaxed), 2); + budget.reset(); + assert!(store.get(b"x").await.unwrap().is_none()); + assert_eq!(inner.gets.load(Ordering::Relaxed), 3); +} diff --git a/app/crowdb-access-server/tests/iceberg_gc_capacity_test.rs b/app/crowdb-access-server/tests/iceberg_gc_capacity_test.rs new file mode 100644 index 000000000..3a169b60f --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_gc_capacity_test.rs @@ -0,0 +1,419 @@ +#[path = "common/iceberg_stack.rs"] +#[allow(dead_code)] +mod common; +#[path = "common/iceberg_gc_capacity.rs"] +mod gc_capacity; + +use std::sync::Arc; + +use common::TestIcebergStack; +use crowdb_access_iceberg::{ + catalog::{CatalogContext, CatalogRepository, CatalogStore, ClearBounds, ManagementPrivilege}, + file::{ + file_key, ContentFormat, FileContent, FileIdentity, FileKind, FileReader, FileRecord, FileRepository, + FileTreeWriter, NativeFileBlocks, TableLocation, + }, + gc::{GcLimits, GcPhase, GcRepository, GcStalledReason, GcTask, GcWorker}, + key::{FileId, OperationId, TableId}, + operation::{mutation_identity, ManagementAction, ManagementRequest, RequestIdentity}, + record::StorageRecord, + table::{head_key, TableHead, TableLifecycle, TablePurgeTask}, +}; +use crowdb_chunk_client::{ChunkIoClient, ChunkIoClientConfig, SmallWritePolicy}; +use crowdb_diskdb_client::{DiskdbClient, DiskdbClientError, DiskdbRpcTransport}; +use crowdb_protocol::{ + common::ChunkId, + diskdb::rpc::{ + AllocateBlocksRequest, CommitBlocksRequest, CompactZoneRequest, FreeBlocksRequest, Segment, + }, +}; +use sha2::{Digest, Sha256}; + +async fn seed_catalog(stack: &TestIcebergStack) -> crowdb_access_iceberg::catalog::CatalogContext { + let repository = CatalogRepository::new(stack.store().await, ClearBounds::default()).unwrap(); + let now = common::now_ms(); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now, + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "capacity".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + now, + ) + .await + .unwrap(); + repository.status().await.unwrap().0.context +} + +async fn chunks(stack: &TestIcebergStack) -> ChunkIoClient { + ChunkIoClient::connect(ChunkIoClientConfig { + management_seeds: stack.cluster.mgmt_endpoints.clone(), + diskio_connections_per_endpoint: 1, + diskio_rpc_workers: 1, + small_write: SmallWritePolicy { + min_pipelines: 1, + max_pipelines: 1, + memory_budget: 8 * 1024 * 1024, + chunk_capacity: 1024 * 1024 * 1024, + mirror_copies: 1, + conversion_enabled: false, + ..SmallWritePolicy::default() + }, + }) + .await + .unwrap() +} + +async fn fill_disk(client: &DiskdbClient) -> Vec { + let mut held = Vec::new(); + for sequence in 1..=256_u64 { + let mut allocated = None; + for units in [1024, 128, 1] { + match client + .allocate_blocks(AllocateBlocksRequest { + disk_group_id: 100, + unit_count: units, + count: 1, + exclude_disk_ids: Vec::new(), + owner_chunk: Some(ChunkId { + high: 77, + low: sequence, + }), + allow_disk_reuse: false, + }) + .await + { + Ok(response) => { + allocated = Some(response.segments); + break; + } + Err(DiskdbClientError::NoSpace(_)) => {} + Err(error) => panic!("disk allocation failed unexpectedly: {error}"), + } + } + let Some(segments) = allocated else { break }; + assert_eq!( + client + .commit_blocks(CommitBlocksRequest { + segments: segments.clone() + }) + .await + .unwrap() + .committed_count, + u32::try_from(segments.len()).unwrap() + ); + held.extend(segments); + } + assert!(!held.is_empty()); + assert_eq!( + client.query_disk_group(100).await.unwrap().disk_groups[0].free_bytes, + 0 + ); + held +} + +async fn write_file( + stack: &TestIcebergStack, + client: &ChunkIoClient, + owner: FileIdentity, +) -> Result { + let blocks = Arc::new(NativeFileBlocks::new(client.clone(), stack.store().await)); + let mut writer = FileTreeWriter::new(blocks, owner, 16 * 1024).unwrap(); + writer.push(&vec![31; 32 * 1024]).await?; + let tree = writer.finish().await?; + Ok(FileRecord { + file: owner.file, + location: owner.table.file("data/capacity.parquet").unwrap(), + kind: FileKind::Data, + format: ContentFormat::Parquet, + length: tree.length, + digest: tree.digest, + content: FileContent::Chunks { root: tree.root }, + hint: None, + }) +} + +async fn read_file(stack: &TestIcebergStack, client: &ChunkIoClient, file: FileRecord) -> Vec { + let blocks = Arc::new(NativeFileBlocks::new(client.clone(), stack.store().await)); + let mut reader = FileReader::new(blocks, file, None, 4096).unwrap(); + let mut bytes = Vec::new(); + while let Some(frame) = reader.next().await.unwrap() { + bytes.extend_from_slice(&frame); + } + bytes +} + +async fn seed_gc_workspace_task( + stack: &TestIcebergStack, + context: CatalogContext, +) -> ( + Arc, + GcRepository, + GcTask, + FileId, + GcLimits, +) { + let store = stack.store().await; + let table = TableLocation { + catalog: context.catalog, + table: TableId::random(), + }; + let metadata = FileRecord { + file: FileId::random(), + location: table.file("metadata/gc-candidate.json").unwrap(), + kind: FileKind::Metadata, + format: ContentFormat::Json, + length: 2, + digest: Sha256::digest(b"{}").into(), + content: FileContent::select_inline(FileKind::Metadata, b"{}").unwrap(), + hint: None, + }; + FileRepository::new(store.clone()) + .publish(context, &metadata) + .await + .unwrap(); + let head = TableHead { + catalog: context.catalog, + table: table.table, + namespace: crowdb_access_iceberg::key::NamespaceId::random(), + name: "gc-capacity".into(), + name_epoch: 1, + lifecycle: TableLifecycle::Tombstone, + generation: 1, + metadata_file: metadata.file, + metadata_location: metadata.location, + metadata_digest: metadata.digest, + format_version: 1, + table_uuid: None, + operation_fence: 2, + pending_operation: Some(OperationId::random()), + }; + let key = head_key(context.catalog, table.table).encode().unwrap(); + let bytes = StorageRecord::TableHead(Box::new(head.clone())).encode().unwrap(); + store + .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) + .await + .unwrap(); + let marker = TablePurgeTask { + activation_epoch: context.activation_epoch, + head: head.clone(), + }; + let key = marker.key().encode().unwrap(); + let bytes = StorageRecord::TablePurgeTask(Box::new(marker)).encode().unwrap(); + store + .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) + .await + .unwrap(); + let workspace = Arc::new(gc_capacity::TestGcWorkspace::new(store)); + let repository = GcRepository::new(workspace.clone()); + let limits = GcLimits { + minimum_retention_ms: 1, + ..GcLimits::default() + }; + let task = GcTask::plan( + context, + OperationId::random(), + Some(head), + common::now_ms(), + limits, + ) + .unwrap(); + repository.create(&task).await.unwrap(); + (workspace, repository, task, metadata.file, limits) +} + +async fn run_gc_until_resource_stall(worker: &GcWorker, mut task: GcTask) -> GcTask { + for _ in 0..20 { + task = worker + .run(&task, common::now_ms().max(task.retry_at_ms)) + .await + .unwrap(); + if task.stalled == GcStalledReason::Resource { + break; + } + } + assert_eq!(task.phase, GcPhase::Discover); + assert_eq!(task.stalled, GcStalledReason::Resource); + assert_eq!(task.deleted, 0); + task +} + +async fn run_gc_until_complete(worker: &GcWorker, mut task: GcTask) -> GcTask { + for _ in 0..300 { + task = worker + .run(&task, common::now_ms().max(task.retry_at_ms)) + .await + .unwrap(); + if task.phase == GcPhase::Complete { + break; + } + } + assert_eq!(task.phase, GcPhase::Complete, "{task:?}"); + task +} + +async fn assert_gc_workspace_stall( + stack: &TestIcebergStack, + client: &ChunkIoClient, + workspace: &Arc, + repository: &GcRepository, + task: GcTask, + file: FileId, + limits: GcLimits, +) -> GcTask { + workspace.deny(true); + let blocks = Arc::new(NativeFileBlocks::new(client.clone(), stack.store().await)); + let worker = GcWorker::new(repository.clone(), blocks, limits).unwrap(); + let stalled = run_gc_until_resource_stall(&worker, task).await; + assert_eq!( + worker.run(&stalled, stalled.retry_at_ms - 1).await.unwrap(), + stalled + ); + assert!(stack + .store() + .await + .get(&file_key(stalled.context.catalog, file).encode().unwrap()) + .await + .unwrap() + .is_some()); + assert_eq!( + repository + .task(stalled.context.catalog, stalled.identity) + .await + .unwrap(), + Some(stalled.clone()) + ); + stalled +} + +async fn assert_gc_workspace_recovered( + stack: &TestIcebergStack, + client: &ChunkIoClient, + workspace: &Arc, + repository: GcRepository, + task: GcTask, + file: FileId, + limits: GcLimits, +) { + workspace.deny(false); + let blocks = Arc::new(NativeFileBlocks::new(client.clone(), stack.store().await)); + let worker = GcWorker::new(repository, blocks, limits).unwrap(); + let finished = run_gc_until_complete(&worker, task).await; + assert!(finished.deleted >= 1); + assert!(stack + .store() + .await + .get(&file_key(finished.context.catalog, file).encode().unwrap()) + .await + .unwrap() + .is_none()); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn full_simulated_disk_preserves_file_authority_then_recovers_after_compaction() { + let stack = TestIcebergStack::start().await; + let context = seed_catalog(&stack).await; + let owner = FileIdentity { + table: TableLocation { + catalog: context.catalog, + table: TableId::random(), + }, + file: FileId::random(), + }; + let committed_owner = FileIdentity { + table: owner.table, + file: FileId::random(), + }; + let client = chunks(&stack).await; + let mut committed = write_file(&stack, &client, committed_owner).await.unwrap(); + committed.location = owner.table.file("data/committed.parquet").unwrap(); + let files = FileRepository::new(stack.store().await); + files.publish(context, &committed).await.unwrap(); + let (workspace, gc_repository, gc_task, gc_file, gc_limits) = + seed_gc_workspace_task(&stack, context).await; + client.shutdown_small_writes().await.unwrap(); + drop(client); + let disk = DiskdbClient::new( + stack.cluster.make_service_registry_client(), + Arc::new(DiskdbRpcTransport::new()), + ); + disk.refresh_endpoints().await.unwrap(); + let held = fill_disk(&disk).await; + let client = chunks(&stack).await; + let failure = write_file(&stack, &client, owner).await.unwrap_err(); + assert!(matches!( + failure, + crowdb_access_iceberg::file::FileIoError::Write(_) + )); + let location = owner.table.file("data/capacity.parquet").unwrap(); + assert!(files.load(context, &location).await.unwrap().is_none()); + assert_eq!( + files.load(context, &committed.location).await.unwrap(), + Some(committed.clone()) + ); + assert_eq!( + read_file(&stack, &client, committed.clone()).await, + vec![31; 32 * 1024] + ); + let stalled = assert_gc_workspace_stall( + &stack, + &client, + &workspace, + &gc_repository, + gc_task, + gc_file, + gc_limits, + ) + .await; + drop(client); + for batch in held.chunks(100) { + assert_eq!( + disk.free_blocks(FreeBlocksRequest { + segments: batch.to_vec() + }) + .await + .unwrap() + .freed_count as usize, + batch.len() + ); + } + let compacted = disk + .compact_zone(CompactZoneRequest { + disk_id: held[0].disk_id, + zone_indices: Vec::new(), + }) + .await + .unwrap(); + assert!(compacted.zones.iter().all(|zone| zone.success)); + assert!(disk.query_disk_group(100).await.unwrap().disk_groups[0].free_bytes > 0); + let client = chunks(&stack).await; + let file = write_file(&stack, &client, owner).await.unwrap(); + files.publish(context, &file).await.unwrap(); + assert_eq!(files.load(context, &location).await.unwrap(), Some(file.clone())); + assert_eq!(read_file(&stack, &client, file).await, vec![31; 32 * 1024]); + assert_gc_workspace_recovered( + &stack, + &client, + &workspace, + gc_repository, + stalled, + gc_file, + gc_limits, + ) + .await; + assert_eq!( + files.load(context, &committed.location).await.unwrap(), + Some(committed.clone()) + ); + assert_eq!(read_file(&stack, &client, committed).await, vec![31; 32 * 1024]); + client.shutdown_small_writes().await.unwrap(); +} diff --git a/app/crowdb-access-server/tests/iceberg_gc_control_test.rs b/app/crowdb-access-server/tests/iceberg_gc_control_test.rs new file mode 100644 index 000000000..d9c6b6732 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_gc_control_test.rs @@ -0,0 +1,529 @@ +#[path = "common/iceberg_stack.rs"] +#[allow(dead_code)] +mod common; +#[path = "common/iceberg_process.rs"] +#[allow(dead_code)] +mod process; + +use crowdb_access_iceberg::{ + catalog::{ + CatalogContext, CatalogRepository, CatalogStore, ClearBounds, ManagementPrivilege, RoutedCatalogStore, + }, + file::{file_key, ContentFormat, FileContent, FileKind, FileRecord, TableLocation}, + gc::{GcLimits, GcRepository, GcStalledReason, ReaderPins}, + key::{FileId, NamespaceId, OperationId, TableId}, + operation::{mutation_identity, ManagementAction, ManagementRequest, RequestIdentity}, + record::StorageRecord, + table::{head_key, TableHead, TableLifecycle, TablePurgeTask}, +}; +use sha2::{Digest, Sha256}; +use std::sync::Arc; + +fn command(stack: &common::TestIcebergStack, token: char, arguments: &[&str]) -> std::process::Output { + process::command(&stack.cluster.mgmt_endpoints) + .env("CROWDB_ICEBERG_TOKEN", token.to_string().repeat(32)) + .arg("gc") + .args(arguments) + .output() + .unwrap() +} + +fn response(output: std::process::Output) -> serde_json::Value { + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let stdout = String::from_utf8(output.stdout).unwrap(); + let json = stdout + .lines() + .find(|line| line.starts_with('{')) + .expect("GC command JSON is missing"); + serde_json::from_str(json).unwrap() +} + +async fn seed_table(stack: &common::TestIcebergStack) -> (Arc, CatalogContext, TableId) { + let store = stack.store().await; + let catalog = CatalogRepository::new( + store.clone(), + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + ) + .unwrap(); + let now = common::now_ms(); + catalog + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now, + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "gc-control".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + now, + ) + .await + .unwrap(); + common::activate(&catalog).await; + let context = catalog.status().await.unwrap().0.context; + let table = seed_head(store.as_ref(), context).await; + (store, context, table) +} + +async fn seed_head(store: &RoutedCatalogStore, context: CatalogContext) -> TableId { + let table = TableId::random(); + let location = TableLocation { + catalog: context.catalog, + table, + }; + let head = TableHead { + catalog: context.catalog, + table, + namespace: NamespaceId::random(), + name: "items".into(), + name_epoch: 1, + lifecycle: TableLifecycle::Ready, + generation: 1, + metadata_file: FileId::random(), + metadata_location: location.file("metadata/first.json").unwrap(), + metadata_digest: [7; 32], + format_version: 1, + table_uuid: None, + operation_fence: 1, + pending_operation: None, + }; + let key = head_key(context.catalog, table).encode().unwrap(); + let bytes = StorageRecord::TableHead(Box::new(head)).encode().unwrap(); + store + .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) + .await + .unwrap(); + table +} + +async fn check_operator_pin( + stack: &common::TestIcebergStack, + store: Arc, + context: CatalogContext, + table: TableId, +) { + let pin_id = OperationId::random().to_string(); + let table_id = table.to_string(); + let catalog_id = context.catalog.to_string(); + response(command(stack, 'm', &["pin", &pin_id, &table_id])); + let pin_identity = pin_id.parse().unwrap(); + let pins = ReaderPins::new(store); + assert!(pins + .get(context.catalog, table, pin_identity) + .await + .unwrap() + .unwrap() + .protects(common::now_ms())); + response(command(stack, 'm', &["unpin", &catalog_id, &table_id, &pin_id])); + assert!(!pins + .get(context.catalog, table, pin_identity) + .await + .unwrap() + .unwrap() + .protects(common::now_ms())); +} + +async fn tombstone_head(store: &RoutedCatalogStore, context: CatalogContext, table: TableId) { + let key = head_key(context.catalog, table); + let previous = store.get(&key.encode().unwrap()).await.unwrap().unwrap(); + let StorageRecord::TableHead(mut head) = StorageRecord::decode(&key, &previous.bytes).unwrap() else { + panic!("table head"); + }; + head.lifecycle = TableLifecycle::Tombstone; + head.operation_fence += 1; + head.pending_operation = Some(OperationId::random()); + let next = StorageRecord::TableHead(head).encode().unwrap(); + let key = key.encode().unwrap(); + store + .compare_exchange( + &key, + Some(&previous.bytes), + &next, + mutation_identity(&key, Some(&previous.bytes), &next), + ) + .await + .unwrap(); +} + +async fn seed_files(store: &RoutedCatalogStore, context: CatalogContext, table: TableId, count: usize) { + for index in 0..count { + let payload_json = format!("{{\"index\":{index}}}"); + let file = FileRecord { + file: FileId::random(), + location: TableLocation { + catalog: context.catalog, + table, + } + .file(&format!("metadata/backlog-{index}.json")) + .unwrap(), + kind: FileKind::Metadata, + format: ContentFormat::Json, + length: payload_json.len() as u64, + digest: Sha256::digest(payload_json.as_bytes()).into(), + content: FileContent::select_inline(FileKind::Metadata, payload_json.as_bytes()).unwrap(), + hint: None, + }; + let key = file_key(context.catalog, file.file).encode().unwrap(); + let bytes = StorageRecord::File(Box::new(file)).encode().unwrap(); + store + .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) + .await + .unwrap(); + } +} + +async fn seed_purge_marker(store: &RoutedCatalogStore, context: CatalogContext, table: TableId) { + tombstone_head(store, context, table).await; + let key = head_key(context.catalog, table); + let head = store.get(&key.encode().unwrap()).await.unwrap().unwrap(); + let StorageRecord::TableHead(head) = StorageRecord::decode(&key, &head.bytes).unwrap() else { + panic!("table head"); + }; + let marker = TablePurgeTask { + activation_epoch: context.activation_epoch, + head: *head, + }; + let encoded = marker.key().encode().unwrap(); + let bytes = StorageRecord::TablePurgeTask(Box::new(marker)).encode().unwrap(); + store + .compare_exchange(&encoded, None, &bytes, mutation_identity(&encoded, None, &bytes)) + .await + .unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn authenticated_gc_controls_survive_separate_processes() { + let stack = common::TestIcebergStack::start().await; + let (store, context, table) = seed_table(&stack).await; + let identity = OperationId::random().to_string(); + let table_id = table.to_string(); + let catalog_id = context.catalog.to_string(); + let denied = command(&stack, 'w', &["start-table", &identity, &table_id]); + assert!(!denied.status.success()); + let live = command(&stack, 'm', &["start-table", &identity, &table_id]); + assert!(!live.status.success()); + assert!(String::from_utf8_lossy(&live.stderr).contains("live-table GC is disabled")); + + check_operator_pin(&stack, store.clone(), context, table).await; + tombstone_head(store.as_ref(), context, table).await; + let created = response(command(&stack, 'm', &["start-table", &identity, &table_id])); + assert_eq!(created["phase"], "Discover"); + assert_eq!(created["task_id"], identity); + let repeated = response(command(&stack, 'm', &["start-table", &identity, &table_id])); + assert_eq!(repeated["revision"], created["revision"]); + let paused = response(command(&stack, 'm', &["pause", &catalog_id, &identity])); + assert_eq!(paused["paused"], true); + let resumed = response(command(&stack, 'm', &["resume", &catalog_id, &identity])); + assert_eq!(resumed["paused"], false); + + let gc = GcRepository::new(store.clone()); + let current = gc + .task(context.catalog, identity.parse().unwrap()) + .await + .unwrap() + .unwrap(); + let stalled = gc + .defer( + ¤t, + GcStalledReason::Storage, + common::now_ms(), + GcLimits::default(), + ) + .await + .unwrap(); + let inspected = response(command(&stack, 'm', &["inspect", &catalog_id, &identity])); + assert_eq!(inspected["revision"], stalled.revision); + assert_eq!(inspected["stalled"], "Storage"); + let retried = response(command(&stack, 'm', &["retry", &catalog_id, &identity])); + assert_eq!(retried["stalled"], "None"); + assert_eq!(retried["attempts"], 0); + + let server = process::TestIcebergProcess::start_with_gc(&stack.cluster.mgmt_endpoints, true).await; + check_foreground_namespace(&server); + let client = reqwest::Client::new(); + for _ in 0..3 { + let response = client + .get(format!("http://{}/v1/config", server.address)) + .bearer_auth("r".repeat(32)) + .send() + .await + .unwrap(); + assert_eq!(response.status(), reqwest::StatusCode::OK); + } + let progress = tokio::time::timeout(std::time::Duration::from_secs(10), async { + loop { + let progress = gc + .task(context.catalog, identity.parse().unwrap()) + .await + .unwrap() + .unwrap(); + if progress.revision > retried["revision"].as_u64().unwrap() { + break progress; + } + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); + assert!(progress.revision > stalled.revision); + let available = command(&stack, 'm', &["inspect", &catalog_id, &identity]); + assert!(available.status.success()); + drop(server); + let restarted = process::TestIcebergProcess::start_with_gc(&stack.cluster.mgmt_endpoints, true).await; + let persisted = gc + .task(context.catalog, identity.parse().unwrap()) + .await + .unwrap() + .unwrap(); + assert!(persisted.revision >= progress.revision); + drop(restarted); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn enabled_scheduler_admits_durable_purge_markers_once() { + let stack = common::TestIcebergStack::start().await; + let (store, context, table) = seed_table(&stack).await; + seed_purge_marker(store.as_ref(), context, table).await; + let server = process::TestIcebergProcess::start_with_gc(&stack.cluster.mgmt_endpoints, true).await; + let repository = GcRepository::new(store); + let identity = OperationId::from_bytes(table.as_bytes()).unwrap(); + let task = tokio::time::timeout(std::time::Duration::from_secs(10), async { + loop { + if let Some(task) = repository.task(context.catalog, identity).await.unwrap() { + break task; + } + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); + assert_eq!(task.kind, crowdb_access_iceberg::gc::GcTaskKind::PurgeTable); + assert_eq!(task.head.as_ref().unwrap().table, table); + drop(server); + let restarted = process::TestIcebergProcess::start_with_gc(&stack.cluster.mgmt_endpoints, true).await; + let resumed = repository.task(context.catalog, identity).await.unwrap().unwrap(); + assert_eq!(resumed.created_ms, task.created_ms); + assert_eq!(resumed.head, task.head); + drop(restarted); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn enabled_scheduler_admits_and_advances_completed_clear() { + let stack = common::TestIcebergStack::start().await; + let (store, old, _) = seed_table(&stack).await; + seed_files(store.as_ref(), old, TableId::random(), 48).await; + let catalog = CatalogRepository::new( + store.clone(), + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + ) + .unwrap(); + let now_ms = common::now_ms(); + let identity = OperationId::random(); + let clear = ManagementRequest { + identity: RequestIdentity { + operation: identity, + issued_ms: now_ms, + }, + principal: "manager".into(), + action: ManagementAction::Clear, + expected_epoch: old.activation_epoch, + display_name: "gc-replacement".into(), + confirmation: Some(old.catalog), + capabilities: None, + }; + assert!(catalog + .execute(clear.clone(), ManagementPrivilege::Clear, now_ms) + .await + .is_err()); + let crowdb_access_iceberg::catalog::RootState::Published(transition) = + catalog.status().await.unwrap().0.state + else { + panic!("expected published maintenance"); + }; + catalog + .execute(clear, ManagementPrivilege::Clear, transition.complete_after_ms) + .await + .unwrap(); + let active = catalog.status().await.unwrap().0.context; + let table = seed_head(store.as_ref(), active).await; + seed_purge_marker(store.as_ref(), active, table).await; + let server = process::TestIcebergProcess::start_with_gc_settings( + &stack.cluster.mgmt_endpoints, + true, + &[("CROWDB_ICEBERG_GC_PAGE_ITEMS", "1")], + ) + .await; + let repository = GcRepository::new(store); + let task = tokio::time::timeout(std::time::Duration::from_secs(15), async { + loop { + if let Some(task) = repository.task(old.catalog, identity).await.unwrap() { + if task.revision > 1 { + break task; + } + } + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); + assert_eq!(task.kind, crowdb_access_iceberg::gc::GcTaskKind::RetiredCatalog); + let active_task = tokio::time::timeout(std::time::Duration::from_secs(10), async { + loop { + if let Some(task) = repository + .task(active.catalog, OperationId::from_bytes(table.as_bytes()).unwrap()) + .await + .unwrap() + { + if task.revision > 1 { + break task; + } + } + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); + assert_eq!( + active_task.kind, + crowdb_access_iceberg::gc::GcTaskKind::PurgeTable + ); + let retired = repository.task(old.catalog, identity).await.unwrap().unwrap(); + assert_eq!(retired.phase, crowdb_access_iceberg::gc::GcPhase::Discover); + drop(server); +} + +fn check_foreground_namespace(server: &process::TestIcebergProcess) { + if let Ok(python) = std::env::var("CROWDB_ICEBERG_E2E_PYTHON") { + let script = r#"import sys +from pyiceberg.catalog import load_catalog +from pyiceberg.schema import Schema +from pyiceberg.types import LongType, NestedField +catalog = load_catalog("crowdb", type="rest", uri=sys.argv[1], token="w" * 32) +namespace = ("gc_foreground",) +catalog.create_namespace(namespace) +assert catalog.namespace_exists(namespace) +identifier = namespace + ("events",) +table = catalog.create_table(identifier, Schema(NestedField(field_id=1, name="id", field_type=LongType(), required=True))) +table.transaction().set_properties({"gc-probe": "committed"}).commit_transaction() +assert catalog.load_table(identifier).properties["gc-probe"] == "committed" +catalog.drop_table(identifier) +catalog.drop_namespace(namespace) +assert not catalog.namespace_exists(namespace) +"#; + let status = std::process::Command::new(python) + .arg("-c") + .arg(script) + .arg(format!("http://{}", server.address)) + .status() + .unwrap(); + assert!(status.success()); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires the pinned PyIceberg environment"] +async fn official_sdk_foreground_progresses_under_gc_backlog() { + let python = std::env::var_os("CROWDB_ICEBERG_E2E_PYTHON") + .expect("run with the pinned iceberg-e2e pixi environment"); + let stack = common::TestIcebergStack::start().await; + let (store, context, table) = seed_table(&stack).await; + seed_files(store.as_ref(), context, table, 128).await; + seed_purge_marker(store.as_ref(), context, table).await; + let server = process::TestIcebergProcess::start_with_gc_settings( + &stack.cluster.mgmt_endpoints, + true, + &[("CROWDB_ICEBERG_GC_PAGE_ITEMS", "1")], + ) + .await; + let script = r#"import sys +from concurrent.futures import ThreadPoolExecutor +import requests +from pyiceberg.catalog import load_catalog +from pyiceberg.schema import Schema +from pyiceberg.types import LongType, NestedField + +def run(worker): + catalog = load_catalog(f"gc-{worker}", type="rest", uri=sys.argv[1], token="w" * 32) + namespace = (f"gc-pressure-{worker}",) + catalog.create_namespace(namespace) + for index in range(3): + identifier = namespace + (f"events-{index}",) + table = catalog.create_table(identifier, Schema(NestedField(field_id=1, name="id", field_type=LongType(), required=True))) + table.transaction().set_properties({"gc-probe": str(index)}).commit_transaction() + loaded = catalog.load_table(identifier) + assert loaded.properties["gc-probe"] == str(index) + response = requests.get( + f"{sys.argv[1]}/v1/namespaces/{namespace[0]}/tables/{identifier[1]}/credentials", + headers={"Authorization": "Bearer " + "w" * 32}, + timeout=5, + ) + response.raise_for_status() + loaded.io.properties.update(response.json()["storage-credentials"][0]["config"]) + with loaded.io.new_input(loaded.metadata_location).open() as stream: + assert stream.read().startswith(b"{") + catalog.drop_table(identifier) + catalog.drop_namespace(namespace) + +with ThreadPoolExecutor(max_workers=4) as executor: + list(executor.map(run, range(4))) +"#; + let mut client = std::process::Command::new(python) + .arg("-c") + .arg(script) + .arg(format!("http://{}", server.address)) + .spawn() + .unwrap(); + let repository = GcRepository::new(store); + let identity = OperationId::from_bytes(table.as_bytes()).unwrap(); + let overlapped = tokio::time::timeout(std::time::Duration::from_secs(90), async { + loop { + if client.try_wait().unwrap().is_some() { + break false; + } + if repository + .task(context.catalog, identity) + .await + .unwrap() + .is_some_and(|task| task.revision > 1) + { + break true; + } + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); + let status = tokio::time::timeout(std::time::Duration::from_secs(90), async { + loop { + if let Some(status) = client.try_wait().unwrap() { + break status; + } + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); + assert!(status.success(), "official SDK foreground operations failed"); + assert!( + overlapped, + "GC did not advance while the SDK requests were active" + ); +} diff --git a/app/crowdb-access-server/tests/iceberg_http_test.rs b/app/crowdb-access-server/tests/iceberg_http_test.rs new file mode 100644 index 000000000..b5c7a0232 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_http_test.rs @@ -0,0 +1,123 @@ +#[path = "common/iceberg_store.rs"] +mod common; + +use crowdb_access_iceberg::catalog::{CatalogRepository, ClearBounds, ManagementPrivilege}; +use crowdb_access_iceberg::key::OperationId; +use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; +use crowdb_access_iceberg::wire::BearerAuthenticator; +use crowdb_access_server::iceberg::{serve, IcebergHttpService}; +use std::sync::Arc; +use std::time::Duration; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::{TcpListener, TcpStream}; + +#[tokio::test] +async fn authenticated_config_warehouse_errors_and_shutdown_use_real_http() { + let store = Arc::new(common::TestStore::default()); + let repository = Arc::new(CatalogRepository::new(store.clone(), ClearBounds::default()).unwrap()); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: 100, + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "catalog".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + 100, + ) + .await + .unwrap(); + common::activate(&repository).await; + let authentication = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let service = Arc::new(IcebergHttpService::new( + repository, + authentication, + Duration::from_secs(2), + )); + let observed = service.clone(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let (stop, stopped) = tokio::sync::oneshot::channel(); + let server = tokio::spawn(async move { + serve(listener, service, async { + let _ = stopped.await; + }) + .await + .unwrap(); + }); + for (path, expected) in [ + ("/v1/config", 200), + ("/v1/config?warehouse=", 200), + ("/v1/config?warehouse=unknown", 404), + ("/v1/config?warehouse=%ZZ", 400), + ("/v1/config?warehouse=&warehouse=", 400), + ("/v1/namespaces", 406), + ] { + let response = get(address, path, &"r".repeat(32)).await; + assert!( + response.starts_with(&format!("HTTP/1.1 {expected}")), + "{response}" + ); + let body = response.split_once("\r\n\r\n").unwrap().1; + let json: serde_json::Value = serde_json::from_str(body).unwrap(); + if expected == 200 { + assert_eq!(json["endpoints"], serde_json::json!([])); + } + if expected == 404 { + assert_eq!(json["error"]["type"], "NoSuchWarehouseException"); + } + } + assert!(get(address, "/v1/config", &"w".repeat(32)) + .await + .starts_with("HTTP/1.1 200")); + assert!(get(address, "/v1/config", "wrong") + .await + .starts_with("HTTP/1.1 401")); + let mut incomplete = TcpStream::connect(address).await.unwrap(); + incomplete + .write_all(b"GET /v1/config HTTP/1.1\r\nHost: localhost\r\n") + .await + .unwrap(); + let mut bytes = Vec::new(); + tokio::time::timeout(Duration::from_millis(2500), incomplete.read_to_end(&mut bytes)) + .await + .unwrap() + .unwrap(); + assert!(bytes.is_empty()); + store + .read_delay_ms + .store(3000, std::sync::atomic::Ordering::SeqCst); + let expired = tokio::time::timeout( + Duration::from_millis(2500), + get(address, "/v1/config", &"r".repeat(32)), + ) + .await + .unwrap(); + assert!(!expired.starts_with("HTTP/1.1 200")); + store.read_delay_ms.store(0, std::sync::atomic::Ordering::SeqCst); + stop.send(()).unwrap(); + tokio::time::timeout(Duration::from_secs(3), server) + .await + .unwrap() + .unwrap(); + let metrics = observed.metrics_snapshot(); + assert_eq!(metrics.routes[0][0].requests, 3); + assert_eq!(metrics.routes[0][1].requests, 1); + assert_eq!(metrics.routes[0][4].requests + metrics.routes[0][6].requests, 1); +} + +async fn get(address: std::net::SocketAddr, path: &str, token: &str) -> String { + let mut stream = TcpStream::connect(address).await.unwrap(); + stream.write_all(format!("GET {path} HTTP/1.1\r\nHost: localhost\r\nAuthorization: Bearer {token}\r\nConnection: close\r\n\r\n").as_bytes()).await.unwrap(); + let mut bytes = Vec::new(); + stream.read_to_end(&mut bytes).await.unwrap(); + String::from_utf8(bytes).unwrap() +} diff --git a/app/crowdb-access-server/tests/iceberg_java_response_loss_test.rs b/app/crowdb-access-server/tests/iceberg_java_response_loss_test.rs new file mode 100644 index 000000000..e83ee7905 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_java_response_loss_test.rs @@ -0,0 +1,73 @@ +#![cfg(feature = "iceberg-e2e")] + +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; +#[path = "common/iceberg_response_loss.rs"] +mod response_loss; + +use std::{sync::Arc, time::Duration}; + +use crowdb_access_iceberg::{ + catalog::{CatalogRepository, ClearBounds}, + wire::BearerAuthenticator, +}; +use crowdb_access_server::iceberg::{serve, IcebergHttpService}; +use response_loss::TestResponseLossProxy; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires Maven and pinned Apache Iceberg Java dependencies"] +async fn official_java_client_observes_lost_create_reply_on_another_listener() { + let fixture = fixture::TestTableHttp::writable().await; + let service = IcebergHttpService::new( + Arc::new(CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap()), + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(), + Duration::from_secs(2), + ) + .with_namespaces(fixture.store.clone()) + .unwrap() + .with_tables(fixture.store.clone(), Arc::new(blocks::TestFileBlocks::default())) + .unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let second_origin = format!("http://{}", listener.local_addr().unwrap()); + let (stop, stopped) = tokio::sync::oneshot::channel(); + let server = tokio::spawn(async move { + serve(listener, Arc::new(service), async { + let _ = stopped.await; + }) + .await + .unwrap(); + }); + let proxy = TestResponseLossProxy::start(fixture.endpoint(), "/v1/namespaces/analytics/tables").await; + let origin = proxy.origin.clone(); + let status = tokio::task::spawn_blocking(move || { + let maven = std::env::var_os("CROWDB_ICEBERG_E2E_MVN").unwrap_or_else(|| "mvn".into()); + std::process::Command::new("timeout") + .arg("60") + .arg(maven) + .args(["-o", "--batch-mode", "--no-transfer-progress", "-f"]) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_java/pom.xml" + )) + .args(["compile", "exec:java", "-Dexec.mainClass=TestIcebergResponseLoss"]) + .arg(format!("-Dexec.args={origin} {second_origin}")) + .status() + .unwrap() + }) + .await + .unwrap(); + proxy.assert_dropped(); + stop.send(()).unwrap(); + server.await.unwrap(); + assert!( + status.success(), + "official Java REST client response-loss acceptance failed" + ); + fixture.finish().await; +} diff --git a/app/crowdb-access-server/tests/iceberg_namespace_http_test.rs b/app/crowdb-access-server/tests/iceberg_namespace_http_test.rs new file mode 100644 index 000000000..d25a30390 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_namespace_http_test.rs @@ -0,0 +1,275 @@ +#[path = "common/iceberg_store.rs"] +mod common; + +use crowdb_access_iceberg::catalog::{CatalogContext, CatalogRepository, ClearBounds, ManagementPrivilege}; +use crowdb_access_iceberg::key::OperationId; +use crowdb_access_iceberg::namespace::{ + NamespaceCreateRequest, NamespaceCreator, NamespaceIdentifier, NamespaceProperties, +}; +use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; +use crowdb_access_iceberg::wire::BearerAuthenticator; +use crowdb_access_server::iceberg::{serve, IcebergHttpService}; +use std::sync::{atomic::Ordering, Arc}; +use std::time::Duration; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::{TcpListener, TcpStream}; + +async fn setup() -> ( + Arc, + CatalogContext, + std::net::SocketAddr, + tokio::sync::oneshot::Sender<()>, + tokio::task::JoinHandle<()>, +) { + let store = Arc::new(common::TestStore::default()); + let repository = Arc::new(CatalogRepository::new(store.clone(), ClearBounds::default()).unwrap()); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: 100, + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "catalog".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + 100, + ) + .await + .unwrap(); + common::activate(&repository).await; + let context = repository.status().await.unwrap().0.context; + let auth = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let service = Arc::new( + IcebergHttpService::new(repository, auth, Duration::from_secs(2)) + .with_namespaces(store.clone()) + .unwrap(), + ); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let (stop, stopped) = tokio::sync::oneshot::channel(); + let server = tokio::spawn(async move { + serve(listener, service, async { + let _ = stopped.await; + }) + .await + .unwrap(); + }); + (store, context, address, stop, server) +} + +async fn create(store: Arc, context: CatalogContext, names: &[&str]) { + let request = NamespaceCreateRequest { + context, + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: 100, + }, + principal: "writer".into(), + identifier: NamespaceIdentifier::new(names.iter().map(|name| (*name).into()).collect()).unwrap(), + properties: NamespaceProperties::default(), + }; + NamespaceCreator::new(store).create(&request).await.unwrap(); +} + +async fn send(address: std::net::SocketAddr, method: &str, path: &str) -> (u16, String) { + let mut stream = TcpStream::connect(address).await.unwrap(); + stream.write_all(format!("{method} {path} HTTP/1.1\r\nHost: localhost\r\nAuthorization: Bearer {}\r\nConnection: close\r\n\r\n", "r".repeat(32)).as_bytes()).await.unwrap(); + let mut bytes = Vec::new(); + stream.read_to_end(&mut bytes).await.unwrap(); + let response = String::from_utf8(bytes).unwrap(); + let (headers, body) = response.split_once("\r\n\r\n").unwrap(); + ( + headers.split_whitespace().nth(1).unwrap().parse().unwrap(), + body.to_owned(), + ) +} + +#[tokio::test] +async fn namespace_reads_preserve_single_decoding_and_page_token_semantics() { + let (store, context, address, stop, server) = setup().await; + assert_eq!(send(address, "GET", "/v1/namespaces/parent/tables").await.0, 406); + for names in [&["parent"][..], &["parent", "a+b"], &["parent", "%2F"]] { + create(store.clone(), context, names).await; + } + let (status, body) = send(address, "GET", "/v1/namespaces/parent%1Fa+b").await; + assert_eq!(status, 200); + assert_eq!( + serde_json::from_str::(&body).unwrap()["namespace"], + serde_json::json!(["parent", "a+b"]) + ); + assert_eq!(send(address, "GET", "/v1/namespaces/parent%1F%252F").await.0, 200); + assert_eq!(send(address, "GET", "/v1/namespaces/parent%1F%2F").await.0, 404); + assert_eq!( + send(address, "HEAD", "/v1/namespaces/parent").await, + (204, String::new()) + ); + assert_eq!( + send(address, "HEAD", "/v1/namespaces/absent").await, + (404, String::new()) + ); + let (status, body) = send(address, "GET", "/v1/namespaces?parent=parent&pageSize=1").await; + assert_eq!(status, 200); + let complete: serde_json::Value = serde_json::from_str(&body).unwrap(); + assert_eq!(complete["namespaces"].as_array().unwrap().len(), 2); + assert!(complete["next-page-token"].is_null()); + let (_, body) = send( + address, + "GET", + "/v1/namespaces?parent=parent&pageSize=1&pageToken=", + ) + .await; + let page: serde_json::Value = serde_json::from_str(&body).unwrap(); + assert_eq!(page["namespaces"].as_array().unwrap().len(), 1); + let token = page["next-page-token"].as_str().unwrap(); + assert_eq!( + send( + address, + "GET", + &format!("/v1/namespaces?parent=parent&pageSize=2&pageToken={token}") + ) + .await + .0, + 400 + ); + assert_eq!( + send( + address, + "GET", + &format!("/v1/namespaces?parent=parent&pageSize=1&pageToken={token}") + ) + .await + .0, + 200 + ); + stop.send(()).unwrap(); + server.await.unwrap(); +} + +#[tokio::test] +async fn complete_list_spool_admission_is_bounded_and_released() { + let (store, _, address, stop, server) = setup().await; + store.scan_delay_ms.store(500, Ordering::SeqCst); + let mut requests = Vec::new(); + for _ in 0..4 { + requests.push(tokio::spawn(send(address, "GET", "/v1/namespaces"))); + } + tokio::time::timeout(Duration::from_secs(1), async { + while store.scans.load(Ordering::SeqCst) < 4 { + tokio::time::sleep(Duration::from_millis(1)).await; + } + }) + .await + .unwrap(); + assert_eq!(send(address, "GET", "/v1/namespaces").await.0, 503); + for request in requests { + assert_eq!(request.await.unwrap().0, 200); + } + store.scan_delay_ms.store(0, Ordering::SeqCst); + assert_eq!(send(address, "GET", "/v1/namespaces").await.0, 200); + stop.send(()).unwrap(); + server.await.unwrap(); +} + +#[tokio::test] +async fn complete_list_deadline_returns_503_and_releases_all_spool_slots() { + let (store, _, address, stop, server) = setup().await; + store.scan_delay_ms.store(2500, Ordering::SeqCst); + let (status, body) = send(address, "GET", "/v1/namespaces").await; + assert_eq!(status, 503); + let error: serde_json::Value = serde_json::from_str(&body).unwrap(); + assert_eq!(error["error"]["code"], 503); + assert!(error.get("namespaces").is_none()); + store.scan_delay_ms.store(100, Ordering::SeqCst); + let mut requests = Vec::new(); + for _ in 0..4 { + requests.push(tokio::spawn(send(address, "GET", "/v1/namespaces"))); + } + for request in requests { + assert_eq!(request.await.unwrap().0, 200); + } + stop.send(()).unwrap(); + server.await.unwrap(); +} + +#[tokio::test] +async fn complete_list_exhaustion_never_returns_a_truncated_success() { + use crowdb_access_iceberg::catalog::StoredValue; + use crowdb_access_iceberg::key::NamespaceId; + use crowdb_access_iceberg::namespace::{ + authority_key, name_key, NamespaceAuthority, NamespaceLifecycle, NamespaceMapping, + NamespaceMappingState, + }; + use crowdb_access_iceberg::record::StorageRecord; + let (store, context, address, stop, server) = setup().await; + let baseline = store.values.load_full(); + for (count, padding, stale) in [(1025, 0, false), (400, 3500, false), (4100, 0, true)] { + let mut values = (*baseline).clone(); + for index in 0..count { + let name = format!("{index:04}{}", "\"".repeat(padding)); + let namespace = NamespaceId::random(); + let mapping = NamespaceMapping { + catalog: context.catalog, + parent: None, + name: name.clone(), + namespace, + name_epoch: 1, + operation: OperationId::random(), + state: NamespaceMappingState::Published, + }; + values.insert( + name_key(context.catalog, None, &name).unwrap().encode().unwrap(), + StoredValue { + bytes: StorageRecord::NamespaceMapping(mapping).encode().unwrap(), + revision: 1, + }, + ); + if !stale { + let authority = NamespaceAuthority { + catalog: context.catalog, + namespace, + parent: None, + identifier: NamespaceIdentifier::new(vec![name]).unwrap(), + name_epoch: 1, + property_revision: 1, + admission_fence: 1, + mutation_revision: 1, + lifecycle: NamespaceLifecycle::Ready, + pending_operation: None, + properties: NamespaceProperties::default(), + }; + values.insert( + authority_key(context.catalog, namespace).encode().unwrap(), + StoredValue { + bytes: StorageRecord::NamespaceAuthority(Box::new(authority)) + .encode() + .unwrap(), + revision: 1, + }, + ); + } + } + store.values.store(Arc::new(values)); + let (status, body) = send(address, "GET", "/v1/namespaces").await; + assert_eq!(status, 503, "count={count}, padding={padding}, stale={stale}"); + let error: serde_json::Value = serde_json::from_str(&body).unwrap(); + assert_eq!(error["error"]["code"], 503); + assert!(error.get("namespaces").is_none()); + assert_eq!( + send(address, "GET", "/v1/namespaces?pageSize=1&pageToken=") + .await + .0, + 200 + ); + store.values.store(baseline.clone()); + assert_eq!(send(address, "GET", "/v1/namespaces").await.0, 200); + } + stop.send(()).unwrap(); + server.await.unwrap(); +} diff --git a/app/crowdb-access-server/tests/iceberg_namespace_limits_test.rs b/app/crowdb-access-server/tests/iceberg_namespace_limits_test.rs new file mode 100644 index 000000000..a7d065a9a --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_namespace_limits_test.rs @@ -0,0 +1,137 @@ +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +use std::collections::BTreeMap; + +use crowdb_access_iceberg::namespace::{NamespaceIdentifier, NamespaceProperties, NamespaceRepository}; +use crowdb_access_iceberg::record::{StorageRecord, MAX_RECORD_BYTES}; +use fixture::TestTableHttp; +use reqwest::Method; +use serde_json::{json, Value}; + +const NAMESPACE: &str = "/v1/namespaces/analytics"; +const PROPERTIES: &str = "/v1/namespaces/analytics/properties"; + +async fn properties(test: &TestTableHttp) -> Value { + let response = test.request(Method::GET, NAMESPACE, "r", None).await; + assert_eq!(response.status().as_u16(), 200); + response.json::().await.unwrap()["properties"].clone() +} + +async fn update(test: &TestTableHttp, changes: Value, expected: u16) -> Value { + let before = properties(test).await; + let authority_key = crowdb_access_iceberg::namespace::authority_key(test.context.catalog, test.namespace) + .encode() + .unwrap(); + let authority = test.store.values.load()[&authority_key].clone(); + let response = test.post(PROPERTIES, "w", None, &changes).await; + let status = response.status().as_u16(); + let body = response.text().await.unwrap(); + assert_eq!(status, expected, "{body}"); + let result: Value = serde_json::from_str(&body).unwrap(); + if expected != 200 { + assert_eq!(result["error"]["code"], expected); + assert_eq!(properties(test).await, before); + let after = test.store.values.load()[&authority_key].clone(); + assert_eq!(after.bytes, authority.bytes); + assert_eq!(after.revision, authority.revision); + } + result +} + +#[tokio::test] +async fn property_cardinality_replacement_and_overlap_are_atomic_over_http() { + let test = TestTableHttp::new().await; + let mut full: BTreeMap<_, _> = (0..256).map(|index| (index.to_string(), "value")).collect(); + update(&test, json!({"updates": full}), 200).await; + assert_eq!(properties(&test).await, json!(full)); + update(&test, json!({"updates":{"overflow":"value"}}), 400).await; + let result = update( + &test, + json!({"removals":["0"],"updates":{"replacement":"new"}}), + 200, + ) + .await; + assert_eq!(result["removed"], json!(["0"])); + assert_eq!(result["updated"], json!(["replacement"])); + let retained = properties(&test).await; + assert_eq!(retained.as_object().unwrap().len(), 256); + assert!(retained.get("0").is_none()); + assert_eq!(retained["replacement"], "new"); + full.remove("0"); + full.insert("replacement".into(), "new"); + assert_eq!(retained, json!(full)); + update( + &test, + json!({"removals":["replacement"],"updates":{"replacement":"bad"}}), + 422, + ) + .await; + test.finish().await; +} + +#[tokio::test] +async fn property_utf8_key_and_value_byte_limits_fail_without_mutation() { + let test = TestTableHttp::new().await; + let key = format!("{}a", "键".repeat(341)); + let value = "值".repeat(2730) + "ab"; + update(&test, json!({"updates":{key.clone():value.clone()}}), 200).await; + assert_eq!(properties(&test).await, json!({key.clone():value.clone()})); + for (invalid_key, invalid_value) in [ + (key.clone() + "b", value.clone()), + (key.clone(), value + "c"), + ("nul\0key".into(), "valid".into()), + ("valid".into(), "nul\0value".into()), + ] { + update(&test, json!({"updates":{invalid_key:invalid_value}}), 400).await; + } + test.finish().await; +} + +#[tokio::test] +async fn encoded_authority_limit_is_checked_before_http_property_publication() { + let test = TestTableHttp::new().await; + let mut authority = NamespaceRepository::new(test.store.clone()) + .load( + test.context, + &NamespaceIdentifier::new(vec!["analytics".into()]).unwrap(), + ) + .await + .unwrap() + .unwrap(); + let mut entries: BTreeMap<_, _> = (0..7) + .map(|index| (index.to_string(), "v".repeat(8192))) + .collect(); + let mut lower = 0_usize; + let mut upper = 8192_usize; + while lower < upper { + let middle = (lower + upper).div_ceil(2); + entries.insert("tail".into(), "v".repeat(middle)); + authority.properties = NamespaceProperties::new(entries.clone()).unwrap(); + if StorageRecord::NamespaceAuthority(Box::new(authority.clone())) + .encode() + .is_ok() + { + lower = middle; + } else { + upper = middle - 1; + } + } + assert!(lower > 0 && lower < 8192); + entries.insert("tail".into(), "v".repeat(lower)); + authority.properties = NamespaceProperties::new(entries.clone()).unwrap(); + let encoded = StorageRecord::NamespaceAuthority(Box::new(authority)) + .encode() + .unwrap(); + assert!(encoded.len() <= MAX_RECORD_BYTES && encoded.len() + 8 > MAX_RECORD_BYTES); + update(&test, json!({"updates":entries}), 200).await; + assert_eq!(properties(&test).await, json!(entries)); + update(&test, json!({"updates":{"tail":"v".repeat(lower + 1)}}), 400).await; + test.finish().await; +} diff --git a/app/crowdb-access-server/tests/iceberg_namespace_sdk_test.rs b/app/crowdb-access-server/tests/iceberg_namespace_sdk_test.rs new file mode 100644 index 000000000..ed852053f --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_namespace_sdk_test.rs @@ -0,0 +1,167 @@ +#![cfg(feature = "iceberg-e2e")] + +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +use std::sync::{atomic::Ordering, Arc}; + +use crowdb_access_iceberg::{ + catalog::StoredValue, + key::{NamespaceId, OperationId}, + namespace::{ + authority_key, name_key, NamespaceAuthority, NamespaceIdentifier, NamespaceLifecycle, + NamespaceMapping, NamespaceMappingState, NamespaceProperties, + }, + record::StorageRecord, +}; +use fixture::TestTableHttp; + +fn install(test: &TestTableHttp, count: usize, padding: usize, stale: bool) { + let mut values = (*test.store.values.load_full()).clone(); + for index in 0..count { + let name = format!("{index:04}{}", "\"".repeat(padding)); + let namespace = NamespaceId::random(); + let mapping = NamespaceMapping { + catalog: test.context.catalog, + parent: None, + name: name.clone(), + namespace, + name_epoch: 1, + operation: OperationId::random(), + state: NamespaceMappingState::Published, + }; + values.insert( + name_key(test.context.catalog, None, &name) + .unwrap() + .encode() + .unwrap(), + StoredValue { + bytes: StorageRecord::NamespaceMapping(mapping).encode().unwrap(), + revision: 1, + }, + ); + if !stale { + let authority = NamespaceAuthority { + catalog: test.context.catalog, + namespace, + parent: None, + identifier: NamespaceIdentifier::new(vec![name]).unwrap(), + name_epoch: 1, + property_revision: 1, + admission_fence: 1, + mutation_revision: 1, + lifecycle: NamespaceLifecycle::Ready, + pending_operation: None, + properties: NamespaceProperties::default(), + }; + values.insert( + authority_key(test.context.catalog, namespace).encode().unwrap(), + StoredValue { + bytes: StorageRecord::NamespaceAuthority(Box::new(authority)) + .encode() + .unwrap(), + revision: 1, + }, + ); + } + } + test.store.values.store(Arc::new(values)); +} + +async fn python(test: &TestTableHttp, mode: &str, count: usize) { + let endpoint = test.endpoint(); + let mode = mode.to_owned(); + let status = tokio::task::spawn_blocking(move || { + let python = std::env::var_os("CROWDB_ICEBERG_E2E_PYTHON").expect("set pinned Python path"); + std::process::Command::new("timeout") + .arg("60") + .arg(python) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_namespace_client.py" + )) + .args([endpoint, mode, count.to_string()]) + .status() + .unwrap() + }) + .await + .unwrap(); + assert!( + status.success(), + "official namespace complete-list acceptance failed" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires pinned PyIceberg environment"] +async fn official_complete_listing_rejects_each_spool_limit_and_releases_resources() { + let test = TestTableHttp::new().await; + let baseline = test.store.values.load_full(); + test.store.scan_delay_ms.store(500, Ordering::SeqCst); + python(&test, "concurrency", 1).await; + test.store.scan_delay_ms.store(0, Ordering::SeqCst); + python(&test, "complete", 1).await; + for (count, padding, stale) in [(1024, 0, false), (400, 3500, false), (4100, 0, true)] { + install(&test, count, padding, stale); + python(&test, "overflow", 0).await; + test.store.values.store(baseline.clone()); + python(&test, "complete", 1).await; + } + test.store.scan_delay_ms.store(2500, Ordering::SeqCst); + python(&test, "overflow", 0).await; + test.store.scan_delay_ms.store(0, Ordering::SeqCst); + python(&test, "complete", 1).await; + install(&test, 10, 0, false); + python(&test, "complete", 11).await; + test.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires Maven and pinned Apache Iceberg Java dependencies"] +async fn official_catalog_continues_through_empty_namespace_pages() { + let test = TestTableHttp::new().await; + install(&test, 3, 0, true); + let mapping = NamespaceMapping { + catalog: test.context.catalog, + parent: None, + name: "zzzzzzzzz".into(), + namespace: NamespaceId::random(), + name_epoch: 1, + operation: OperationId::random(), + state: NamespaceMappingState::Published, + }; + test.put( + &name_key(test.context.catalog, None, &mapping.name).unwrap(), + &StorageRecord::NamespaceMapping(mapping.clone()), + ); + let endpoint = test.endpoint(); + let status = tokio::task::spawn_blocking(move || { + let maven = std::env::var_os("CROWDB_ICEBERG_E2E_MVN").expect("set pinned Maven path"); + std::process::Command::new("timeout") + .arg("60") + .arg(maven) + .args(["-o", "--batch-mode", "--no-transfer-progress", "-f"]) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_java/pom.xml" + )) + .args(["compile", "exec:java", "-Dexec.mainClass=TestIcebergNamespaces"]) + .arg(format!("-Dexec.args={endpoint}")) + .status() + .unwrap() + }) + .await + .unwrap(); + assert_eq!(test.store.scans.load(Ordering::SeqCst), 7); + test.finish().await; + assert!( + status.success(), + "official namespace pagination acceptance failed" + ); +} diff --git a/app/crowdb-access-server/tests/iceberg_namespace_write_http_test.rs b/app/crowdb-access-server/tests/iceberg_namespace_write_http_test.rs new file mode 100644 index 000000000..e73cd62a0 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_namespace_write_http_test.rs @@ -0,0 +1,429 @@ +#[path = "common/iceberg_store.rs"] +mod common; + +use crowdb_access_iceberg::catalog::{CatalogRepository, CatalogStore, ClearBounds, ManagementPrivilege}; +use crowdb_access_iceberg::key::{IcebergKey, OperationId, SystemScope}; +use crowdb_access_iceberg::operation::{ledger_key, ManagementAction, ManagementRequest, RequestIdentity}; +use crowdb_access_iceberg::wire::BearerAuthenticator; +use crowdb_access_server::iceberg::{serve, IcebergHttpService}; +use std::sync::Arc; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::{TcpListener, TcpStream}; + +async fn setup() -> ( + Arc, + std::net::SocketAddr, + tokio::sync::oneshot::Sender<()>, + tokio::task::JoinHandle<()>, +) { + let store = Arc::new(common::TestStore::default()); + let repository = Arc::new(CatalogRepository::new(store.clone(), ClearBounds::default()).unwrap()); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: 100, + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "catalog".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + 100, + ) + .await + .unwrap(); + common::activate(&repository).await; + let auth = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let service = Arc::new( + IcebergHttpService::new(repository, auth, Duration::from_secs(2)) + .with_namespaces(store.clone()) + .unwrap(), + ); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let (stop, stopped) = tokio::sync::oneshot::channel(); + let server = tokio::spawn(async move { + serve(listener, service, async { + let _ = stopped.await; + }) + .await + .unwrap(); + }); + (store, address, stop, server) +} + +async fn fresh_key(store: &common::TestStore) -> String { + for _ in 0..32 { + let mut bytes = *OperationId::random().as_bytes(); + let now = u64::try_from(SystemTime::now().duration_since(UNIX_EPOCH).unwrap().as_millis()).unwrap(); + bytes[..6].copy_from_slice(&now.to_be_bytes()[2..]); + bytes[6] = (bytes[6] & 15) | 0x70; + bytes[8] = (bytes[8] & 63) | 0x80; + let operation = OperationId::from_bytes(&bytes).unwrap(); + if store + .get( + &ledger_key(SystemScope::RetryBinding, operation) + .unwrap() + .encode() + .unwrap(), + ) + .await + .unwrap() + .is_some() + { + continue; + } + return wire_key(operation); + } + panic!("no free retry fixture slot"); +} + +fn wire_key(operation: OperationId) -> String { + let hex = operation.to_string(); + format!( + "{}-{}-{}-{}-{}", + &hex[..8], + &hex[8..12], + &hex[12..16], + &hex[16..20], + &hex[20..] + ) +} + +async fn send( + address: std::net::SocketAddr, + method: &str, + path: &str, + principal: &str, + key: Option<&str>, + body: &str, +) -> (u16, String) { + let mut stream = TcpStream::connect(address).await.unwrap(); + let idempotency = key.map_or_else(String::new, |key| format!("Idempotency-Key: {key}\r\n")); + stream.write_all(format!("{method} {path} HTTP/1.1\r\nHost: localhost\r\nAuthorization: Bearer {}\r\nContent-Type: application/json\r\nContent-Length: {}\r\n{idempotency}Connection: close\r\n\r\n{body}", principal.repeat(32), body.len()).as_bytes()).await.unwrap(); + let mut bytes = Vec::new(); + stream.read_to_end(&mut bytes).await.unwrap(); + let response = String::from_utf8(bytes).unwrap(); + let (headers, body) = response.split_once("\r\n\r\n").unwrap(); + ( + headers.split_whitespace().nth(1).unwrap().parse().unwrap(), + body.to_owned(), + ) +} + +#[tokio::test] +async fn namespace_writes_use_distinct_credentials_and_replay_success_and_terminal_errors() { + let (store, address, stop, server) = setup().await; + let body = r#"{"namespace":["parent"],"properties":{"owner":"original"}}"#; + let before = store.values.load().len(); + for principal in ["r", "m", "c"] { + assert_eq!( + send(address, "POST", "/v1/namespaces", principal, None, body) + .await + .0, + 403 + ); + } + assert_eq!(store.values.load().len(), before); + let key = fresh_key(&store).await; + let created = send(address, "POST", "/v1/namespaces", "w", Some(&key), body).await; + assert_eq!(created.0, 200); + assert_eq!( + send(address, "POST", "/v1/namespaces", "w", Some(&key), body).await, + created + ); + assert_eq!( + send( + address, + "POST", + "/v1/namespaces", + "w", + Some(&key), + r#"{"namespace":["different"]}"# + ) + .await + .0, + 409 + ); + let duplicate_key = fresh_key(&store).await; + let duplicate = send(address, "POST", "/v1/namespaces", "w", Some(&duplicate_key), body).await; + assert_eq!(duplicate.0, 409); + verify_property_replay(&store, address).await; + verify_drop_replay(&store, address, &duplicate_key, &duplicate, body).await; + stop.send(()).unwrap(); + server.await.unwrap(); +} + +async fn verify_property_replay(store: &common::TestStore, address: std::net::SocketAddr) { + let update_key = fresh_key(store).await; + let changes = r#"{"removals":["owner","missing"],"updates":{"value":"kept"}}"#; + let updated = send( + address, + "POST", + "/v1/namespaces/parent/properties", + "w", + Some(&update_key), + changes, + ) + .await; + assert_eq!(updated.0, 200); + assert_eq!( + send( + address, + "POST", + "/v1/namespaces/parent/properties", + "w", + Some(&update_key), + changes + ) + .await, + updated + ); + assert_eq!( + send( + address, + "POST", + "/v1/namespaces/parent/properties", + "w", + None, + r#"{"removals":["value"],"updates":{"value":"bad"}}"# + ) + .await + .0, + 422 + ); +} + +async fn verify_drop_replay( + store: &common::TestStore, + address: std::net::SocketAddr, + duplicate_key: &str, + duplicate: &(u16, String), + body: &str, +) { + let drop_key = fresh_key(store).await; + assert_eq!( + send( + address, + "DELETE", + "/v1/namespaces/parent", + "w", + Some(&drop_key), + "" + ) + .await + .0, + 204 + ); + assert_eq!( + send(address, "POST", "/v1/namespaces", "w", Some(duplicate_key), body).await, + *duplicate + ); + assert_eq!( + send(address, "POST", "/v1/namespaces", "w", None, body).await.0, + 200 + ); + assert_eq!( + send( + address, + "DELETE", + "/v1/namespaces/parent", + "w", + Some(&drop_key), + "" + ) + .await + .0, + 204 + ); + assert_eq!( + send(address, "GET", "/v1/namespaces/parent", "r", None, "") + .await + .0, + 200 + ); +} + +#[tokio::test] +async fn malformed_and_missing_parent_results_are_retained_before_any_later_retry() { + let (store, address, stop, server) = setup().await; + let key = fresh_key(&store).await; + let body = r#"{"namespace":["missing","child"]}"#; + let failed = send(address, "POST", "/v1/namespaces", "w", Some(&key), body).await; + assert_eq!(failed.0, 400); + assert_eq!( + send( + address, + "POST", + "/v1/namespaces", + "w", + None, + r#"{"namespace":["missing"]}"# + ) + .await + .0, + 200 + ); + assert_eq!( + send(address, "POST", "/v1/namespaces", "w", Some(&key), body).await, + failed + ); + let invalid_key = fresh_key(&store).await; + let malformed = send(address, "POST", "/v1/namespaces", "w", Some(&invalid_key), "{").await; + assert_eq!(malformed.0, 400); + assert_eq!( + send(address, "POST", "/v1/namespaces", "w", Some(&invalid_key), "{").await, + malformed + ); + assert_eq!( + send(address, "POST", "/v1/namespaces", "w", Some("invalid"), "{}") + .await + .0, + 400 + ); + assert_eq!( + send(address, "POST", "/v1/namespaces", "w", None, body).await.0, + 200 + ); + assert_eq!( + send(address, "DELETE", "/v1/namespaces/missing", "w", None, "") + .await + .0, + 409 + ); + stop.send(()).unwrap(); + server.await.unwrap(); +} + +#[tokio::test] +async fn colliding_uuidv7_headers_have_independent_durable_http_replay() { + let (store, address, stop, server) = setup().await; + let now = u64::try_from(SystemTime::now().duration_since(UNIX_EPOCH).unwrap().as_millis()).unwrap(); + let mut seen = std::collections::BTreeMap::new(); + let mut collision = None; + for sequence in 0_u16..=4096 { + let mut bytes = [0; 16]; + bytes[..6].copy_from_slice(&now.to_be_bytes()[2..]); + bytes[6] = 0x70; + bytes[8] = 0x80; + bytes[14..].copy_from_slice(&sequence.to_be_bytes()); + let operation = OperationId::from_bytes(&bytes).unwrap(); + let slot = ledger_key(SystemScope::RetryBinding, operation) + .unwrap() + .encode() + .unwrap(); + if let Some(first) = seen.insert(slot, operation) { + collision = Some((first, operation)); + break; + } + } + let (first, second) = collision.expect("4097 identities must collide in 4096 slots"); + let first_key = wire_key(first); + let second_key = wire_key(second); + let first_body = r#"{"namespace":["first"]}"#; + let second_body = r#"{"namespace":["second"]}"#; + let first_response = send( + address, + "POST", + "/v1/namespaces", + "w", + Some(&first_key), + first_body, + ) + .await; + let second_response = send( + address, + "POST", + "/v1/namespaces", + "w", + Some(&second_key), + second_body, + ) + .await; + assert_eq!(first_response.0, 200); + assert_eq!(second_response.0, 200); + assert_eq!( + send( + address, + "POST", + "/v1/namespaces", + "w", + Some(&first_key), + first_body + ) + .await, + first_response + ); + assert_eq!( + send( + address, + "POST", + "/v1/namespaces", + "w", + Some(&second_key), + second_body + ) + .await, + second_response + ); + let overflow = IcebergKey::System { + scope: SystemScope::RetryOverflow, + suffix: second.as_bytes().to_vec(), + }; + assert!(store.values.load().contains_key(&overflow.encode().unwrap())); + stop.send(()).unwrap(); + server.await.unwrap(); +} + +#[tokio::test] +async fn server_errors_leave_recoverable_publication_and_large_final_responses() { + let (store, address, stop, server) = setup().await; + let properties: std::collections::BTreeMap<_, _> = (0..7) + .map(|index| (format!("key{index}"), "v".repeat(8190))) + .collect(); + for mode in [1, 2, 3] { + let key = fresh_key(&store).await; + let namespace = format!("lost-{mode}"); + let body = + serde_json::to_string(&serde_json::json!({"namespace": [namespace], "properties": properties})) + .unwrap(); + store + .lose_reply_kind + .store(mode, std::sync::atomic::Ordering::SeqCst); + assert_eq!( + send(address, "POST", "/v1/namespaces", "w", Some(&key), &body) + .await + .0, + 503 + ); + let resumed = send(address, "POST", "/v1/namespaces", "w", Some(&key), &body).await; + assert_eq!(resumed.0, 200); + assert!(resumed.1.len() > 16 * 1024); + assert_eq!( + send(address, "POST", "/v1/namespaces", "w", Some(&key), &body).await, + resumed + ); + assert_eq!( + send( + address, + "GET", + &format!("/v1/namespaces/{namespace}"), + "r", + None, + "" + ) + .await + .0, + 200 + ); + } + stop.send(()).unwrap(); + server.await.unwrap(); +} diff --git a/app/crowdb-access-server/tests/iceberg_rck_test.rs b/app/crowdb-access-server/tests/iceberg_rck_test.rs new file mode 100644 index 000000000..d8941e3fd --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_rck_test.rs @@ -0,0 +1,114 @@ +#![cfg(feature = "iceberg-e2e")] + +#[path = "common/iceberg_stack.rs"] +#[allow(dead_code)] +mod common; +#[path = "common/iceberg_process.rs"] +#[allow(dead_code)] +mod process; + +use std::time::Duration; + +use common::{now_ms, TestIcebergStack}; +use crowdb_access_iceberg::{ + catalog::{CatalogRepository, ClearBounds, ManagementPrivilege}, + key::OperationId, + operation::{ManagementAction, ManagementRequest, RequestIdentity}, +}; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires pinned Apache Iceberg 1.11.0 source, Gradle and native storage"] +async fn apache_rest_compatibility_kit_supported_catalog_surface() { + let source = + std::env::var("CROWDB_ICEBERG_RCK_ROOT").expect("set the pinned Apache Iceberg 1.11.0 source root"); + let revision = std::process::Command::new("git") + .args(["-C", &source, "rev-parse", "HEAD"]) + .output() + .unwrap(); + assert!(revision.status.success()); + assert_eq!( + String::from_utf8(revision.stdout).unwrap().trim(), + "6976e020b894f6a6777704df2b8c4458cb291ae9" + ); + let stack = TestIcebergStack::start().await; + let repository = CatalogRepository::new( + stack.store().await, + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + ) + .unwrap(); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "rest-kit".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + now_ms(), + ) + .await + .unwrap(); + common::activate(&repository).await; + let process = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let origin = format!("http://{}", process.address); + let selectors = std::env::var("CROWDB_ICEBERG_RCK_SELECTOR").unwrap_or_else(|_| { + [ + "testCreateNamespace", + "testBasicCreateTable", + "testRenameTable", + "testDropTable", + "testDropMissingTable", + "testListTables", + ] + .iter() + .map(|name| format!("org.apache.iceberg.rest.RESTCompatibilityKitCatalogTests.{name}")) + .collect::>() + .join(",") + }); + let result = tokio::task::spawn_blocking(move || { + let mut command = std::process::Command::new("timeout"); + command.arg("900").arg("./gradlew").arg(":iceberg-open-api:test"); + for selector in selectors.split(',') { + command.arg("--tests").arg(selector); + } + command + .args([ + "--no-daemon", + "-Drck.local=false", + "-Drck.requires-namespace-create=true", + ]) + .env("CATALOG_URI", origin) + .env("CATALOG_WAREHOUSE", "") + .env("CATALOG_IO__IMPL", "org.apache.iceberg.aws.s3.S3FileIO") + .env("CATALOG_TOKEN", "w".repeat(32)) + .env( + "JAVA_HOME", + concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../.pixi/envs/iceberg-e2e/lib/jvm" + ), + ) + .current_dir(source) + .status() + .unwrap() + }); + let status = tokio::time::timeout(Duration::from_secs(930), result) + .await + .unwrap() + .unwrap(); + assert!( + status.success(), + "Apache Iceberg 1.11.0 REST Compatibility Kit selected catalog tests failed" + ); +} diff --git a/app/crowdb-access-server/tests/iceberg_route_test.rs b/app/crowdb-access-server/tests/iceberg_route_test.rs new file mode 100644 index 000000000..fd1d17247 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_route_test.rs @@ -0,0 +1,300 @@ +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; + +use std::{sync::Arc, time::Duration}; + +use crowdb_access_iceberg::{ + catalog::{CatalogRepository, ClearBounds, ManagementPrivilege}, + key::OperationId, + operation::{ManagementAction, ManagementRequest, RequestIdentity}, + wire::BearerAuthenticator, +}; +use crowdb_access_server::iceberg::{serve, IcebergHttpService}; +use reqwest::{ + header::{HeaderMap, HeaderValue, AUTHORIZATION}, + Client, Method, StatusCode, +}; +use tokio::net::TcpListener; + +async fn start( + namespaces: bool, + tables: bool, + credentials: bool, +) -> ( + Arc, + Arc, + String, + tokio::sync::oneshot::Sender<()>, + tokio::task::JoinHandle<()>, +) { + start_with_capabilities(namespaces, tables, credentials, Some(0x3fff)).await +} + +async fn start_with_capabilities( + namespaces: bool, + tables: bool, + credentials: bool, + capability_bits: Option, +) -> ( + Arc, + Arc, + String, + tokio::sync::oneshot::Sender<()>, + tokio::task::JoinHandle<()>, +) { + let store = Arc::new(common::TestStore::default()); + let repository = Arc::new(CatalogRepository::new(store.clone(), ClearBounds::default()).unwrap()); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: 100, + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "catalog".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + 100, + ) + .await + .unwrap(); + if let Some(bits) = capability_bits { + common::activate_bits(&repository, bits).await; + } + let authentication = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let mut service = IcebergHttpService::new(repository, authentication, Duration::from_secs(2)); + if namespaces { + service = service.with_namespaces(store.clone()).unwrap(); + } + if tables { + service = service + .with_tables(store.clone(), Arc::new(blocks::TestFileBlocks::default())) + .unwrap(); + } + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let origin = format!("http://{}", listener.local_addr().unwrap()); + if credentials { + service = service + .with_table_credentials(store.clone(), origin.clone()) + .unwrap(); + } + let (stop, stopped) = tokio::sync::oneshot::channel(); + let service = Arc::new(service); + let observed = service.clone(); + let server = tokio::spawn(async move { + serve(listener, service, async { + let _ = stopped.await; + }) + .await + .unwrap(); + }); + (store, observed, origin, stop, server) +} + +async fn send(client: &Client, origin: &str, method: Method, path: &str, token: &str) -> reqwest::Response { + client + .request(method, format!("{origin}{path}")) + .bearer_auth(token.repeat(32)) + .send() + .await + .unwrap() +} + +async fn reject_unsupported(client: &Client, origin: &str, store: &common::TestStore) { + let authority = store.values.load_full(); + for (method, path) in [ + (Method::POST, "/v1/namespaces/analytics/tables/events/plan"), + (Method::POST, "/v1/namespaces/analytics/tables/events/metrics"), + (Method::POST, "/v1/namespaces/analytics/register"), + (Method::POST, "/v1/transactions/commit"), + (Method::POST, "/v1/oauth/tokens"), + (Method::DELETE, "/v1/namespaces/analytics/tables"), + (Method::POST, "/v1/namespaces/analytics/tables/events/credentials"), + ] { + let response = send(client, origin, method, path, "w").await; + assert_eq!(response.status(), StatusCode::NOT_ACCEPTABLE, "{path}"); + assert_eq!( + response.json::().await.unwrap()["error"]["type"], + "UnsupportedOperationException" + ); + assert_eq!(*store.values.load_full(), *authority, "{path}"); + } +} + +async fn check_admin_metrics(client: &Client, origin: &str) { + for role in ["r", "w", "c"] { + assert_eq!( + send(client, origin, Method::GET, "/_crowdb/metrics", role) + .await + .status(), + StatusCode::FORBIDDEN + ); + } + let diagnostic = send(client, origin, Method::GET, "/_crowdb/metrics", "m").await; + assert_eq!(diagnostic.status(), StatusCode::OK); + let diagnostic: serde_json::Value = diagnostic.json().await.unwrap(); + assert_eq!(diagnostic["routes"].as_array().unwrap().len(), 9); +} + +#[tokio::test] +async fn explicit_partial_activation_limits_discovery_and_table_admission() { + let client = Client::new(); + let (_, _, origin, stop, server) = start_with_capabilities(true, true, true, None).await; + assert_eq!( + send(&client, &origin, Method::GET, "/v1/config", "r") + .await + .status(), + 503 + ); + assert_eq!( + send(&client, &origin, Method::GET, "/v1/namespaces/a/tables/t", "r") + .await + .status(), + StatusCode::NOT_ACCEPTABLE + ); + stop.send(()).unwrap(); + server.await.unwrap(); + + let (_, _, origin, stop, server) = start_with_capabilities(true, true, true, Some(0x0033)).await; + let config = send(&client, &origin, Method::GET, "/v1/config", "r") + .await + .json::() + .await + .unwrap(); + assert_eq!(config["overrides"]["crowdb.iceberg.v1.read"], "true"); + assert_eq!(config["overrides"]["crowdb.iceberg.v2.create"], "false"); + assert_eq!(config["endpoints"].as_array().unwrap().len(), 10); + assert!(!config["endpoints"].as_array().unwrap().iter().any(|endpoint| { + endpoint + .as_str() + .unwrap() + .starts_with("POST /v1/{prefix}/namespaces/{namespace}/tables") + })); + assert_eq!( + send(&client, &origin, Method::POST, "/v1/namespaces/a/tables", "w") + .await + .status(), + StatusCode::NOT_ACCEPTABLE + ); + assert_eq!( + send(&client, &origin, Method::GET, "/v1/namespaces/a/tables/t", "r") + .await + .status(), + StatusCode::NOT_FOUND + ); + stop.send(()).unwrap(); + server.await.unwrap(); +} + +#[tokio::test] +async fn discovery_uses_installed_routes_and_unsupported_paths_leave_no_record() { + let client = Client::builder().timeout(Duration::from_secs(3)).build().unwrap(); + for (namespaces, tables, credentials, expected) in [ + (false, false, false, 0), + (false, true, false, 0), + (true, false, false, 6), + (true, true, true, 14), + ] { + let (store, service, origin, stop, server) = start(namespaces, tables, credentials).await; + let config = send(&client, &origin, Method::GET, "/v1/config", "r") + .await + .json::() + .await + .unwrap(); + let endpoints = config["endpoints"].as_array().unwrap(); + assert_eq!(endpoints.len(), expected); + if !namespaces { + assert!(config.get("idempotency-key-lifetime").is_none()); + } + for endpoint in endpoints { + let template = endpoint.as_str().unwrap(); + assert!(!template.contains("/_crowdb/")); + assert!(!template.contains("/plan")); + assert!(!template.contains("/metrics")); + assert!(!template.contains("/register")); + assert!(!template.contains("/oauth")); + } + reject_unsupported(&client, &origin, &store).await; + store + .read_delay_ms + .store(5_000, std::sync::atomic::Ordering::SeqCst); + check_admin_metrics(&client, &origin).await; + store.read_delay_ms.store(0, std::sync::atomic::Ordering::SeqCst); + let unauthenticated = client + .post(format!("{origin}/v1/namespaces/analytics/register")) + .send() + .await + .unwrap(); + assert_eq!(unauthenticated.status(), StatusCode::UNAUTHORIZED); + let mut duplicate = HeaderMap::new(); + duplicate.append( + AUTHORIZATION, + HeaderValue::from_str(&format!("Bearer {}", "r".repeat(32))).unwrap(), + ); + duplicate.append( + AUTHORIZATION, + HeaderValue::from_str(&format!("Bearer {}", "w".repeat(32))).unwrap(), + ); + let duplicate = client + .get(format!("{origin}/v1/config")) + .headers(duplicate) + .send() + .await + .unwrap(); + assert_eq!(duplicate.status(), StatusCode::UNAUTHORIZED); + let table_path = "/v1/namespaces/analytics/tables/events"; + let table_read = send(&client, &origin, Method::GET, table_path, "r").await; + assert_eq!( + table_read.status(), + if namespaces && tables { + StatusCode::NOT_FOUND + } else { + StatusCode::NOT_ACCEPTABLE + } + ); + let _ = table_read.bytes().await.unwrap(); + let mut admitted_bytes = None; + if namespaces { + let body = br#"{"namespace":["analytics"]}"#; + let created = client + .post(format!("{origin}/v1/namespaces")) + .bearer_auth("w".repeat(32)) + .body(body.as_slice()) + .send() + .await + .unwrap(); + assert_eq!(created.status(), StatusCode::OK); + let _ = created.bytes().await.unwrap(); + let head = send(&client, &origin, Method::HEAD, "/v1/namespaces/analytics", "r").await; + assert_eq!(head.status(), StatusCode::NO_CONTENT); + assert!(head.bytes().await.unwrap().is_empty()); + admitted_bytes = Some(body.len() as u64); + } + stop.send(()).unwrap(); + server.await.unwrap(); + let snapshot = service.metrics_snapshot(); + if let Some(bytes) = admitted_bytes { + assert_eq!(snapshot.routes[2][0].requests, 1); + assert_eq!(snapshot.routes[2][0].request_bytes, bytes); + assert!(snapshot.routes[2][0].response_bytes > 0); + assert_eq!(snapshot.retry_new, 1); + assert_eq!(snapshot.routes[1][0].requests, 1); + assert_eq!(snapshot.routes[1][0].response_bytes, 0); + } + assert_eq!(snapshot.routes[0][0].requests, 1); + assert!(snapshot.routes[0][0].response_bytes > 0); + assert_eq!(snapshot.routes[0][1].requests, 1); + assert_eq!(snapshot.routes[8][3].requests, 7); + assert_eq!(snapshot.routes[7][0].requests, 1); + assert_eq!(snapshot.routes[7][1].requests, 3); + } +} diff --git a/app/crowdb-access-server/tests/iceberg_rust_retired_sdk_test.rs b/app/crowdb-access-server/tests/iceberg_rust_retired_sdk_test.rs new file mode 100644 index 000000000..4bea404e4 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_rust_retired_sdk_test.rs @@ -0,0 +1,174 @@ +#![cfg(feature = "iceberg-e2e")] + +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; +#[path = "common/iceberg_stack.rs"] +#[allow(dead_code)] +mod native_stack; +#[path = "common/iceberg_process.rs"] +#[allow(dead_code)] +mod process; + +use std::{sync::Arc, time::Duration}; + +use crowdb_access_iceberg::{ + catalog::{CatalogError, CatalogRepository, ClearBounds, ManagementPrivilege}, + key::OperationId, + operation::{ManagementAction, ManagementRequest, RequestIdentity}, + wire::BearerAuthenticator, +}; +use crowdb_access_server::iceberg::{serve, IcebergHttpService}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "builds the pinned official Apache Iceberg Rust client"] +async fn official_rust_client_rejects_retired_catalog_after_clear() { + let fixture = fixture::TestTableHttp::writable().await; + let repository = Arc::new(CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap()); + let service = IcebergHttpService::new( + repository.clone(), + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(), + Duration::from_secs(2), + ) + .with_namespaces(fixture.store.clone()) + .unwrap() + .with_tables(fixture.store.clone(), Arc::new(blocks::TestFileBlocks::default())) + .unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let second_origin = format!("http://{}", listener.local_addr().unwrap()); + let (stop, stopped) = tokio::sync::oneshot::channel(); + let server = tokio::spawn(async move { + serve(listener, Arc::new(service), async { + let _ = stopped.await; + }) + .await + .unwrap(); + }); + run_client_across_clear(&fixture.endpoint(), &second_origin, &repository, 60).await; + stop.send(()).unwrap(); + server.await.unwrap(); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires pinned Apache Iceberg Rust client and native retirement grace"] +async fn official_rust_client_rejects_retired_native_catalog_after_full_grace() { + let stack = native_stack::TestIcebergStack::start().await; + let repository = CatalogRepository::new( + stack.store().await, + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + ) + .unwrap(); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "rust-retired-native".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + now_ms(), + ) + .await + .unwrap(); + native_stack::activate(&repository).await; + let first = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let second = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + run_client_across_clear( + &format!("http://{}", first.address), + &format!("http://{}", second.address), + &repository, + 1_500, + ) + .await; +} + +async fn run_client_across_clear( + origin: &str, + second_origin: &str, + repository: &CatalogRepository, + timeout_seconds: u64, +) { + let control = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let control_address = control.local_addr().unwrap().to_string(); + let origin = origin.to_owned(); + let second_origin = second_origin.to_owned(); + let client = tokio::task::spawn_blocking(move || { + std::process::Command::new("timeout") + .arg(timeout_seconds.to_string()) + .arg("pixi") + .args(["run", "--"]) + .arg(std::env::var_os("CROWDB_ICEBERG_RUST_CLIENT_BIN").expect("build Rust SDK fixture first")) + .env("CROWDB_ICEBERG_RUST_ORIGIN", origin) + .env("CROWDB_ICEBERG_RUST_SECOND_ORIGIN", second_origin) + .env("CROWDB_ICEBERG_RUST_TOKEN", "w".repeat(32)) + .env("CROWDB_ICEBERG_RUST_NAMESPACE", "rust_sdk_retired") + .env("CROWDB_ICEBERG_RUST_RETIRE_CONTROL", control_address) + .status() + .unwrap() + }); + let (mut signal, _) = tokio::time::timeout(Duration::from_secs(20), control.accept()) + .await + .unwrap() + .unwrap(); + signal.read_exact(&mut [0]).await.unwrap(); + let (root, authority) = repository.status().await.unwrap(); + let clear = ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "manager".into(), + action: ManagementAction::Clear, + expected_epoch: root.context.activation_epoch, + display_name: authority.display_name, + confirmation: Some(root.context.catalog), + capabilities: None, + }; + assert!(matches!( + repository + .execute(clear.clone(), ManagementPrivilege::Clear, now_ms()) + .await, + Err(CatalogError::Busy) + )); + let (root, _) = repository.status().await.unwrap(); + let crowdb_access_iceberg::catalog::RootState::Published(transition) = root.state else { + panic!("clear did not publish its durable grace boundary"); + }; + let remaining = transition.complete_after_ms.saturating_sub(now_ms()); + assert!(remaining < (timeout_seconds - 10) * 1_000); + tokio::time::sleep(Duration::from_millis(remaining + 10)).await; + repository + .execute(clear, ManagementPrivilege::Clear, now_ms()) + .await + .unwrap(); + common::activate(repository).await; + signal.write_all(&[1]).await.unwrap(); + assert!(client.await.unwrap().success()); +} + +fn now_ms() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis() + .try_into() + .unwrap() +} diff --git a/app/crowdb-access-server/tests/iceberg_rust_sdk_test.rs b/app/crowdb-access-server/tests/iceberg_rust_sdk_test.rs new file mode 100644 index 000000000..48e2bfea0 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_rust_sdk_test.rs @@ -0,0 +1,206 @@ +#![cfg(feature = "iceberg-e2e")] + +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; +#[path = "common/iceberg_stack.rs"] +#[allow(dead_code)] +mod native_stack; +#[path = "common/iceberg_process.rs"] +#[allow(dead_code)] +mod process; +#[path = "common/iceberg_response_loss.rs"] +mod response_loss; + +use crowdb_access_iceberg::{ + catalog::{CatalogRepository, ClearBounds, ManagementPrivilege}, + key::OperationId, + operation::{ManagementAction, ManagementRequest, RequestIdentity}, + wire::BearerAuthenticator, +}; +use crowdb_access_server::iceberg::{serve, IcebergHttpService}; +use fixture::TestTableHttp; +use response_loss::TestResponseLossProxy; +use std::{sync::Arc, time::Duration}; + +#[tokio::test] +#[ignore = "builds the pinned official Apache Iceberg Rust client"] +async fn official_rust_client_namespace_and_table_lifecycle() { + run_official_client(false).await; +} + +#[tokio::test] +#[ignore = "builds the pinned official Apache Iceberg Rust client"] +async fn official_rust_client_observes_lost_create_reply_on_another_listener() { + run_official_client(true).await; +} + +async fn run_official_client(response_loss: bool) { + let fixture = TestTableHttp::writable().await; + let backend_origin = fixture.endpoint(); + let service = IcebergHttpService::new( + Arc::new(CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap()), + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(), + Duration::from_secs(2), + ) + .with_namespaces(fixture.store.clone()) + .unwrap() + .with_tables(fixture.store.clone(), Arc::new(blocks::TestFileBlocks::default())) + .unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let second_origin = format!("http://{}", listener.local_addr().unwrap()); + let (stop, stopped) = tokio::sync::oneshot::channel(); + let server = tokio::spawn(async move { + serve(listener, Arc::new(service), async { + let _ = stopped.await; + }) + .await + .unwrap(); + }); + let (origin, proxy) = if response_loss { + let proxy = TestResponseLossProxy::start(backend_origin, "/v1/namespaces/rust_sdk_loss/tables").await; + (proxy.origin.clone(), Some(proxy)) + } else { + (backend_origin, None) + }; + let status = tokio::task::spawn_blocking(move || { + let mut command = std::process::Command::new("timeout"); + command + .arg("600") + .arg("pixi") + .args(["run", "--"]) + .arg(std::env::var_os("CROWDB_ICEBERG_RUST_CLIENT_BIN").expect("build Rust SDK fixture first")) + .env("CROWDB_ICEBERG_RUST_ORIGIN", origin) + .env("CROWDB_ICEBERG_RUST_SECOND_ORIGIN", second_origin) + .env("CROWDB_ICEBERG_RUST_TOKEN", "w".repeat(32)) + .env( + "CROWDB_ICEBERG_RUST_NAMESPACE", + if response_loss { + "rust_sdk_loss" + } else { + "rust_sdk" + }, + ); + if response_loss { + command.env("CROWDB_ICEBERG_RUST_RESPONSE_LOSS", "1"); + } + command.status().unwrap() + }) + .await + .unwrap(); + if let Some(proxy) = proxy { + proxy.assert_dropped(); + } + stop.send(()).unwrap(); + server.await.unwrap(); + assert!(status.success(), "official Rust REST client failed"); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires pinned Apache Iceberg Rust client and native storage"] +async fn official_rust_client_lost_reply_survives_native_storage_restart() { + let mut stack = native_stack::TestIcebergStack::start().await; + let repository = CatalogRepository::new( + stack.store().await, + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + ) + .unwrap(); + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: native_stack::now_ms(), + }, + principal: "manager".into(), + action: ManagementAction::Initialize, + expected_epoch: 0, + display_name: "rust-native".into(), + confirmation: None, + capabilities: None, + }, + ManagementPrivilege::Manage, + native_stack::now_ms(), + ) + .await + .unwrap(); + native_stack::activate(&repository).await; + let first = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let second = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let proxy = TestResponseLossProxy::start( + format!("http://{}", first.address), + "/v1/namespaces/rust_sdk_loss/tables", + ) + .await; + assert!( + run_rust_fixture( + &proxy.origin, + &format!("http://{}", second.address), + true, + false, + true + ) + .await + ); + proxy.assert_dropped(); + drop(first); + drop(second); + stack.chunk_kv.restart().await; + let first = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let second = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + assert!( + run_rust_fixture( + &format!("http://{}", first.address), + &format!("http://{}", second.address), + false, + true, + false + ) + .await + ); +} + +async fn run_rust_fixture( + origin: &str, + second_origin: &str, + response_loss: bool, + verify_existing: bool, + keep_table: bool, +) -> bool { + let origin = origin.to_owned(); + let second_origin = second_origin.to_owned(); + tokio::task::spawn_blocking(move || { + let mut command = std::process::Command::new("timeout"); + command + .arg("600") + .arg("pixi") + .args(["run", "--"]) + .arg(std::env::var_os("CROWDB_ICEBERG_RUST_CLIENT_BIN").expect("build Rust SDK fixture first")) + .env("CROWDB_ICEBERG_RUST_ORIGIN", origin) + .env("CROWDB_ICEBERG_RUST_SECOND_ORIGIN", second_origin) + .env("CROWDB_ICEBERG_RUST_TOKEN", "w".repeat(32)) + .env("CROWDB_ICEBERG_RUST_NAMESPACE", "rust_sdk_loss"); + if response_loss { + command.env("CROWDB_ICEBERG_RUST_RESPONSE_LOSS", "1"); + } + if verify_existing { + command.env("CROWDB_ICEBERG_RUST_VERIFY_EXISTING", "1"); + } + if keep_table { + command.env("CROWDB_ICEBERG_RUST_KEEP_TABLE", "1"); + } + command.status().unwrap().success() + }) + .await + .unwrap() +} diff --git a/app/crowdb-access-server/tests/iceberg_table_acceptance_test.rs b/app/crowdb-access-server/tests/iceberg_table_acceptance_test.rs new file mode 100644 index 000000000..919215108 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_table_acceptance_test.rs @@ -0,0 +1,258 @@ +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +use std::{collections::BTreeMap, sync::atomic::Ordering, time::Duration}; + +use crowdb_access_iceberg::{ + catalog::StoredValue, + key::{CatalogScope, IcebergKey, OperationId, TableId}, + record::StorageRecord, + table::{head_key, name_key, TableHead, TableLifecycle, TableMapping, TableMappingState}, +}; +use fixture::TestTableHttp; +use reqwest::Method; +use serde_json::{json, Value}; + +const TABLES: &str = "/v1/namespaces/analytics/tables"; +const TABLE: &str = "/v1/namespaces/analytics/tables/events"; + +async fn value(response: reqwest::Response, status: u16) -> Value { + let actual = response.status().as_u16(); + let body = response.text().await.unwrap(); + assert_eq!(actual, status, "{body}"); + serde_json::from_str(&body).unwrap() +} + +async fn create(test: &TestTableHttp) -> Value { + value( + test.post( + TABLES, + "w", + None, + &json!({"name":"events", "schema":{ + "type":"struct","schema-id":0,"fields":[{"id":1,"name":"id","type":"long","required":true}]}}), + ) + .await, + 200, + ) + .await +} + +#[tokio::test] +async fn concurrent_commit_cannot_mix_all_refs_or_conditional_http_loads() { + let test = TestTableHttp::writable().await; + create(&test).await; + for mode in ["all", "refs"] { + for conditional in [false, true] { + let path = format!("{TABLE}?snapshots={mode}"); + let before = test.request(Method::GET, &path, "r", None).await; + assert_eq!(before.status(), 200); + let etag = before.headers()["etag"].to_str().unwrap().to_owned(); + before.bytes().await.unwrap(); + test.store.pause_file_read.store(true, Ordering::SeqCst); + let read = test.request(Method::GET, &path, "r", conditional.then_some(etag.as_str())); + tokio::pin!(read); + tokio::select! { + response = &mut read => panic!("read completed before barrier: {}", response.status()), + () = test.store.file_read_entered.notified() => {}, + () = tokio::time::sleep(Duration::from_secs(1)) => panic!("file read did not reach barrier"), + } + let marker = format!("{mode}-{conditional}"); + value( + test.post( + TABLE, + "w", + None, + &json!({"requirements":[], "updates":[ + {"action":"set-properties","updates":{"marker":marker}} + ]}), + ) + .await, + 200, + ) + .await; + test.store.file_read_release.notify_one(); + let failed = value(read.await, 503).await; + assert!(failed.get("metadata").is_none()); + let after = test.request(Method::GET, &path, "r", Some(&etag)).await; + assert_ne!(after.headers()["etag"], etag); + assert_eq!( + value(after, 200).await["metadata"]["properties"]["marker"], + marker + ); + } + } + test.finish().await; +} + +fn mapping(test: &TestTableHttp, head: &TableHead, name: &str, state: TableMappingState) { + test.put( + &name_key(head.catalog, head.namespace, name).unwrap(), + &StorageRecord::TableMapping(TableMapping { + catalog: head.catalog, + namespace: head.namespace, + name: name.into(), + table: head.table, + name_epoch: head.name_epoch, + operation: OperationId::random(), + state, + }), + ); +} + +#[tokio::test] +async fn mixed_mapping_pages_and_exists_expose_only_current_table_heads() { + let test = TestTableHttp::new().await; + let (head, _) = test.install("events").await; + mapping(&test, &head, "alias", TableMappingState::Published); + mapping(&test, &head, "reserved", TableMappingState::Reserved); + let mut absent = head.clone(); + absent.table = TableId::random(); + mapping(&test, &absent, "missing", TableMappingState::Published); + let (mut dropped, _) = test.install("dropped").await; + dropped.lifecycle = TableLifecycle::Tombstone; + dropped.pending_operation = Some(OperationId::random()); + test.put( + &head_key(dropped.catalog, dropped.table), + &StorageRecord::TableHead(Box::new(dropped)), + ); + let mut token = String::new(); + let mut names = Vec::new(); + let mut pages = 0; + let scans = test.store.scans.load(Ordering::SeqCst); + loop { + let path = format!("{TABLES}?pageSize=1&pageToken={token}"); + let page = value(test.request(Method::GET, &path, "r", None).await, 200).await; + names.extend(page["identifiers"].as_array().unwrap().iter().cloned()); + pages += 1; + assert!(pages <= 5); + match page["next-page-token"].as_str() { + Some(next) => token = next.into(), + None => break, + } + } + assert_eq!(pages, 5); + assert_eq!(test.store.scans.load(Ordering::SeqCst) - scans, 5); + assert_eq!(names, vec![json!({"namespace":["analytics"],"name":"events"})]); + let complete = value(test.request(Method::GET, TABLES, "r", None).await, 200).await; + assert_eq!(complete["identifiers"], json!(names)); + assert!(complete["next-page-token"].is_null()); + for name in ["alias", "reserved", "missing", "dropped", "events"] { + let response = test + .request(Method::HEAD, &format!("{TABLES}/{name}"), "r", None) + .await; + assert_eq!( + response.status().as_u16(), + if name == "events" { 204 } else { 404 } + ); + assert!(response.bytes().await.unwrap().is_empty()); + } + test.finish().await; +} + +fn authority(test: &TestTableHttp) -> BTreeMap, StoredValue> { + test.store + .values + .load() + .iter() + .filter(|(key, _)| { + matches!( + IcebergKey::decode(key), + Ok(IcebergKey::Catalog { + scope: CatalogScope::TableHead + | CatalogScope::TableName + | CatalogScope::File + | CatalogScope::FileLocation + | CatalogScope::Reclamation, + .. + }) + ) + }) + .map(|(key, value)| (key.clone(), value.clone())) + .collect() +} + +#[tokio::test] +async fn unsupported_table_operations_do_not_change_any_table_or_file_authority() { + let test = TestTableHttp::writable().await; + create(&test).await; + let before = authority(&test); + for path in [ + "/v1/namespaces/analytics/register", + "/v1/namespaces/analytics/tables/events/unknown", + ] { + let error = value(test.post(path, "w", None, &json!({})).await, 406).await; + assert_eq!(error["error"]["type"], "UnsupportedOperationException"); + assert_eq!(authority(&test), before); + } + test.finish().await; +} + +#[tokio::test] +async fn lost_drop_publication_reply_replays_once_and_preserves_files_for_both_purge_modes() { + for purge in [false, true] { + let test = TestTableHttp::writable().await; + create(&test).await; + let before = authority(&test); + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(); + let identity = format!("{:08x}-{:04x}-7000-8000-000000000001", now >> 16, now & 0xffff); + let path = format!("{}{TABLE}?purgeRequested={purge}", test.endpoint()); + let client = reqwest::Client::new(); + test.store.lose_reply_kind.store(5, Ordering::SeqCst); + let failed = client + .delete(&path) + .bearer_auth("w".repeat(32)) + .header("idempotency-key", &identity) + .send() + .await + .unwrap(); + value(failed, 503).await; + assert_eq!(test.request(Method::HEAD, TABLE, "r", None).await.status(), 404); + for _ in 0..2 { + let response = client + .delete(&path) + .bearer_auth("w".repeat(32)) + .header("idempotency-key", &identity) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 204); + assert!(response.bytes().await.unwrap().is_empty()); + } + let after = authority(&test); + for (key, value) in before { + if matches!( + IcebergKey::decode(&key).unwrap(), + IcebergKey::Catalog { + scope: CatalogScope::File | CatalogScope::FileLocation, + .. + } + ) { + assert_eq!(after.get(&key), Some(&value)); + } + } + let tasks = after + .keys() + .filter(|key| { + matches!( + IcebergKey::decode(key).unwrap(), + IcebergKey::Catalog { + scope: CatalogScope::Reclamation, + .. + } + ) + }) + .count(); + assert_eq!(tasks, usize::from(purge)); + test.finish().await; + } +} diff --git a/app/crowdb-access-server/tests/iceberg_table_admission_test.rs b/app/crowdb-access-server/tests/iceberg_table_admission_test.rs new file mode 100644 index 000000000..20fce5d0f --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_table_admission_test.rs @@ -0,0 +1,147 @@ +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +use crowdb_access_iceberg::key::{CatalogScope, IcebergKey}; +use fixture::TestTableHttp; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +const TABLE: &str = "/v1/namespaces/analytics/tables/events"; + +async fn fixture() -> TestTableHttp { + let fixture = TestTableHttp::writable().await; + let response = fixture + .post( + "/v1/namespaces/analytics/tables", + "w", + None, + &serde_json::json!({"name":"events","schema":{"type":"struct","schema-id":0,"fields":[ + {"id":1,"name":"id","type":"long","required":true}]}}), + ) + .await; + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + fixture +} + +fn mutations(fixture: &TestTableHttp) -> Vec<(Vec, Vec)> { + fixture + .store + .values + .load() + .iter() + .filter(|(key, _)| { + matches!( + IcebergKey::decode(key), + Ok(IcebergKey::Catalog { + scope: CatalogScope::TableHead + | CatalogScope::TableCommitOperation + | CatalogScope::File + | CatalogScope::FileLocation, + .. + }) + ) + }) + .map(|(key, value)| (key.clone(), value.bytes.clone())) + .collect() +} + +#[tokio::test] +async fn commit_request_byte_limits_reject_before_candidate_or_operation_creation() { + let fixture = fixture().await; + let mut body = r#"{"requirements":[],"updates":[]}"#.as_bytes().to_vec(); + body.resize(2 * 1024 * 1024 - 64 * 1024, b' '); + let client = reqwest::Client::new(); + let response = client + .post(format!("{}{TABLE}", fixture.endpoint())) + .bearer_auth("w".repeat(32)) + .header("content-type", "application/json") + .body(body.clone()) + .send() + .await + .unwrap(); + let status = response.status(); + let response = response.text().await.unwrap(); + assert_eq!(status, 200, "{response}"); + let before = mutations(&fixture); + for length in [body.len() + 1, 2 * 1024 * 1024 + 1] { + body.resize(length, b' '); + let response = client + .post(format!("{}{TABLE}", fixture.endpoint())) + .bearer_auth("w".repeat(32)) + .header("content-type", "application/json") + .body(body.clone()) + .send() + .await + .unwrap(); + let status = response.status(); + let response = response.text().await.unwrap(); + assert_eq!(status, 400, "{response}"); + assert_eq!(mutations(&fixture), before); + } + let response = fixture + .post( + TABLE, + "w", + None, + &serde_json::json!({"requirements":[],"updates":[]}), + ) + .await; + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + fixture.finish().await; +} + +async fn pending_body(endpoint: &str) -> tokio::net::TcpStream { + let address = endpoint.strip_prefix("http://").unwrap(); + let mut stream = tokio::net::TcpStream::connect(address).await.unwrap(); + stream.write_all(format!("POST {TABLE} HTTP/1.1\r\nHost: {address}\r\nAuthorization: Bearer {}\r\nContent-Length: 1\r\nExpect: 100-continue\r\nConnection: close\r\n\r\n", "w".repeat(32)).as_bytes()).await.unwrap(); + let mut header = [0; 25]; + stream.read_exact(&mut header).await.unwrap(); + assert_eq!(&header, b"HTTP/1.1 100 Continue\r\n\r\n"); + stream +} + +#[tokio::test] +async fn pending_commit_bodies_share_admission_and_errors_release_every_slot() { + let fixture = fixture().await; + let before = mutations(&fixture); + let mut streams = Vec::new(); + for _ in 0..4 { + streams.push(pending_body(&fixture.endpoint()).await); + } + let response = fixture + .post( + TABLE, + "w", + None, + &serde_json::json!({"requirements":[],"updates":[]}), + ) + .await; + assert_eq!(response.status(), 503, "{}", response.text().await.unwrap()); + assert_eq!(mutations(&fixture), before); + for mut stream in streams { + stream.write_all(b"x").await.unwrap(); + let mut response = Vec::new(); + stream.read_to_end(&mut response).await.unwrap(); + assert!( + response.starts_with(b"HTTP/1.1 400"), + "{}", + String::from_utf8_lossy(&response) + ); + } + assert_eq!(mutations(&fixture), before); + let response = fixture + .post( + TABLE, + "w", + None, + &serde_json::json!({"requirements":[],"updates":[]}), + ) + .await; + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + fixture.finish().await; +} diff --git a/app/crowdb-access-server/tests/iceberg_table_credentials_test.rs b/app/crowdb-access-server/tests/iceberg_table_credentials_test.rs new file mode 100644 index 000000000..b0dfd6690 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_table_credentials_test.rs @@ -0,0 +1,272 @@ +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +use crowdb_access_iceberg::{ + catalog::{CatalogRepository, ClearBounds}, + file::{FileGrantIssuer, FileOperation, TableLocation}, + wire::BearerAuthenticator, +}; +use fixture::TestTableHttp; +use reqwest::Method; +use serde_json::{json, Value}; + +#[tokio::test] +async fn credential_refresh_requires_selected_version_read_capability() { + let fixture = TestTableHttp::vending_with_capabilities(0x0033).await; + fixture.install("events").await; + let path = "/v1/namespaces/analytics/tables/events/credentials"; + assert_eq!(fixture.request(Method::GET, path, "r", None).await.status(), 406); + let repository = CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap(); + common::activate_bits(&repository, 0x3fff).await; + assert_eq!(fixture.request(Method::GET, path, "r", None).await.status(), 200); + fixture.finish().await; +} + +#[tokio::test] +async fn read_only_format_profile_never_vends_file_mutation_permission() { + let fixture = TestTableHttp::vending_with_capabilities(0x0300).await; + fixture.install("events").await; + let path = "/v1/namespaces/analytics/tables/events/credentials"; + let response = fixture.request(Method::GET, path, "w", None).await; + assert_eq!(response.status(), 200); + let body: Value = response.json().await.unwrap(); + let config = &body["storage-credentials"][0]["config"]; + let authentication = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let issuer = FileGrantIssuer::new(authentication.namespace_token_key(), 900_000).unwrap(); + let now = u64::try_from( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap(); + let grant = issuer + .verify( + config["s3.access-key-id"].as_str().unwrap(), + config["s3.session-token"].as_str().unwrap(), + fixture.context, + now, + ) + .unwrap(); + assert!(grant.grant().operations.allows(FileOperation::Get)); + assert!(!grant.grant().operations.allows(FileOperation::Put)); + let repository = CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap(); + common::activate_bits(&repository, 0x3fff).await; + let response = fixture.request(Method::GET, path, "w", None).await; + assert_eq!(response.status(), 200); + let body: Value = response.json().await.unwrap(); + let config = &body["storage-credentials"][0]["config"]; + let now = u64::try_from( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap(); + let grant = issuer + .verify( + config["s3.access-key-id"].as_str().unwrap(), + config["s3.session-token"].as_str().unwrap(), + fixture.context, + now, + ) + .unwrap(); + assert!(grant.grant().operations.allows(FileOperation::Put)); + fixture.finish().await; +} + +async fn draft(fixture: &TestTableHttp) -> Value { + let response = fixture + .post( + "/v1/namespaces/analytics/tables", + "w", + None, + &json!({"name":"events","stage-create":true,"schema":{"type":"struct","fields":[]}}), + ) + .await; + let status = response.status(); + let bytes = response.bytes().await.unwrap(); + assert_eq!(status, 200, "{}", String::from_utf8_lossy(&bytes)); + serde_json::from_slice(&bytes).unwrap() +} + +#[tokio::test] +async fn same_name_drafts_refresh_only_the_exact_original_writer_scope() { + let fixture = TestTableHttp::vending().await; + let auth = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let issuer = FileGrantIssuer::new(auth.namespace_token_key(), 900_000).unwrap(); + let first = draft(&fixture).await; + let second = draft(&fixture).await; + assert_ne!(first["metadata"]["location"], second["metadata"]["location"]); + for draft in [&first, &second] { + let path = draft["config"]["client.refresh-credentials-endpoint"] + .as_str() + .unwrap(); + for role in ["r", "m", "c"] { + assert_eq!(fixture.request(Method::GET, path, role, None).await.status(), 404); + } + let response = fixture.request(Method::GET, path, "w", None).await; + assert_eq!(response.status(), 200); + let value: Value = serde_json::from_slice(&response.bytes().await.unwrap()).unwrap(); + assert_eq!(value["storage-credentials"].as_array().unwrap().len(), 1); + let credential = &value["storage-credentials"][0]; + assert_eq!( + credential["prefix"], + format!("{}/", draft["metadata"]["location"].as_str().unwrap()) + ); + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis() + .try_into() + .unwrap(); + let grant = issuer + .verify( + credential["config"]["s3.access-key-id"].as_str().unwrap(), + credential["config"]["s3.session-token"].as_str().unwrap(), + fixture.context, + now, + ) + .unwrap(); + let table: TableLocation = format!("{}/", draft["metadata"]["location"].as_str().unwrap()) + .parse() + .unwrap(); + assert_eq!(grant.grant().table, table.table); + assert_credential_pin(&fixture, grant.grant()).await; + assert!(grant.grant().operations.allows(FileOperation::Put)); + let wrong = path.replace("/events/", "/other/"); + assert_eq!( + fixture.request(Method::GET, &wrong, "w", None).await.status(), + 404 + ); + assert_eq!( + fixture + .request( + Method::GET, + &format!("{path}&table-id={}", table.table), + "w", + None + ) + .await + .status(), + 400 + ); + } + assert_eq!( + fixture + .request( + Method::GET, + "/v1/namespaces/analytics/tables/events/credentials", + "w", + None + ) + .await + .status(), + 404 + ); + expire_first(&fixture, &first, &second).await; + fixture.finish().await; +} + +async fn assert_credential_pin(fixture: &TestTableHttp, grant: &crowdb_access_iceberg::file::FileGrant) { + use crowdb_access_iceberg::{key::IcebergKey, record::StorageRecord}; + let repository = CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap(); + let (_, authority) = repository.status().await.unwrap(); + let expiry = + grant.expires_ms + authority.admission_bounds.request_ms + authority.admission_bounds.clock_skew_ms; + assert!(fixture.store.values.load().iter().any(|(key, value)| { + matches!(IcebergKey::decode(key).and_then(|key| StorageRecord::decode(&key, &value.bytes)), + Ok(StorageRecord::GcPin(pin)) if pin.head.table == grant.table && pin.expires_ms == expiry + && pin.protects_uploads && !pin.released) + })); +} + +async fn expire_first(fixture: &TestTableHttp, first: &Value, second: &Value) { + let first_table: TableLocation = format!("{}/", first["metadata"]["location"].as_str().unwrap()) + .parse() + .unwrap(); + let creator = crowdb_access_iceberg::commit::TableCreator::new( + fixture.store.clone(), + std::sync::Arc::new(blocks::TestFileBlocks::default()), + ); + assert!(creator + .expire_stage(fixture.context, first_table.table, i64::MAX) + .await + .unwrap()); + assert_eq!( + fixture + .request( + Method::GET, + first["config"]["client.refresh-credentials-endpoint"] + .as_str() + .unwrap(), + "w", + None + ) + .await + .status(), + 404 + ); + assert_eq!( + fixture + .request( + Method::GET, + second["config"]["client.refresh-credentials-endpoint"] + .as_str() + .unwrap(), + "w", + None + ) + .await + .status(), + 200 + ); +} + +#[tokio::test] +async fn published_table_credentials_preserve_read_only_roles() { + let fixture = TestTableHttp::vending().await; + let (head, _) = fixture.install("events").await; + let auth = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let issuer = FileGrantIssuer::new(auth.namespace_token_key(), 900_000).unwrap(); + for role in ["r", "w", "m", "c"] { + let response = fixture + .request( + Method::GET, + "/v1/namespaces/analytics/tables/events/credentials", + role, + None, + ) + .await; + assert_eq!(response.status(), 200); + let body: Value = serde_json::from_slice(&response.bytes().await.unwrap()).unwrap(); + let config = &body["storage-credentials"][0]["config"]; + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis() + .try_into() + .unwrap(); + let grant = issuer + .verify( + config["s3.access-key-id"].as_str().unwrap(), + config["s3.session-token"].as_str().unwrap(), + fixture.context, + now, + ) + .unwrap(); + assert_eq!(grant.grant().table, head.table); + assert_eq!(grant.grant().operations.allows(FileOperation::Put), role == "w"); + assert!(grant.grant().operations.allows(FileOperation::Get)); + } + fixture.finish().await; +} diff --git a/app/crowdb-access-server/tests/iceberg_table_http_test.rs b/app/crowdb-access-server/tests/iceberg_table_http_test.rs new file mode 100644 index 000000000..f1ce90034 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_table_http_test.rs @@ -0,0 +1,302 @@ +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +use crowdb_access_iceberg::catalog::{CatalogRepository, ClearBounds}; +use fixture::TestTableHttp; +use reqwest::Method; +use serde_json::Value; + +const PATH: &str = "/v1/namespaces/analytics/tables/events"; + +#[tokio::test] +async fn selected_version_requires_read_even_for_head_and_conditional_load() { + let fixture = TestTableHttp::with_capabilities(0x0033).await; + fixture.install("events").await; + for method in [Method::HEAD, Method::GET] { + let response = fixture.request(method, PATH, "r", Some("*")).await; + assert_eq!(response.status(), 406); + } + let repository = CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap(); + common::activate_bits(&repository, 0x3fff).await; + assert_eq!(fixture.request(Method::HEAD, PATH, "r", None).await.status(), 204); + assert_eq!(fixture.request(Method::GET, PATH, "r", None).await.status(), 200); + fixture.finish().await; +} + +#[tokio::test] +async fn optional_access_delegation_list_does_not_change_table_identity() { + let fixture = TestTableHttp::vending().await; + fixture.install("events").await; + let client = reqwest::Client::new(); + let origin = fixture.endpoint(); + let plain = client + .get(format!("{origin}{PATH}?snapshots=refs")) + .bearer_auth("r".repeat(32)) + .send() + .await + .unwrap(); + assert_eq!(plain.status(), 200); + let plain: Value = plain.json().await.unwrap(); + let delegated = client + .get(format!("{origin}{PATH}?snapshots=refs")) + .bearer_auth("r".repeat(32)) + .header("X-Iceberg-Access-Delegation", "vended-credentials,remote-signing") + .send() + .await + .unwrap(); + assert_eq!(delegated.status(), 200); + let delegated: Value = delegated.json().await.unwrap(); + assert_eq!(plain, delegated); + fixture.finish().await; +} + +#[tokio::test] +async fn successful_load_counts_selected_version_and_emitted_response() { + let fixture = TestTableHttp::new().await; + fixture.install("events").await; + let service = fixture.service.clone(); + let response = fixture.request(Method::GET, PATH, "r", None).await; + assert_eq!(response.status(), 200); + let length = response.bytes().await.unwrap().len() as u64; + fixture.finish().await; + let snapshot = service.metrics_snapshot(); + assert_eq!(snapshot.selected_versions, [0, 0, 1]); + assert_eq!(snapshot.routes[3][0].requests, 1); + assert_eq!(snapshot.routes[3][0].response_bytes, length); + assert!(snapshot.routes[3][0].dispatch_latency_ns > 0); + assert!(snapshot.routes[3][0].lifetime_ns >= snapshot.routes[3][0].dispatch_latency_ns); +} + +#[tokio::test] +async fn configured_load_etag_includes_sdk_configuration_and_preserves_conditionals() { + use sha2::{Digest, Sha256}; + let fixture = TestTableHttp::vending().await; + let (head, _) = fixture.install("events").await; + let mut digest = Sha256::new(); + digest.update(b"crowdb-iceberg-table-load-v1"); + digest.update(head.catalog.as_bytes()); + digest.update(head.table.as_bytes()); + digest.update(head.generation.to_be_bytes()); + digest.update(head.metadata_digest); + digest.update([0]); + let metadata_etag = format!("\"{:x}\"", digest.finalize()); + let loaded = fixture + .request(Method::GET, PATH, "r", Some(&metadata_etag)) + .await; + assert_eq!(loaded.status(), 200); + let etag = loaded.headers()["etag"].to_str().unwrap().to_owned(); + let bytes = loaded.bytes().await.unwrap(); + let mut digest = Sha256::new(); + digest.update(metadata_etag.as_bytes()); + digest.update(&bytes); + assert_eq!(etag, format!("\"{:x}\"", digest.finalize())); + assert_ne!(etag, metadata_etag); + let body: Value = serde_json::from_slice(&bytes).unwrap(); + assert!(body["config"]["s3.endpoint"].is_string()); + let unchanged = fixture + .request(Method::GET, PATH, "r", Some(&format!("W/{etag}"))) + .await; + assert_eq!(unchanged.status(), 304); + assert_eq!(unchanged.headers()["etag"], etag); + assert!(unchanged.bytes().await.unwrap().is_empty()); + fixture.finish().await; +} + +#[tokio::test] +async fn table_load_preserves_raw_metadata_and_mode_specific_conditional_responses() { + let fixture = TestTableHttp::new().await; + let (_, bytes) = fixture.install("events").await; + let loaded = fixture.request(Method::GET, PATH, "r", None).await; + assert_eq!(loaded.status(), 200); + let etag = loaded.headers()["etag"].to_str().unwrap().to_owned(); + let body = loaded.text().await.unwrap(); + assert!(body.contains(std::str::from_utf8(&bytes).unwrap())); + let value: Value = serde_json::from_str(&body).unwrap(); + assert_eq!(value["metadata"]["snapshots"].as_array().unwrap().len(), 3); + let loaded = fixture + .request(Method::GET, &format!("{PATH}?snapshots=refs"), "r", Some(&etag)) + .await; + assert_eq!(loaded.status(), 200); + let refs_etag = loaded.headers()["etag"].to_str().unwrap().to_owned(); + assert_ne!(refs_etag, etag); + let body = loaded.text().await.unwrap(); + assert!(body.contains("123456789012345678901234567890")); + let value: Value = serde_json::from_str(&body).unwrap(); + assert_eq!(value["metadata"]["snapshots"].as_array().unwrap().len(), 2); + let unchanged = fixture + .request(Method::GET, PATH, "r", Some(&format!("W/{etag}"))) + .await; + assert_eq!(unchanged.status(), 304); + assert_eq!(unchanged.headers()["etag"], etag); + assert!(unchanged.bytes().await.unwrap().is_empty()); + let exists = fixture.request(Method::HEAD, PATH, "r", None).await; + assert_eq!(exists.status(), 204); + assert!(exists.bytes().await.unwrap().is_empty()); + fixture.finish().await; +} + +#[tokio::test] +async fn table_list_has_complete_and_paged_modes_with_bound_tokens() { + let fixture = TestTableHttp::new().await; + fixture.install("a+b").await; + fixture.install("%2F").await; + for path in [ + "/v1/namespaces/analytics/tables/a+b", + "/v1/namespaces/analytics/tables/%252F", + ] { + assert_eq!(fixture.request(Method::HEAD, path, "r", None).await.status(), 204); + } + let path = "/v1/namespaces/analytics/tables?pageSize=1"; + let response = fixture.request(Method::GET, path, "r", None).await; + assert_eq!(response.status(), 200); + let complete: Value = serde_json::from_str(&response.text().await.unwrap()).unwrap(); + assert_eq!(complete["identifiers"].as_array().unwrap().len(), 2); + assert!(complete["next-page-token"].is_null()); + let response = fixture + .request(Method::GET, &format!("{path}&pageToken="), "r", None) + .await; + let page: Value = serde_json::from_str(&response.text().await.unwrap()).unwrap(); + assert_eq!(page["identifiers"].as_array().unwrap().len(), 1); + let token = page["next-page-token"].as_str().unwrap(); + assert_eq!( + fixture + .request(Method::GET, &format!("{path}&pageToken={token}"), "r", None) + .await + .status(), + 200 + ); + assert_eq!( + fixture + .request( + Method::GET, + &format!("/v1/namespaces/analytics/tables?pageSize=2&pageToken={token}"), + "r", + None + ) + .await + .status(), + 400 + ); + fixture.finish().await; +} + +#[tokio::test] +async fn read_routes_authenticate_reject_bad_parameters_and_advertise_only_test_reads() { + let fixture = TestTableHttp::new().await; + fixture.install("events").await; + for role in ["r", "w", "m", "c"] { + assert_eq!(fixture.request(Method::GET, PATH, role, None).await.status(), 200); + } + assert_eq!( + fixture.request(Method::GET, PATH, "invalid", None).await.status(), + 401 + ); + for query in ["snapshots=unknown", "snapshots=all&snapshots=refs", "pageSize=2"] { + assert_eq!( + fixture + .request(Method::GET, &format!("{PATH}?{query}"), "r", None) + .await + .status(), + 400 + ); + } + assert_eq!(fixture.request(Method::POST, PATH, "w", None).await.status(), 406); + assert_eq!( + fixture + .request(Method::HEAD, "/v1/namespaces/analytics/tables/absent", "r", None) + .await + .status(), + 404 + ); + let response = fixture + .request(Method::GET, "/v1/namespaces/absent/tables", "r", None) + .await; + assert_eq!(response.status(), 404); + assert!(response + .text() + .await + .unwrap() + .contains("NoSuchNamespaceException")); + let config = fixture + .request(Method::GET, "/v1/config", "r", None) + .await + .text() + .await + .unwrap(); + assert!(config.contains("GET /v1/{prefix}/namespaces/{namespace}/tables/{table}")); + assert!(!config.contains("POST /v1/{prefix}/namespaces/{namespace}/tables")); + fixture.finish().await; +} + +#[tokio::test] +async fn corrupt_authority_cannot_become_not_modified() { + let fixture = TestTableHttp::new().await; + let (mut head, _) = fixture.install("events").await; + head.metadata_digest[0] ^= 1; + fixture.put( + &crowdb_access_iceberg::table::head_key(head.catalog, head.table), + &crowdb_access_iceberg::record::StorageRecord::TableHead(Box::new(head)), + ); + assert_eq!( + fixture.request(Method::GET, PATH, "r", Some("*")).await.status(), + 503 + ); + fixture.finish().await; +} + +#[tokio::test] +async fn read_admission_is_shared_by_complete_and_paged_lists_and_releases_after_errors() { + use std::sync::{atomic::Ordering, Arc}; + use std::time::Duration; + + let fixture = Arc::new(TestTableHttp::new().await); + fixture.store.scan_delay_ms.store(500, Ordering::SeqCst); + let mut readers = Vec::new(); + for index in 0..4 { + let fixture = fixture.clone(); + readers.push(tokio::spawn(async move { + let path = if index % 2 == 0 { + "/v1/namespaces/analytics/tables" + } else { + "/v1/namespaces/analytics/tables?pageToken=" + }; + fixture.request(Method::GET, path, "r", None).await.status() + })); + } + tokio::time::timeout(Duration::from_secs(1), async { + while fixture.store.scans.load(Ordering::SeqCst) < 4 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert_eq!(fixture.request(Method::GET, PATH, "r", None).await.status(), 503); + for reader in readers { + assert_eq!(reader.await.unwrap(), 200); + } + fixture.store.scan_delay_ms.store(0, Ordering::SeqCst); + for _ in 0..8 { + assert_eq!( + fixture + .request(Method::GET, &format!("{PATH}?snapshots=bad"), "r", None) + .await + .status(), + 400 + ); + } + assert_eq!( + fixture + .request(Method::GET, PATH, "r", Some(&"x".repeat(8193))) + .await + .status(), + 400 + ); + assert_eq!(fixture.request(Method::GET, PATH, "r", None).await.status(), 404); + Arc::try_unwrap(fixture).ok().unwrap().finish().await; +} diff --git a/app/crowdb-access-server/tests/iceberg_table_lifecycle_test.rs b/app/crowdb-access-server/tests/iceberg_table_lifecycle_test.rs new file mode 100644 index 000000000..06da2ac96 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_table_lifecycle_test.rs @@ -0,0 +1,252 @@ +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +use fixture::TestTableHttp; +use reqwest::Method; +use serde_json::{json, Value}; + +const TABLES: &str = "/v1/namespaces/analytics/tables"; +const TABLE: &str = "/v1/namespaces/analytics/tables/events"; +const RENAME: &str = "/v1/tables/rename"; + +fn key() -> String { + static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(); + format!( + "{:08x}-{:04x}-7000-8000-{:012x}", + now >> 16, + now & 0xffff, + NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed) + ) +} + +fn create() -> Value { + json!({"name":"events", "schema":{"type":"struct","schema-id":0, + "fields":[{"id":1,"name":"id","type":"long","required":true}]}}) +} + +async fn value(response: reqwest::Response, status: u16) -> Value { + let actual = response.status(); + let bytes = response.text().await.unwrap(); + assert_eq!(actual.as_u16(), status, "{bytes}"); + serde_json::from_str(&bytes).unwrap() +} + +async fn empty(response: reqwest::Response) { + let status = response.status(); + let bytes = response.bytes().await.unwrap(); + assert_eq!(status.as_u16(), 204, "{bytes:?}"); + assert!(bytes.is_empty()); +} + +async fn delete(test: &TestTableHttp, path: &str, role: &str, identity: &str) -> reqwest::Response { + reqwest::Client::new() + .delete(format!("{}{path}", test.endpoint())) + .bearer_auth(role.repeat(32)) + .header("idempotency-key", identity) + .send() + .await + .unwrap() +} + +#[tokio::test] +async fn drop_enforces_writer_and_replays_without_deleting_recreated_table() { + for purge in [false, true] { + let test = TestTableHttp::writable().await; + let before = value(test.post(TABLES, "w", None, &create()).await, 200).await; + let path = format!("{TABLE}?purgeRequested={purge}"); + for role in ["r", "m", "c"] { + value(delete(&test, &path, role, &key()).await, 403).await; + } + let identity = key(); + empty(delete(&test, &path, "w", &identity).await).await; + value(test.request(Method::GET, TABLE, "r", None).await, 404).await; + let after = value(test.post(TABLES, "w", None, &create()).await, 200).await; + assert_ne!(before["metadata"]["table-uuid"], after["metadata"]["table-uuid"]); + empty(delete(&test, &path, "w", &identity).await).await; + assert_eq!( + value(test.request(Method::GET, TABLE, "r", None).await, 200).await, + after + ); + value( + delete( + &test, + &format!("{TABLE}?purgeRequested={}", !purge), + "w", + &identity, + ) + .await, + 409, + ) + .await; + test.finish().await; + } +} + +#[tokio::test] +async fn rename_preserves_metadata_supports_cross_namespace_and_never_aliases_old_name() { + let test = TestTableHttp::writable().await; + let before = value(test.post(TABLES, "w", None, &create()).await, 200).await; + let source = json!({"namespace":["analytics"], "name":"events"}); + let target = json!({"namespace":["analytics"], "name":"renamed"}); + let rename = json!({"source":source,"destination":target}); + for role in ["r", "m", "c"] { + value(test.post(RENAME, role, None, &rename).await, 403).await; + } + let identity = key(); + empty(test.post(RENAME, "w", Some(&identity), &rename).await).await; + value(test.request(Method::GET, TABLE, "r", None).await, 404).await; + let renamed = "/v1/namespaces/analytics/tables/renamed"; + assert_eq!( + value(test.request(Method::GET, renamed, "r", None).await, 200).await, + before + ); + value( + test.post(TABLE, "w", None, &json!({"requirements":[],"updates":[]})) + .await, + 404, + ) + .await; + value(test.post(TABLES, "w", None, &create()).await, 200).await; + empty(test.post(RENAME, "w", Some(&identity), &rename).await).await; + value( + test.post("/v1/namespaces", "w", None, &json!({"namespace":["destination"]})) + .await, + 200, + ) + .await; + empty( + test.post( + RENAME, + "w", + None, + &json!({"source":target, + "destination":{"namespace":["destination"],"name":"moved"}}), + ) + .await, + ) + .await; + value(test.request(Method::GET, renamed, "r", None).await, 404).await; + let moved = "/v1/namespaces/destination/tables/moved"; + assert_eq!( + value(test.request(Method::GET, moved, "r", None).await, 200).await, + before + ); + let update = + json!({"requirements":[],"updates":[{"action":"set-properties","updates":{"renamed":"yes"}}]}); + let committed = value(test.post(moved, "w", None, &update).await, 200).await; + assert_eq!(committed["metadata"]["properties"]["renamed"], "yes"); + assert_eq!(committed["metadata"]["location"], before["metadata"]["location"]); + let config = value(test.request(Method::GET, "/v1/config", "r", None).await, 200).await; + assert!(config["endpoints"] + .as_array() + .unwrap() + .contains(&json!("POST /v1/{prefix}/tables/rename"))); + test.finish().await; +} + +#[tokio::test] +async fn lifecycle_rejects_malformed_or_unsupported_requests_without_mutation() { + let test = TestTableHttp::writable().await; + let before = value(test.post(TABLES, "w", None, &create()).await, 200).await; + value(test.request(Method::GET, RENAME, "w", None).await, 406).await; + value(delete(&test, RENAME, "w", &key()).await, 406).await; + value(delete(&test, TABLES, "w", &key()).await, 406).await; + for query in [ + "purgeRequested=1", + "purgeRequested=true&purgeRequested=false", + "unknown=true", + ] { + value(delete(&test, &format!("{TABLE}?{query}"), "w", &key()).await, 400).await; + } + for body in [ + json!({}), + json!({"source":{"namespace":[],"name":"events"},"destination":{"namespace":["analytics"],"name":"other"}}), + ] { + value(test.post(RENAME, "w", None, &body).await, 400).await; + } + let source = json!({"namespace":["analytics"],"name":"events"}); + value( + test.post( + RENAME, + "w", + None, + &json!({"source":source,"destination":{"namespace":["absent"],"name":"events"}}), + ) + .await, + 404, + ) + .await; + empty( + test.post(RENAME, "w", None, &json!({"source":source,"destination":source})) + .await, + ) + .await; + value( + test.post("/v1/namespaces/analytics/register", "w", None, &json!({})) + .await, + 406, + ) + .await; + assert_eq!( + value(test.request(Method::GET, TABLE, "r", None).await, 200).await, + before + ); + test.finish().await; +} + +#[tokio::test] +async fn drop_accepts_boolean_query_spelling_from_official_client() { + let test = TestTableHttp::writable().await; + value(test.post(TABLES, "w", None, &create()).await, 200).await; + empty(delete(&test, &format!("{TABLE}?purgeRequested=False"), "w", &key()).await).await; + value(test.request(Method::GET, TABLE, "r", None).await, 404).await; + test.finish().await; +} + +#[tokio::test] +async fn credential_refresh_follows_exact_renamed_identity_and_stops_after_drop() { + let test = TestTableHttp::vending().await; + let created = value(test.post(TABLES, "w", None, &create()).await, 200).await; + let old = created["config"]["client.refresh-credentials-endpoint"] + .as_str() + .unwrap(); + value(test.request(Method::GET, old, "w", None).await, 200).await; + empty( + test.post( + RENAME, + "w", + None, + &json!({"source":{"namespace":["analytics"],"name":"events"}, + "destination":{"namespace":["analytics"],"name":"renamed"}}), + ) + .await, + ) + .await; + value(test.request(Method::GET, old, "w", None).await, 404).await; + let renamed = "/v1/namespaces/analytics/tables/renamed"; + let moved = value(test.request(Method::GET, renamed, "w", None).await, 200).await; + let current = moved["config"]["client.refresh-credentials-endpoint"] + .as_str() + .unwrap(); + assert_ne!(old, current); + value(test.request(Method::GET, current, "w", None).await, 200).await; + empty(delete(&test, renamed, "w", &key()).await).await; + value(test.request(Method::GET, current, "w", None).await, 404).await; + let replacement = value(test.post(TABLES, "w", None, &create()).await, 200).await; + assert_ne!( + replacement["metadata"]["location"], + created["metadata"]["location"] + ); + value(test.request(Method::GET, old, "w", None).await, 404).await; + test.finish().await; +} diff --git a/app/crowdb-access-server/tests/iceberg_table_sdk_test.rs b/app/crowdb-access-server/tests/iceberg_table_sdk_test.rs new file mode 100644 index 000000000..e33c01b00 --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_table_sdk_test.rs @@ -0,0 +1,127 @@ +#![cfg(feature = "iceberg-e2e")] + +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires Maven and pinned Apache Iceberg Java dependencies"] +async fn official_commit_errors_preserve_heads_and_enforce_count_boundaries() { + let fixture = fixture::TestTableHttp::writable().await; + let endpoint = fixture.endpoint(); + let status = tokio::task::spawn_blocking(move || { + let maven = std::env::var_os("CROWDB_ICEBERG_E2E_MVN").unwrap_or_else(|| "mvn".into()); + std::process::Command::new("timeout") + .arg("60") + .arg(maven) + .args(["-o", "--batch-mode", "--no-transfer-progress", "-f"]) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_java/pom.xml" + )) + .args(["compile", "exec:java", "-Dexec.mainClass=TestIcebergCommitErrors"]) + .arg(format!("-Dexec.args={endpoint}")) + .status() + .unwrap() + }) + .await + .unwrap(); + fixture.finish().await; + assert!(status.success(), "official commit error acceptance failed"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires Maven and pinned Apache Iceberg Java dependencies"] +async fn official_rest_catalog_reads_fixture_generations_without_fileio() { + let fixture = fixture::TestTableHttp::new().await; + for name in ["events", "a+b", "%2F"] { + fixture.install(name).await; + } + let endpoint = fixture.endpoint(); + let status = tokio::task::spawn_blocking(move || { + let maven = std::env::var_os("CROWDB_ICEBERG_E2E_MVN").unwrap_or_else(|| "mvn".into()); + std::process::Command::new("timeout") + .arg("60") + .arg(maven) + .args(["-o", "--batch-mode", "--no-transfer-progress", "-f"]) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_java/pom.xml" + )) + .args(["compile", "exec:java", "-Dexec.mainClass=TestIcebergCatalogReads"]) + .arg(format!("-Dexec.args={endpoint}")) + .status() + .unwrap() + }) + .await + .unwrap(); + fixture.finish().await; + assert!(status.success(), "official RESTCatalog read acceptance failed"); +} + +#[tokio::test] +#[ignore = "requires Maven and pinned Apache Iceberg Java dependencies"] +async fn official_staged_catalog_preserves_exact_draft_credential_refresh_uri() { + let status = tokio::task::spawn_blocking(|| { + let maven = std::env::var_os("CROWDB_ICEBERG_E2E_MVN").unwrap_or_else(|| "mvn".into()); + std::process::Command::new("timeout") + .arg("60") + .arg(maven) + .args(["-o", "--batch-mode", "--no-transfer-progress", "-f"]) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_java/pom.xml" + )) + .args([ + "compile", + "exec:java", + "-Dexec.mainClass=TestIcebergDraftCredentials", + ]) + .status() + .unwrap() + }) + .await + .unwrap(); + assert!( + status.success(), + "official staged credential refresh acceptance failed" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires Maven and pinned Apache Iceberg Java dependencies"] +async fn official_catalog_creates_commits_upgrades_stages_and_refreshes_native_credentials() { + let fixture = fixture::TestTableHttp::vending().await; + let endpoint = fixture.endpoint(); + let status = tokio::task::spawn_blocking(move || { + let maven = std::env::var_os("CROWDB_ICEBERG_E2E_MVN").unwrap_or_else(|| "mvn".into()); + std::process::Command::new("timeout") + .arg("60") + .arg(maven) + .args(["-o", "--batch-mode", "--no-transfer-progress", "-f"]) + .arg(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_java/pom.xml" + )) + .args([ + "compile", + "exec:java", + "-Dexec.mainClass=TestIcebergCatalogWrites", + ]) + .arg(format!("-Dexec.args={endpoint}")) + .status() + .unwrap() + }) + .await + .unwrap(); + fixture.finish().await; + assert!( + status.success(), + "official native catalog write acceptance failed" + ); +} diff --git a/app/crowdb-access-server/tests/iceberg_table_write_test.rs b/app/crowdb-access-server/tests/iceberg_table_write_test.rs new file mode 100644 index 000000000..7b047690c --- /dev/null +++ b/app/crowdb-access-server/tests/iceberg_table_write_test.rs @@ -0,0 +1,268 @@ +#[path = "common/iceberg_file_blocks.rs"] +#[allow(dead_code)] +mod blocks; +#[path = "common/iceberg_store.rs"] +mod common; +#[path = "common/iceberg_table_http.rs"] +#[allow(dead_code)] +mod fixture; + +use crowdb_access_iceberg::catalog::{CatalogRepository, ClearBounds}; +use fixture::TestTableHttp; +use reqwest::Method; +use serde_json::{json, Value}; + +const TABLES: &str = "/v1/namespaces/analytics/tables"; +const TABLE: &str = "/v1/namespaces/analytics/tables/events"; + +fn key() -> String { + static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(); + format!( + "{:08x}-{:04x}-7000-8000-{:012x}", + now >> 16, + now & 0xffff, + NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed) + ) +} + +fn create(staged: bool) -> Value { + json!({"name":"events", "stage-create":staged, "schema":{"type":"struct","schema-id":0, + "fields":[{"id":91,"name":"id","type":"long","required":true}]}}) +} + +async fn value(response: reqwest::Response, status: u16) -> Value { + let actual = response.status(); + let text = response.text().await.unwrap(); + assert_eq!(actual.as_u16(), status, "{text}"); + serde_json::from_str(&text).unwrap() +} + +#[tokio::test] +async fn create_and_upgrade_follow_the_selected_persisted_version_profile() { + let fixture = TestTableHttp::writable_with_capabilities(0x003f).await; + let refused = value(fixture.post(TABLES, "w", None, &create(false)).await, 406).await; + assert_eq!(refused["error"]["type"], "UnsupportedOperationException"); + let mut v1 = create(false); + v1["properties"] = json!({"format-version":"1"}); + let created = value(fixture.post(TABLES, "w", None, &v1).await, 200).await; + assert_eq!(created["metadata"]["format-version"], 1); + let upgrade = + json!({"requirements":[],"updates":[{"action":"upgrade-format-version","format-version":2}]}); + value(fixture.post(TABLE, "w", None, &upgrade).await, 406).await; + let repository = CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap(); + common::activate_bits(&repository, 0x1fff).await; + let upgraded = value(fixture.post(TABLE, "w", None, &upgrade).await, 200).await; + assert_eq!(upgraded["metadata"]["format-version"], 2); + fixture.finish().await; +} + +#[tokio::test] +async fn direct_v1_to_v3_upgrade_requires_both_persisted_edges() { + let fixture = TestTableHttp::writable_with_capabilities(0x1fff).await; + let mut v1 = create(false); + v1["properties"] = json!({"format-version":"1"}); + value(fixture.post(TABLES, "w", None, &v1).await, 200).await; + let upgrade = + json!({"requirements":[],"updates":[{"action":"upgrade-format-version","format-version":3}]}); + value(fixture.post(TABLE, "w", None, &upgrade).await, 406).await; + let repository = CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap(); + common::activate_bits(&repository, 0x3fff).await; + let updated = value(fixture.post(TABLE, "w", None, &upgrade).await, 200).await; + assert_eq!(updated["metadata"]["format-version"], 3); + fixture.finish().await; +} + +#[tokio::test] +async fn create_and_update_replay_exact_results_and_enforce_independent_writer() { + let fixture = TestTableHttp::writable().await; + for role in ["r", "m", "c"] { + value(fixture.post(TABLES, role, None, &create(false)).await, 403).await; + } + let identity = key(); + let created = value( + fixture.post(TABLES, "w", Some(&identity), &create(false)).await, + 200, + ) + .await; + assert_eq!(created["metadata"]["schemas"][0]["fields"][0]["id"], 1); + assert_eq!( + value( + fixture.post(TABLES, "w", Some(&identity), &create(false)).await, + 200 + ) + .await, + created + ); + let mut changed = create(false); + changed["name"] = json!("other"); + value(fixture.post(TABLES, "w", Some(&identity), &changed).await, 409).await; + let identity = key(); + let update = json!({"requirements":[{"type":"assert-table-uuid","uuid":created["metadata"]["table-uuid"]}], + "updates":[{"action":"set-properties","updates":{"owner":"writer"}}]}); + let committed = value(fixture.post(TABLE, "w", Some(&identity), &update).await, 200).await; + assert_eq!(committed["metadata"]["properties"]["owner"], "writer"); + assert_eq!( + value(fixture.post(TABLE, "w", Some(&identity), &update).await, 200).await, + committed + ); + assert_eq!( + value(fixture.request(Method::GET, TABLE, "r", None).await, 200).await, + committed + ); + fixture.finish().await; +} + +#[tokio::test] +async fn failed_requirement_is_durable_and_does_not_publish_or_rebase() { + let fixture = TestTableHttp::writable().await; + let before = value(fixture.post(TABLES, "w", None, &create(false)).await, 200).await; + let identity = key(); + let update = json!({"requirements":[{"type":"assert-current-schema-id","current-schema-id":123}], + "updates":[{"action":"set-properties","updates":{"bad":"value"}}]}); + let rejected = value(fixture.post(TABLE, "w", Some(&identity), &update).await, 409).await; + assert_eq!(rejected["error"]["type"], "CommitFailedException"); + assert_eq!( + value(fixture.post(TABLE, "w", Some(&identity), &update).await, 409).await, + rejected + ); + assert_eq!( + value(fixture.request(Method::GET, TABLE, "r", None).await, 200).await, + before + ); + let operation: crowdb_access_iceberg::key::OperationId = identity.parse().unwrap(); + let journal = crowdb_access_iceberg::commit::TableCommitJournal::new(fixture.store.clone()); + assert_eq!( + journal + .load(fixture.context, operation) + .await + .unwrap() + .unwrap() + .phase, + crowdb_access_iceberg::commit::TableCommitPhase::Rejected + ); + fixture.finish().await; +} + +#[tokio::test] +async fn staged_create_is_invisible_until_standard_assert_create_commit() { + let fixture = TestTableHttp::writable().await; + let draft = value(fixture.post(TABLES, "w", None, &create(true)).await, 200).await; + assert!(draft.get("metadata-location").is_none()); + assert_eq!( + fixture.request(Method::HEAD, TABLE, "r", None).await.status(), + 404 + ); + let metadata = &draft["metadata"]; + let body = json!({"requirements":[{"type":"assert-create"}],"updates":[ + {"action":"assign-uuid","uuid":metadata["table-uuid"]}, + {"action":"upgrade-format-version","format-version":metadata["format-version"]}, + {"action":"add-schema","schema":metadata["schemas"][0]}, + {"action":"set-current-schema","schema-id":-1}, + {"action":"add-spec","spec":metadata["partition-specs"][0]}, + {"action":"set-default-spec","spec-id":-1}, + {"action":"add-sort-order","sort-order":metadata["sort-orders"][0]}, + {"action":"set-default-sort-order","sort-order-id":-1}, + {"action":"set-location","location":metadata["location"]}]}); + let identity = key(); + let committed = value(fixture.post(TABLE, "w", Some(&identity), &body).await, 200).await; + assert_eq!(committed["metadata"]["table-uuid"], metadata["table-uuid"]); + assert_eq!( + value(fixture.post(TABLE, "w", Some(&identity), &body).await, 200).await, + committed + ); + assert_eq!( + fixture.request(Method::HEAD, TABLE, "r", None).await.status(), + 204 + ); + fixture.finish().await; +} + +#[tokio::test] +async fn concurrent_identical_commit_keys_cannot_rebase_or_finalize_a_transient_conflict() { + let fixture = TestTableHttp::writable().await; + value(fixture.post(TABLES, "w", None, &create(false)).await, 200).await; + let identity = key(); + let body = + json!({"requirements":[],"updates":[{"action":"set-properties","updates":{"concurrent":"once"}}]}); + let (first, second) = tokio::join!( + fixture.post(TABLE, "w", Some(&identity), &body), + fixture.post(TABLE, "w", Some(&identity), &body) + ); + for response in [first, second] { + assert!( + matches!(response.status().as_u16(), 200 | 503), + "{}", + response.text().await.unwrap() + ); + } + let result = value(fixture.post(TABLE, "w", Some(&identity), &body).await, 200).await; + assert_eq!(result["metadata"]["properties"]["concurrent"], "once"); + let selected = crowdb_access_iceberg::table::TableRepository::new(fixture.store.clone()) + .select(fixture.context, fixture.namespace, "events") + .await + .unwrap() + .unwrap(); + assert_eq!(selected.head.generation, 2); + value( + fixture + .post( + "/v1/namespaces/analytics/tables/other", + "w", + Some(&identity), + &body, + ) + .await, + 409, + ) + .await; + fixture.finish().await; +} + +#[tokio::test] +async fn foreign_create_location_is_a_replayable_client_error() { + let fixture = TestTableHttp::writable().await; + let mut body = create(false); + body["location"] = json!("s3://external/table"); + let identity = key(); + let first = value(fixture.post(TABLES, "w", Some(&identity), &body).await, 400).await; + assert_eq!( + value(fixture.post(TABLES, "w", Some(&identity), &body).await, 400).await, + first + ); + assert_eq!( + fixture.request(Method::HEAD, TABLE, "r", None).await.status(), + 404 + ); + fixture.finish().await; +} + +#[tokio::test] +async fn missing_selected_file_rejection_survives_lost_durable_reply() { + let fixture = TestTableHttp::writable().await; + let created = value(fixture.post(TABLES, "w", None, &create(false)).await, 200).await; + let identity = key(); + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(); + let body = json!({"requirements":[],"updates":[{"action":"add-snapshot","snapshot":{ + "snapshot-id":123,"sequence-number":1,"timestamp-ms":u64::try_from(now).unwrap(),"schema-id":0, + "summary":{"operation":"append"},"manifest-list":format!("{}/metadata/missing.avro", created["metadata"]["location"].as_str().unwrap())}}, + {"action":"set-snapshot-ref","ref-name":"main","type":"branch","snapshot-id":123}]}); + fixture + .store + .lose_reply_kind + .store(4, std::sync::atomic::Ordering::SeqCst); + value(fixture.post(TABLE, "w", Some(&identity), &body).await, 503).await; + let result = value(fixture.post(TABLE, "w", Some(&identity), &body).await, 400).await; + assert_eq!(result["error"]["type"], "BadRequestException"); + assert_eq!( + value(fixture.request(Method::GET, TABLE, "r", None).await, 200).await, + created + ); + fixture.finish().await; +} diff --git a/app/crowdb-access-server/tests/s3_full_stack_test.rs b/app/crowdb-access-server/tests/s3_full_stack_test.rs index 8514676d8..b1091722c 100644 --- a/app/crowdb-access-server/tests/s3_full_stack_test.rs +++ b/app/crowdb-access-server/tests/s3_full_stack_test.rs @@ -461,6 +461,14 @@ fn assert_native_write_metrics(listen: &str) { } fn issue_credentials(access_binary: &Path, seeds: &str) -> (String, String) { + let missing = Command::new(access_binary) + .args(["lookup-user", "boto3-e2e"]) + .env("CROWDB_MANAGEMENT_SEEDS", seeds) + .env("CROWDB_S3_MASTER_KEY", MASTER_KEY) + .output() + .expect("look up absent S3 user"); + assert!(!missing.status.success()); + assert!(!String::from_utf8_lossy(&missing.stdout).contains("AWS_ACCESS_KEY_ID=")); let issued = Command::new(access_binary) .args(["issue-user", "boto3-e2e"]) .env("CROWDB_MANAGEMENT_SEEDS", seeds) @@ -473,6 +481,23 @@ fn issue_credentials(access_binary: &Path, seeds: &str) -> (String, String) { String::from_utf8_lossy(&issued.stderr) ); let issued = String::from_utf8(issued.stdout).expect("token issuer output is UTF-8"); + for command in ["ensure-user", "lookup-user", "ensure-user"] { + let resumed = Command::new(access_binary) + .args([command, "boto3-e2e"]) + .env("CROWDB_MANAGEMENT_SEEDS", seeds) + .env("CROWDB_S3_MASTER_KEY", MASTER_KEY) + .output() + .expect("run replay-safe S3 user-token issuer"); + assert!( + resumed.status.success(), + "token reconciliation failed:\n{}", + String::from_utf8_lossy(&resumed.stderr) + ); + let resumed = String::from_utf8(resumed.stdout).unwrap(); + for name in ["AWS_ACCESS_KEY_ID", "AWS_SECRET_ACCESS_KEY"] { + assert_eq!(output_value(&resumed, name), output_value(&issued, name)); + } + } ( output_value(&issued, "AWS_ACCESS_KEY_ID").to_owned(), output_value(&issued, "AWS_SECRET_ACCESS_KEY").to_owned(), diff --git a/app/crowdb-chunk-kv-server/tests/transition_worker_test.rs b/app/crowdb-chunk-kv-server/tests/transition_worker_test.rs index 476137984..8c907f8f5 100644 --- a/app/crowdb-chunk-kv-server/tests/transition_worker_test.rs +++ b/app/crowdb-chunk-kv-server/tests/transition_worker_test.rs @@ -444,7 +444,7 @@ async fn source_worker_quiesces_before_returning_release_proof() { durable_tail_offset: 0, } ); - assert_eq!(source.lifecycle(), PartitionLifecycle::WriteStalled); + assert_eq!(source.lifecycle(), PartitionLifecycle::TransferQuiesced); } #[tokio::test] diff --git a/app/crowdb-chunkdb/src/lifecycle/handler.rs b/app/crowdb-chunkdb/src/lifecycle/handler.rs index bfacfee0e..d7e30515d 100644 --- a/app/crowdb-chunkdb/src/lifecycle/handler.rs +++ b/app/crowdb-chunkdb/src/lifecycle/handler.rs @@ -588,9 +588,6 @@ impl LifecycleHandler { } } let now_ms = self.renew_liveness_if_due(chunk_id, writer_epoch).await?; - if acknowledged_cursor == chunk.acknowledged_cursor && closed_strip_sequence.is_none() { - return Ok(chunk); - } if let Some(sequence) = closed_strip_sequence { for strip in &mut chunk.strips { if strip.strip_sequence <= sequence && strip.sealed_ts_ms == 0 { diff --git a/app/crowdb-chunkdb/tests/full_stack_test.rs b/app/crowdb-chunkdb/tests/full_stack_test.rs index 1a01e06de..48019423f 100644 --- a/app/crowdb-chunkdb/tests/full_stack_test.rs +++ b/app/crowdb-chunkdb/tests/full_stack_test.rs @@ -1297,6 +1297,14 @@ async fn chunkdb_full_stack_allocate_seal_delete() { assert_eq!(sealed.capacity, 1024); assert!(sealed.cleanup_intents.is_empty()); eprintln!("chunk sealed"); + let owned_segments: Vec<_> = sealed + .strips + .iter() + .flat_map(|strip| match strip.strip.as_ref().unwrap() { + Strip::MirrorStrip(mirror) => mirror.segments.clone(), + Strip::EcStrip(ec) => ec.segments.clone(), + }) + .collect(); // 8. Delete the chunk. let deleted = harness @@ -1306,6 +1314,25 @@ async fn chunkdb_full_stack_allocate_seal_delete() { .expect("delete_chunk"); assert_eq!(deleted.state, ChunkState::Deleted as i32); eprintln!("chunk deleted"); + assert!(deleted.strips.is_empty()); + let disk_records = cluster.make_ddb_kv_client(); + for segment in &owned_segments { + let records = disk_records + .read_zone_records( + (STORE_ID, DATA_GROUP_ID), + &segment.disk_id.unwrap(), + segment.zone_index, + ) + .await + .unwrap(); + assert!( + records.free.iter().any(|record| { + record.key.unit_offset == segment.unit_offset + && record.key.allocation_ts == segment.allocation_ts + }), + "disk block must be freed before chunk layout is cleared: {segment:?}" + ); + } // 9. Delete again → should return the same idempotent tombstone. let deleted_again = harness @@ -3174,7 +3201,8 @@ async fn chunkdb_shared_writer_cursor_is_fenced_and_orphan_is_sealed() { ) .await .expect("renew liveness without advancing cursor"); - assert_eq!(renewed.modify_ts, advanced.modify_ts); + assert!(renewed.modify_ts > advanced.modify_ts); + assert!(renewed.writer_lease_deadline_ms >= advanced.writer_lease_deadline_ms); assert_eq!(renewed.acknowledged_cursor, advanced.acknowledged_cursor); assert!(matches!( harness diff --git a/app/crowdb-cli/Cargo.toml b/app/crowdb-cli/Cargo.toml index eba71717a..e252b0320 100644 --- a/app/crowdb-cli/Cargo.toml +++ b/app/crowdb-cli/Cargo.toml @@ -60,6 +60,10 @@ path = "tests/kv_cli_test.rs" name = "lifecycle_cli_test" path = "tests/lifecycle_cli_test.rs" +[[test]] +name = "launch_registry_cli_test" +path = "tests/launch_registry_cli_test.rs" + [[test]] name = "mgmt_cli_test" path = "tests/mgmt_cli_test.rs" diff --git a/app/crowdb-cli/src/commands.rs b/app/crowdb-cli/src/commands.rs index cb6a6f9e3..c9129ff19 100644 --- a/app/crowdb-cli/src/commands.rs +++ b/app/crowdb-cli/src/commands.rs @@ -11,6 +11,7 @@ pub(crate) mod bench; pub(crate) mod chunk; pub(crate) mod cluster; pub(crate) mod kv; +pub(crate) mod launch; pub(crate) mod port_alloc; pub(crate) mod s3; @@ -72,7 +73,14 @@ pub(crate) fn op_context(cli: &Cli) -> Result { } /// Load the CLI's internal persisted state from its fixed runtime location. -pub(crate) fn load_config(_cli: &Cli) -> Result { +pub(crate) fn load_config(cli: &Cli) -> Result { + if let Some(path) = &cli.registry { + crowdb_console_shared::config::web::LaunchRegistry::load(path).map_err(|error| { + eprintln!("error: load launch registry: {error}"); + ExitCode::from(2) + })?; + return Ok(crowdb_console_shared::ConsoleConfig::default()); + } let path = config_path(); if !path.exists() { return Ok(crowdb_console_shared::ConsoleConfig::default()); @@ -100,7 +108,11 @@ fn config_path() -> std::path::PathBuf { } /// Persist the config from an [`OpContext`] back to the config file. -pub(crate) fn commit_config(_cli: &Cli, ctx: &OpContext) -> Result<(), ExitCode> { +pub(crate) fn commit_config(cli: &Cli, ctx: &OpContext) -> Result<(), ExitCode> { + if cli.registry.is_some() { + eprintln!("error: launch registry mode cannot persist local cluster topology"); + return Err(ExitCode::from(2)); + } let path = config_path(); if let Some(parent) = path.parent() { std::fs::create_dir_all(parent).map_err(|e| { diff --git a/app/crowdb-cli/src/commands/chunk/diskdb.rs b/app/crowdb-cli/src/commands/chunk/diskdb.rs index 61f317440..fbea07602 100644 --- a/app/crowdb-cli/src/commands/chunk/diskdb.rs +++ b/app/crowdb-cli/src/commands/chunk/diskdb.rs @@ -9,6 +9,7 @@ use clap::Subcommand; use crowdb_console_shared::ops::chunk; +use crate::commands::launch::{self, LaunchVerb}; use crate::commands::op_context; use crate::Cli; @@ -16,19 +17,23 @@ use crate::Cli; pub enum ChunkDiskdbVerb { Deploy { #[arg(short = 'n', long)] - node: String, + node: u64, + }, + Start { + #[arg(short = 'n', long)] + node: u64, }, Restart { #[arg(short = 'n', long)] - node: String, + node: u64, }, Stop { #[arg(short = 'n', long)] - node: String, + node: u64, }, Delete { #[arg(short = 'n', long)] - node: String, + node: u64, }, /// List living diskdb instances discovered from the group-0 /// service registry. Pass `--endpoint` to bypass discovery and @@ -50,6 +55,36 @@ pub enum ChunkDiskdbVerb { pub async fn run_chunk_diskdb_verb(cli: &Cli, verb: ChunkDiskdbVerb) -> ExitCode { match verb { ChunkDiskdbVerb::List { endpoint } => run_list(cli, endpoint.as_deref()).await, + ChunkDiskdbVerb::Deploy { node } | ChunkDiskdbVerb::Start { node } => { + launch::run( + cli, + LaunchVerb::Start { + node, + service: "diskdb".into(), + }, + ) + .await + } + ChunkDiskdbVerb::Restart { node } => { + launch::run( + cli, + LaunchVerb::Restart { + node, + service: "diskdb".into(), + }, + ) + .await + } + ChunkDiskdbVerb::Stop { node } => { + launch::run( + cli, + LaunchVerb::Stop { + node, + service: "diskdb".into(), + }, + ) + .await + } other => { eprintln!("chunk diskdb {other:?} — not yet implemented (Phase 3)"); ExitCode::from(1) diff --git a/app/crowdb-cli/src/commands/chunk/stub.rs b/app/crowdb-cli/src/commands/chunk/stub.rs index 6981f5b7f..825dde45b 100644 --- a/app/crowdb-cli/src/commands/chunk/stub.rs +++ b/app/crowdb-cli/src/commands/chunk/stub.rs @@ -7,14 +7,15 @@ use std::process::ExitCode; use clap::Subcommand; +use crate::commands::launch::{self, LaunchVerb}; use crate::Cli; #[derive(Subcommand, Debug)] pub enum ChunkStubVerb { #[command(subcommand)] - Chunkdb(ChunkdbVerb), + Chunkdb(ServiceVerb), #[command(subcommand)] - Diskio(DiskioVerb), + Diskio(ServiceVerb), Allocate, Free, Write, @@ -23,24 +24,67 @@ pub enum ChunkStubVerb { } #[derive(Subcommand, Debug)] -pub enum ChunkdbVerb { +pub enum ServiceVerb { Deploy { #[arg(short = 'n', long)] - node: String, + node: u64, }, - List, -} - -#[derive(Subcommand, Debug)] -pub enum DiskioVerb { - Deploy { + Start { + #[arg(short = 'n', long)] + node: u64, + }, + Restart { #[arg(short = 'n', long)] - node: String, + node: u64, + }, + Stop { + #[arg(short = 'n', long)] + node: u64, }, List, } -pub async fn run_chunk_stub_verb(_cli: &Cli, verb: ChunkStubVerb) -> ExitCode { - eprintln!("chunk {verb:?} — not yet implemented (Phase 3)"); - ExitCode::from(1) +pub async fn run_chunk_stub_verb(cli: &Cli, verb: ChunkStubVerb) -> ExitCode { + match verb { + ChunkStubVerb::Chunkdb(verb) => run_service(cli, "chunkdb", verb).await, + ChunkStubVerb::Diskio(verb) => run_service(cli, "diskio", verb).await, + other => { + eprintln!("chunk {other:?} — not yet implemented (Phase 3)"); + ExitCode::from(1) + } + } +} + +async fn run_service(cli: &Cli, service: &str, verb: ServiceVerb) -> ExitCode { + if matches!(verb, ServiceVerb::List) { + return run_list(cli, service).await; + } + let service = service.into(); + let verb = match verb { + ServiceVerb::Deploy { node } | ServiceVerb::Start { node } => LaunchVerb::Start { node, service }, + ServiceVerb::Restart { node } => LaunchVerb::Restart { node, service }, + ServiceVerb::Stop { node } => LaunchVerb::Stop { node, service }, + ServiceVerb::List => unreachable!(), + }; + launch::run(cli, verb).await +} + +async fn run_list(cli: &Cli, service: &str) -> ExitCode { + let ctx = match crate::commands::op_context(cli) { + Ok(ctx) => ctx, + Err(code) => return code, + }; + match ctx.sysmd().read_service_instances(service).await { + Ok(instances) => { + println!("living {service} instances ({}):", instances.len()); + for (id, instance) in instances { + println!(" instance {id}: {}", instance.rpc_endpoint); + } + ExitCode::SUCCESS + } + Err(error) => { + eprintln!("error: read {service} registration: {error}"); + ExitCode::from(2) + } + } } diff --git a/app/crowdb-cli/src/commands/kv/logical.rs b/app/crowdb-cli/src/commands/kv/logical.rs index 3ac09e7a2..e5a83a4c7 100644 --- a/app/crowdb-cli/src/commands/kv/logical.rs +++ b/app/crowdb-cli/src/commands/kv/logical.rs @@ -7,7 +7,7 @@ use std::process::ExitCode; use clap::Subcommand; -use crate::commands::{commit_config, op_context}; +use crate::commands::op_context; use crate::Cli; // ── store ──────────────────────────────────────────────────────── @@ -51,9 +51,6 @@ pub async fn run_store_verb(cli: &Cli, verb: StoreVerb) -> ExitCode { }; match crowdb_console_shared::ops::kv_logical::add_store(&ctx, store_id, &node_ids).await { Ok(hosting) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!( "added store {store_id} on nodes: {}", hosting @@ -84,9 +81,6 @@ pub async fn run_store_verb(cli: &Cli, verb: StoreVerb) -> ExitCode { }; match crowdb_console_shared::ops::kv_logical::remove_store(&ctx, store_id).await { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!("removed store {store_id}"); ExitCode::SUCCESS } @@ -203,9 +197,6 @@ pub async fn run_group_verb(cli: &Cli, verb: GroupVerb) -> ExitCode { .await { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!("added group {group_id} in store {store_id}"); ExitCode::SUCCESS } @@ -236,9 +227,6 @@ pub async fn run_group_verb(cli: &Cli, verb: GroupVerb) -> ExitCode { }; match crowdb_console_shared::ops::kv_logical::remove_group(&ctx, store_id, group_id).await { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!("removed group {group_id} in store {store_id}"); ExitCode::SUCCESS } @@ -353,9 +341,6 @@ pub async fn run_replica_verb(cli: &Cli, verb: ReplicaVerb) -> ExitCode { .await { Ok(new_rid) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!("added replica {new_rid} to group {group_id} in store {store_id}"); ExitCode::SUCCESS } @@ -398,9 +383,6 @@ pub async fn run_replica_verb(cli: &Cli, verb: ReplicaVerb) -> ExitCode { match crowdb_console_shared::ops::kv_logical::remove_replica(&ctx, store_id, group_id, rid).await { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!("removed replica {rid} from group {group_id} in store {store_id}"); ExitCode::SUCCESS } diff --git a/app/crowdb-cli/src/commands/kv/server.rs b/app/crowdb-cli/src/commands/kv/server.rs index 37bd69bb1..260fff1e3 100644 --- a/app/crowdb-cli/src/commands/kv/server.rs +++ b/app/crowdb-cli/src/commands/kv/server.rs @@ -13,19 +13,24 @@ use crowdb_protocol::NodeId; use crate::commands::{commit_config, op_context}; use crate::Cli; +mod registry; + #[derive(Subcommand, Debug)] pub enum KvServerVerb { Deploy { #[arg(short = 'n', long)] node: String, #[arg(short = 'r', long)] - rest_port: u16, + rest_port: Option, #[arg(short = 'R', long)] - rpc_port: u16, + rpc_port: Option, #[arg(short = 'b', long)] binary: Option, }, - #[command(alias = "start")] + Start { + #[arg(short = 'n', long)] + node: String, + }, Restart { #[arg(short = 'n', long)] node: String, @@ -43,6 +48,9 @@ pub enum KvServerVerb { #[allow(clippy::too_many_lines)] pub async fn run_kv_server_verb(cli: &Cli, verb: KvServerVerb) -> ExitCode { + if let Some(path) = &cli.registry { + return registry::run(cli, path, verb).await; + } match verb { KvServerVerb::Deploy { node, @@ -50,6 +58,10 @@ pub async fn run_kv_server_verb(cli: &Cli, verb: KvServerVerb) -> ExitCode { rpc_port, binary, } => { + let (Some(rest_port), Some(rpc_port)) = (rest_port, rpc_port) else { + eprintln!("error: deploy requires management and RPC ports without --registry"); + return ExitCode::from(2); + }; let node_id: NodeId = match node.parse() { Ok(n) => n, Err(e) => { @@ -85,7 +97,7 @@ pub async fn run_kv_server_verb(cli: &Cli, verb: KvServerVerb) -> ExitCode { } } } - KvServerVerb::Restart { node } => { + KvServerVerb::Restart { node } | KvServerVerb::Start { node } => { let node_id: NodeId = match node.parse() { Ok(n) => n, Err(e) => { @@ -97,7 +109,7 @@ pub async fn run_kv_server_verb(cli: &Cli, verb: KvServerVerb) -> ExitCode { Ok(c) => c, Err(c) => return c, }; - match crowdb_console_shared::ops::kv_server::restart(&ctx, node_id, None).await { + match crowdb_console_shared::ops::kv_server::restart(&ctx, node_id, None, None, &[]).await { Ok(d) => { if let Err(c) = commit_config(cli, &ctx) { return c; diff --git a/app/crowdb-cli/src/commands/kv/server/registry.rs b/app/crowdb-cli/src/commands/kv/server/registry.rs new file mode 100644 index 000000000..16e6ed993 --- /dev/null +++ b/app/crowdb-cli/src/commands/kv/server/registry.rs @@ -0,0 +1,119 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::path::Path; +use std::process::ExitCode; + +use crowdb_console_shared::config::web::LaunchRegistry; +use crowdb_console_shared::error::{Error, Result}; +use crowdb_console_shared::launch::LaunchRuntime; + +use super::KvServerVerb; +use crate::Cli; + +pub(super) async fn run(cli: &Cli, path: &Path, verb: KvServerVerb) -> ExitCode { + match execute(cli, path, verb).await { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("error: {error}"); + ExitCode::from(2) + } + } +} + +async fn execute(cli: &Cli, path: &Path, verb: KvServerVerb) -> Result<()> { + let mut registry = LaunchRegistry::load(path)?; + let runtime = LaunchRuntime::for_registry(path)?; + if matches!(verb, KvServerVerb::List) { + println!("{:<12} {:<26} PID", "NODE", "HOST"); + for launch in registry + .launches + .iter() + .filter(|launch| launch.service_id == "kv") + { + let identity = runtime.status(launch).await?; + println!( + "{:<12} {:<26} {}", + launch.node_id, + launch.host, + identity.map_or_else(|| "stopped".into(), |identity| identity.pid.to_string()) + ); + } + return Ok(()); + } + let node = match &verb { + KvServerVerb::Deploy { node, .. } + | KvServerVerb::Start { node } + | KvServerVerb::Restart { node } + | KvServerVerb::Stop { node } + | KvServerVerb::Delete { node } => node, + KvServerVerb::List => unreachable!(), + } + .parse::() + .map_err(|error| Error::Validation { + field: "node".into(), + message: error.to_string(), + })?; + let launch = registry + .launches + .iter() + .find(|launch| launch.node_id == node && launch.service_id == "kv") + .cloned() + .ok_or_else(|| Error::NotFound { + kind: "configured kv launch".into(), + id: node.to_string(), + })?; + match verb { + KvServerVerb::Deploy { + rest_port, + rpc_port, + binary, + .. + } => { + if rest_port.is_some() || rpc_port.is_some() || binary.is_some() { + return Err(Error::Validation { + field: "registry".into(), + message: "binary and listener arguments must come from the launch registry".into(), + }); + } + let identity = runtime.start(&launch).await?; + println!("started kv on node {node} (pid {})", identity.pid); + } + KvServerVerb::Start { .. } => { + let identity = runtime.start(&launch).await?; + println!("started kv on node {node} (pid {})", identity.pid); + } + KvServerVerb::Restart { .. } => { + let identity = runtime.restart(&launch).await?; + println!("restarted kv on node {node} (pid {})", identity.pid); + } + KvServerVerb::Stop { .. } => { + runtime.stop(&launch).await?; + println!("stopped kv on node {node}"); + } + KvServerVerb::Delete { .. } => { + let ctx = crate::commands::op_context(cli) + .map_err(|_| Error::Config("cannot initialize authority client".into()))?; + if ctx + .sysmd() + .list_all_replicas() + .await? + .iter() + .any(|replica| replica.node_id == node) + { + return Err(Error::Conflict { + kind: "node with replica membership".into(), + id: node.to_string(), + }); + } + runtime.stop(&launch).await?; + registry + .launches + .retain(|record| record.node_id != node || record.service_id != "kv"); + registry.save(path)?; + println!("removed kv launch for node {node}"); + } + KvServerVerb::List => unreachable!(), + } + Ok(()) +} diff --git a/app/crowdb-cli/src/commands/launch.rs b/app/crowdb-cli/src/commands/launch.rs new file mode 100644 index 000000000..83ccfd36a --- /dev/null +++ b/app/crowdb-cli/src/commands/launch.rs @@ -0,0 +1,101 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Process controls using launch policy, independent of cluster topology. + +use std::process::ExitCode; + +use clap::Subcommand; +use crowdb_console_shared::config::web::LaunchRegistry; +use crowdb_console_shared::error::{Error, Result}; +use crowdb_console_shared::launch::LaunchRuntime; + +use crate::Cli; + +#[derive(Subcommand, Debug)] +pub enum LaunchVerb { + /// Show configured processes and their current runtime identities. + List, + /// Start a configured process, preserving an already running instance. + Start { + #[arg(long)] + node: u64, + #[arg(long)] + service: String, + }, + Restart { + #[arg(long)] + node: u64, + #[arg(long)] + service: String, + }, + Stop { + #[arg(long)] + node: u64, + #[arg(long)] + service: String, + }, +} + +pub(crate) async fn run(cli: &Cli, verb: LaunchVerb) -> ExitCode { + match execute(cli, verb).await { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("error: {error}"); + ExitCode::from(2) + } + } +} + +async fn execute(cli: &Cli, verb: LaunchVerb) -> Result<()> { + let path = cli + .registry + .as_ref() + .ok_or_else(|| Error::Config("process controls require --registry".into()))?; + let registry = LaunchRegistry::load(path)?; + let runtime = LaunchRuntime::for_registry(path)?; + if matches!(verb, LaunchVerb::List) { + println!("{:<12} {:<16} {:<26} PID", "NODE", "SERVICE", "HOST"); + for launch in registry.launches { + let identity = runtime.status(&launch).await?; + println!( + "{:<12} {:<16} {:<26} {}", + launch.node_id, + launch.service_id, + launch.host, + identity.map_or_else(|| "stopped".into(), |identity| identity.pid.to_string()) + ); + } + return Ok(()); + } + let (node, service) = match &verb { + LaunchVerb::Start { node, service } + | LaunchVerb::Restart { node, service } + | LaunchVerb::Stop { node, service } => (*node, service.as_str()), + LaunchVerb::List => unreachable!(), + }; + let launch = registry + .launches + .iter() + .find(|launch| launch.node_id == node && launch.service_id == service) + .ok_or_else(|| Error::NotFound { + kind: "configured launch".into(), + id: format!("{node}/{service}"), + })?; + match &verb { + LaunchVerb::Start { .. } => { + let identity = runtime.start(launch).await?; + println!("started {service} on node {node} (pid {})", identity.pid); + } + LaunchVerb::Restart { .. } => { + let identity = runtime.restart(launch).await?; + println!("restarted {service} on node {node} (pid {})", identity.pid); + } + LaunchVerb::Stop { .. } => { + runtime.stop(launch).await?; + println!("stopped {service} on node {node}"); + } + LaunchVerb::List => unreachable!(), + } + Ok(()) +} diff --git a/app/crowdb-cli/src/main.rs b/app/crowdb-cli/src/main.rs index e27f5d3fe..aa3c066ee 100644 --- a/app/crowdb-cli/src/main.rs +++ b/app/crowdb-cli/src/main.rs @@ -45,6 +45,10 @@ use commands::{ #[derive(Parser, Debug)] #[command(name = "crowdb-cli", version, about = "CrowDB cluster console (CLI)")] struct Cli { + /// Versioned bare-metal launch registry; process identity is stored separately. + #[arg(long, global = true, value_name = "PATH")] + registry: Option, + /// IP address of any system-group node; leader discovery is automatic. #[arg( long, @@ -87,6 +91,9 @@ impl Cli { #[derive(Subcommand, Debug)] enum Domain { + /// Start, restart, stop, or inspect services from the launch registry. + #[command(subcommand)] + Launch(commands::launch::LaunchVerb), /// Hardware topology + cluster-level ops. #[command(alias = "cls")] Cluster { @@ -253,6 +260,7 @@ async fn dispatch(mut cli: Cli) -> ExitCode { }, ); match command { + Domain::Launch(verb) => commands::launch::run(&cli, verb).await, Domain::Cluster { verb } => run_cluster_verb(&cli, verb).await, Domain::Kv { verb } => match verb { KvVerb::Server(sv) => run_kv_server_verb(&cli, sv).await, diff --git a/app/crowdb-cli/tests/common/direct.rs b/app/crowdb-cli/tests/common/direct.rs index 461ab3dbe..3b54c4612 100644 --- a/app/crowdb-cli/tests/common/direct.rs +++ b/app/crowdb-cli/tests/common/direct.rs @@ -162,6 +162,22 @@ pub async fn spawn_group0() -> Option { // Wait for the leader to be elected. wait_for_group0_leader(&client, Duration::from_secs(5)).await; + let context = crowdb_console_shared::ops::OpContext::new( + deployed.rpc_url.trim_start_matches("http://").to_string(), + vec![deployed.mgmt_url.clone()], + cfg, + ); + let deadline = std::time::Instant::now() + Duration::from_secs(5); + loop { + if matches!(context.live_node_mgmt_url(1).await, Ok(url) if url == deployed.mgmt_url) { + break; + } + assert!( + std::time::Instant::now() < deadline, + "node 1 did not register in Group 0" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } Some(Group0 { pid: deployed.pid, diff --git a/app/crowdb-cli/tests/launch_registry_cli_test.rs b/app/crowdb-cli/tests/launch_registry_cli_test.rs new file mode 100644 index 000000000..c3640d115 --- /dev/null +++ b/app/crowdb-cli/tests/launch_registry_cli_test.rs @@ -0,0 +1,171 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +#![cfg(target_os = "linux")] + +mod common; + +use std::os::unix::fs::PermissionsExt; +use std::path::Path; +use std::process::Command; +use std::time::Duration; + +use crowdb_console_shared::config::web::{LaunchRecord, LaunchRegistry}; +use crowdb_console_shared::launch::LaunchRuntime; +use crowdb_console_shared::lifecycle; +use crowdb_test_harness::test_dirs::tempdir_in_test_data; + +struct TestProcesses(Vec); +impl Drop for TestProcesses { + fn drop(&mut self) { + for pid in &self.0 { + if lifecycle::process_is_alive(*pid) { + let _ = lifecycle::stop_pid_with_timeout(*pid, Duration::from_secs(2)); + } + } + } +} + +fn run(registry: &Path, port: u16, args: &[&str]) -> String { + let args: Vec<_> = ["kv", "server"].into_iter().chain(args.iter().copied()).collect(); + run_command(registry, port, &args) +} + +fn run_command(registry: &Path, port: u16, args: &[&str]) -> String { + let output = Command::new(env!("CARGO_BIN_EXE_crowdb-cli")) + .arg("--registry") + .arg(registry) + .arg("--system-port") + .arg(port.to_string()) + .env("CROWDB_CLI_STATE", registry.with_file_name("invalid-legacy.toml")) + .args(args) + .output() + .unwrap(); + assert!( + output.status.success(), + "CLI failed: {}", + String::from_utf8_lossy(&output.stderr) + ); + String::from_utf8(output.stdout).unwrap() +} + +fn record(directory: &Path, service: &str) -> LaunchRecord { + let binary = directory.join("service"); + std::fs::write(&binary, "#!/bin/sh\nexec sleep 60\n").unwrap(); + std::fs::set_permissions(&binary, std::fs::Permissions::from_mode(0o700)).unwrap(); + let config = directory.join("service.toml"); + std::fs::write(&config, "").unwrap(); + LaunchRecord { + node_id: 701, + service_id: service.into(), + host: "localhost".into(), + ssh_credential_ref: None, + ssh_user: None, + ssh_port: 22, + binary_path: binary, + service_config_path: config, + workspace: directory.to_owned(), + auto_start: false, + args: Vec::new(), + readiness_url: None, + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn cli_uses_launch_registry_and_runtime_identity_without_legacy_state() { + let g0 = common::direct::spawn_group0() + .await + .expect("KV server binary must be built"); + let dir = tempdir_in_test_data("cli-launch-registry"); + std::fs::write(dir.path().join("invalid-legacy.toml"), "invalid legacy config").unwrap(); + let record = record(dir.path(), "kv"); + let path = dir.path().join("launches.toml"); + LaunchRegistry { + version: 1, + launches: vec![record.clone()], + } + .save(&path) + .unwrap(); + let runtime = LaunchRuntime::for_registry(&path).unwrap(); + let mut guard = TestProcesses(Vec::new()); + run(&path, g0.mgmt_port, &["deploy", "--node", "701"]); + let first = runtime.status(&record).await.unwrap().unwrap(); + guard.0.push(first.pid); + run(&path, g0.mgmt_port, &["start", "--node", "701"]); + assert_eq!(runtime.status(&record).await.unwrap(), Some(first)); + assert!(run(&path, g0.mgmt_port, &["list"]).contains(&first.pid.to_string())); + run(&path, g0.mgmt_port, &["restart", "--node", "701"]); + let next = runtime.status(&record).await.unwrap().unwrap(); + guard.0.push(next.pid); + assert_ne!(first, next); + run(&path, g0.mgmt_port, &["stop", "--node", "701"]); + assert!(runtime.status(&record).await.unwrap().is_none()); + run(&path, g0.mgmt_port, &["delete", "--node", "701"]); + assert!(LaunchRegistry::load(&path).unwrap().launches.is_empty()); + assert_eq!( + std::fs::read_to_string(dir.path().join("invalid-legacy.toml")).unwrap(), + "invalid legacy config" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn chunk_commands_and_generic_launch_controls_share_process_identity() { + let dir = tempdir_in_test_data("cli-chunk-launch"); + let records: Vec<_> = ["diskdb", "chunkdb", "diskio"] + .iter() + .map(|service| record(dir.path(), service)) + .collect(); + let path = dir.path().join("launches.toml"); + LaunchRegistry { + version: 1, + launches: records.clone(), + } + .save(&path) + .unwrap(); + let runtime = LaunchRuntime::for_registry(&path).unwrap(); + let mut guard = TestProcesses(Vec::new()); + for record in &records { + let mut args = vec!["chunk"]; + if record.service_id != "diskdb" { + args.push("stub"); + } + args.extend([record.service_id.as_str(), "deploy", "--node", "701"]); + run_command(&path, 9, &args); + let first = runtime.status(record).await.unwrap().unwrap(); + guard.0.push(first.pid); + run_command( + &path, + 9, + &[ + "launch", + "start", + "--node", + "701", + "--service", + &record.service_id, + ], + ); + assert_eq!(runtime.status(record).await.unwrap(), Some(first)); + assert!(run_command(&path, 9, &["launch", "list"]).contains(&first.pid.to_string())); + run_command( + &path, + 9, + &[ + "launch", + "restart", + "--node", + "701", + "--service", + &record.service_id, + ], + ); + let next = runtime.status(record).await.unwrap().unwrap(); + guard.0.push(next.pid); + assert_ne!(first, next); + let action = args.len() - 3; + args[action] = "stop"; + run_command(&path, 9, &args); + assert!(runtime.status(record).await.unwrap().is_none()); + } + assert_eq!(LaunchRegistry::load(&path).unwrap().launches, records); +} diff --git a/app/crowdb-diskio/src/engine/uring/uring_engine.cpp b/app/crowdb-diskio/src/engine/uring/uring_engine.cpp index b402c8913..107e244cf 100644 --- a/app/crowdb-diskio/src/engine/uring/uring_engine.cpp +++ b/app/crowdb-diskio/src/engine/uring/uring_engine.cpp @@ -8,6 +8,7 @@ # include "disk/disk.h" # include +# include # include namespace crowdb::diskio @@ -21,6 +22,9 @@ UringEngine::UringEngine(unsigned ring_entries) cfg.mode = crowdb::common::PollingMode::Hybrid; topo.pipelines.push_back(cfg); uring_ = std::make_unique(std::move(topo)); + if (!uring_->valid()) { + throw std::runtime_error("io_uring initialization failed"); + } } UringEngine::UringEngine(unsigned ring_entries, crowdb::common::PollingMode mode, crowdb::common::HybridConfig hybrid, @@ -34,6 +38,9 @@ UringEngine::UringEngine(unsigned ring_entries, crowdb::common::PollingMode mode cfg.sqpoll = sqpoll; topo.pipelines.push_back(cfg); uring_ = std::make_unique(std::move(topo)); + if (!uring_->valid()) { + throw std::runtime_error("io_uring initialization failed"); + } } void UringEngine::submit_write(Disk *disk, off_t phys_offset, const uint8_t *data, size_t size, diff --git a/app/crowdb-diskio/src/rpc/dio_server.cpp b/app/crowdb-diskio/src/rpc/dio_server.cpp index 2aeaafc7d..c92f46d5a 100644 --- a/app/crowdb-diskio/src/rpc/dio_server.cpp +++ b/app/crowdb-diskio/src/rpc/dio_server.cpp @@ -13,9 +13,12 @@ #include #include +#include #include #include +#include #include +#include namespace crowdb::diskio { @@ -255,7 +258,14 @@ crowdb::rpc::OutFrame *DiskioServer::handle_read(crowdb::rpc::Frame *request, cr static_cast(dproto::FBDiskIoRetCode_ZoneNotExist)); return nullptr; } - off_t phys_offset = static_cast(zone->base_offset + zone_offset); + if (zone->base_offset < 0 || zone->capacity < 0 || zone_offset > static_cast(zone->capacity) || + size > static_cast(zone->capacity) - zone_offset || + zone_offset > static_cast(std::numeric_limits::max() - zone->base_offset)) { + delete request; + send_error_response(conn, req_id, create_nano, msg_type, static_cast(dproto::FBDiskIoRetCode_IoError)); + return nullptr; + } + off_t phys_offset = zone->base_offset + static_cast(zone_offset); auto *pool = conn->pool(); auto *read_buf = pool->alloc(size); @@ -265,23 +275,65 @@ crowdb::rpc::OutFrame *DiskioServer::handle_read(crowdb::rpc::Frame *request, cr return nullptr; } + auto io_offset = phys_offset; + size_t io_size = size; + size_t prefix = 0; + std::shared_ptr aligned_data; + if (disk->is_o_direct()) { + const size_t alignment = std::max(4096, disk->block_size()); + if ((alignment & (alignment - 1)) != 0 || alignment > static_cast(std::numeric_limits::max()) || + static_cast(phys_offset) + size > + static_cast(std::numeric_limits::max()) - alignment) { + read_buf->release(); + delete request; + send_error_response(conn, req_id, create_nano, msg_type, + static_cast(dproto::FBDiskIoRetCode_InvalidAlignment)); + return nullptr; + } + prefix = static_cast(phys_offset) % alignment; + io_offset = phys_offset - static_cast(prefix); + io_size = ((prefix + size + alignment - 1) / alignment) * alignment; + if (io_offset < zone->base_offset || + io_size > static_cast(zone->capacity - (io_offset - zone->base_offset))) { + read_buf->release(); + delete request; + send_error_response(conn, req_id, create_nano, msg_type, + static_cast(dproto::FBDiskIoRetCode_InvalidAlignment)); + return nullptr; + } + aligned_data = std::shared_ptr(static_cast(std::aligned_alloc(alignment, io_size)), + [](uint8_t *data) { std::free(data); }); + if (aligned_data == nullptr) { + read_buf->release(); + delete request; + send_error_response(conn, req_id, create_nano, msg_type, + static_cast(dproto::FBDiskIoRetCode_IoError)); + return nullptr; + } + } + delete request; Disk *disk_ptr = disk.get(); auto started = std::chrono::steady_clock::now(); - disk_ptr->engine()->submit_read(disk_ptr, phys_offset, read_buf->data, size, test_pattern_offset, - [this, conn, req_id, create_nano, msg_type, read_buf, size, started](int res) { + auto *io_data = aligned_data ? aligned_data.get() : read_buf->data; + disk_ptr->engine()->submit_read(disk_ptr, io_offset, io_data, io_size, test_pattern_offset, + [this, conn, req_id, create_nano, msg_type, read_buf, size, io_size, prefix, + aligned_data = std::move(aligned_data), disk = std::move(disk), started](int res) { read_latency_->observe(elapsed_nanos(started)); int16_t ret_code = static_cast(dproto::FBDiskIoRetCode_Success); crowdb::rpc::Buffer *data = nullptr; if (res < 0) { ret_code = static_cast(dproto::FBDiskIoRetCode_IoError); } - else if (static_cast(res) < size) { + else if (static_cast(res) < io_size) { ret_code = static_cast(dproto::FBDiskIoRetCode_PartialWrite); } else { - read_buf->len = static_cast(res); + if (aligned_data) { + std::memcpy(read_buf->data, aligned_data.get() + prefix, size); + } + read_buf->len = size; data = read_buf; } auto *pool = conn->pool(); diff --git a/app/crowdb-diskio/tests/dio_server_test.cpp b/app/crowdb-diskio/tests/dio_server_test.cpp index 0d864aee5..4b058b256 100644 --- a/app/crowdb-diskio/tests/dio_server_test.cpp +++ b/app/crowdb-diskio/tests/dio_server_test.cpp @@ -183,6 +183,14 @@ TEST(DiskioServerTest, WriteAndReadRoundTrip) auto disk_set = std::make_shared(); disk_set->add(disk); + std::string direct_path = temp_path(); + ASSERT_EQ(::truncate(direct_path.c_str(), 1 << 16), 0); + std::vector direct_zones; + direct_zones.push_back({0, 0, 1 << 16}); + auto direct_disk = std::make_shared(crowdb::diskio::DiskId{1, 2}, direct_path, engine, + std::move(direct_zones), true); + ASSERT_GE(direct_disk->fd(), 0); + disk_set->add(direct_disk); // Start the RPC server. RpcServer server; @@ -243,6 +251,26 @@ TEST(DiskioServerTest, WriteAndReadRoundTrip) ASSERT_EQ(read_state.recv_data.size(), DATA_SIZE); EXPECT_EQ(std::memcmp(read_state.recv_data.data(), payload.data(), DATA_SIZE), 0); + Buffer *direct_write_ctrl = build_write_request(pool, 21, {1, 2}, 0, 0, DATA_SIZE, wall_time_ms()); + Buffer *direct_write_data = pool->alloc(DATA_SIZE); + direct_write_data->write(payload.data(), DATA_SIZE); + DioState direct_write_state; + ASSERT_TRUE(caller.send(&client_transport, conn.get(), 21, direct_write_ctrl, direct_write_data, + static_cast(rproto::FBMsgType_EDiskWriteRequest), dio_on_complete, + &direct_write_state)); + ASSERT_TRUE(wait_for(direct_write_state)); + ASSERT_EQ(direct_write_state.ret_code.load(), static_cast(dproto::FBDiskIoRetCode_Success)); + + Buffer *direct_read_ctrl = build_read_request(pool, 22, {1, 2}, 0, 2048, 238); + DioState direct_read_state; + ASSERT_TRUE(caller.send(&client_transport, conn.get(), 22, direct_read_ctrl, nullptr, + static_cast(rproto::FBMsgType_EDiskReadRequest), dio_on_complete, + &direct_read_state)); + ASSERT_TRUE(wait_for(direct_read_state)); + ASSERT_EQ(direct_read_state.ret_code.load(), static_cast(dproto::FBDiskIoRetCode_Success)); + ASSERT_EQ(direct_read_state.recv_data.size(), 238); + EXPECT_EQ(std::memcmp(direct_read_state.recv_data.data(), payload.data() + 2048, 238), 0); + client_transport.stop(); server.stop(); } diff --git a/app/crowdb-diskio/tests/uring_engine_test.cpp b/app/crowdb-diskio/tests/uring_engine_test.cpp index 58f77d015..e27c9a582 100644 --- a/app/crowdb-diskio/tests/uring_engine_test.cpp +++ b/app/crowdb-diskio/tests/uring_engine_test.cpp @@ -22,6 +22,7 @@ # include # include # include +# include # include # include # include @@ -119,6 +120,11 @@ class TestDisk : public crowdb::diskio::Disk }; } // namespace +TEST(UringEngine, RejectsUninitializedRing) +{ + EXPECT_THROW(crowdb::diskio::UringEngine(0), std::runtime_error); +} + TEST(UringEngine, WriteReadRoundTrip) { std::string path = temp_path(); diff --git a/app/crowdb-kv-server/Cargo.toml b/app/crowdb-kv-server/Cargo.toml index 8114cda2b..d035aa39f 100644 --- a/app/crowdb-kv-server/Cargo.toml +++ b/app/crowdb-kv-server/Cargo.toml @@ -42,6 +42,10 @@ reqwest = { version = "0.12", features = ["json"] } crowdb-rpc-ffi = { path = "../../lib/crowdb-rpc/ffi", features = ["test-util"] } crowdb-test-harness = { path = "../../lib/crowdb-test-harness" } +[[test]] +name = "group0_discovery_test" +path = "tests/group0_discovery_test.rs" + [[test]] name = "async_ops_test" path = "tests/async_ops_test.rs" diff --git a/app/crowdb-kv-server/src/background.rs b/app/crowdb-kv-server/src/background.rs index 0fffcbdf6..8ad3a1413 100644 --- a/app/crowdb-kv-server/src/background.rs +++ b/app/crowdb-kv-server/src/background.rs @@ -3,5 +3,7 @@ //! Background tasks: service keepalive and binding monitor. +pub(crate) mod discovery; pub mod domain_monitor; +pub mod identity; pub mod keepalive; diff --git a/app/crowdb-kv-server/src/background/discovery.rs b/app/crowdb-kv-server/src/background/discovery.rs new file mode 100644 index 000000000..9898ff7e0 --- /dev/null +++ b/app/crowdb-kv-server/src/background/discovery.rs @@ -0,0 +1,75 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Durable connection hints for processes launched before Group 0 exists. + +use std::io; +use std::path::Path; +use std::sync::atomic::{AtomicU64, Ordering}; + +use crowdb_protocol::mgmt::Group0DiscoveryRequest; +use tokio::io::AsyncWriteExt; + +const FILE_NAME: &str = "group0-discovery.json"; +static NEXT_WRITE: AtomicU64 = AtomicU64::new(0); + +pub(crate) fn validate(request: &Group0DiscoveryRequest) -> io::Result<()> { + if request.management_seeds.is_empty() || request.management_seeds.len() > 16 { + return Err(io::Error::other( + "one to sixteen Group 0 management seeds are required", + )); + } + for seed in &request.management_seeds { + let uri: axum::http::Uri = seed.parse().map_err(io::Error::other)?; + if uri.scheme_str() != Some("http") + || uri.host().map_or(true, str::is_empty) + || uri + .authority() + .map_or(true, |authority| authority.as_str().contains('@')) + || uri.path() != "/" + || uri.query().is_some() + { + return Err(io::Error::other( + "management seed must be an unauthenticated HTTP origin", + )); + } + } + Ok(()) +} + +pub(crate) async fn load(root: &Path) -> io::Result>> { + let body = match tokio::fs::read(root.join(FILE_NAME)).await { + Ok(body) => body, + Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(None), + Err(error) => return Err(error), + }; + let request: Group0DiscoveryRequest = serde_json::from_slice(&body)?; + validate(&request)?; + Ok(Some(request.management_seeds)) +} + +pub(crate) async fn save(root: &Path, request: &Group0DiscoveryRequest) -> io::Result<()> { + validate(request)?; + tokio::fs::create_dir_all(root).await?; + let temporary = root.join(format!( + ".group0-discovery-{}-{}.tmp", + std::process::id(), + NEXT_WRITE.fetch_add(1, Ordering::Relaxed) + )); + let result = async { + let mut file = tokio::fs::OpenOptions::new() + .write(true) + .create_new(true) + .open(&temporary) + .await?; + file.write_all(&serde_json::to_vec(request)?).await?; + file.sync_all().await?; + tokio::fs::rename(&temporary, root.join(FILE_NAME)).await?; + tokio::fs::File::open(root).await?.sync_all().await + } + .await; + if result.is_err() { + let _ = tokio::fs::remove_file(&temporary).await; + } + result +} diff --git a/app/crowdb-kv-server/src/background/domain_monitor/chunk_kv/balance.rs b/app/crowdb-kv-server/src/background/domain_monitor/chunk_kv/balance.rs index a3ac671e7..1f887007a 100644 --- a/app/crowdb-kv-server/src/background/domain_monitor/chunk_kv/balance.rs +++ b/app/crowdb-kv-server/src/background/domain_monitor/chunk_kv/balance.rs @@ -36,9 +36,6 @@ pub async fn plan(control: &Group0ControlPlane, descriptor: &DomainMonitorDescri let Some(mut catalog) = catalog::load_current(control).await? else { return Ok(()); }; - let child_owner_balance_enabled = descriptor.chunk_kv_range_balance.is_some(); - let policy = descriptor.chunk_kv_range_balance.clone().unwrap_or_default(); - policy.validate().map_err(|error| error.to_string())?; let now_ms = wall_time_ms(); let state = planning_state(control, descriptor, now_ms).await?; if state.healthy.is_empty() { @@ -81,14 +78,14 @@ pub async fn plan(control: &Group0ControlPlane, descriptor: &DomainMonitorDescri .iter() .flat_map(|page| page.entries.iter()) .collect(); - if plan_split(control, &entries, &state, &policy, now_ms).await? { + let Some(policy) = descriptor.chunk_kv_range_balance.as_ref() else { + return Ok(()); + }; + policy.validate().map_err(|error| error.to_string())?; + if plan_split(control, &entries, &state, policy, now_ms).await? { return Ok(()); } - if child_owner_balance_enabled { - plan_transfer(control, &entries, &state, &policy, now_ms).await - } else { - Ok(()) - } + plan_transfer(control, &entries, &state, policy, now_ms).await } async fn planning_state( diff --git a/app/crowdb-kv-server/src/background/identity.rs b/app/crowdb-kv-server/src/background/identity.rs new file mode 100644 index 000000000..20fba4614 --- /dev/null +++ b/app/crowdb-kv-server/src/background/identity.rs @@ -0,0 +1,86 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Stable service registration identity owned by one durable node root. + +use std::io::{self, Write}; +use std::path::Path; + +use crowdb_protocol::common::KvServerIdentity; +use serde::{Deserialize, Serialize}; + +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct StoredIdentity { + version: u32, + instance_id: u64, + node_id: Option, +} + +/// Load or durably create a registration identity before the server starts. +/// +/// # Errors +/// Rejects malformed state, changed explicit identities, and persistence errors. +pub fn load_or_create( + root: &Path, + requested: Option, + node_id: Option, +) -> io::Result { + std::fs::create_dir_all(root)?; + let path = root.join("service-identity.json"); + match std::fs::read(&path) { + Ok(body) => return decode(&body, requested, node_id), + Err(error) if error.kind() == io::ErrorKind::NotFound => {} + Err(error) => return Err(error), + } + let generated = crowdb_kv_client::new_client_id(); + let stored = StoredIdentity { + version: 1, + instance_id: requested.unwrap_or(generated), + node_id, + }; + let body = serde_json::to_vec(&stored)?; + let identity = decode(&body, requested, node_id)?; + let temporary = root.join(format!( + ".service-identity-{}-{generated}.tmp", + std::process::id() + )); + let result = (|| { + let mut file = std::fs::OpenOptions::new() + .write(true) + .create_new(true) + .open(&temporary)?; + file.write_all(&body)?; + file.sync_all()?; + // Publish a complete file without replacing another creator's identity. + match std::fs::hard_link(&temporary, &path) { + Ok(()) => { + std::fs::File::open(root)?.sync_all()?; + Ok(identity) + } + Err(error) if error.kind() == io::ErrorKind::AlreadyExists => { + decode(&std::fs::read(&path)?, requested, node_id) + } + Err(error) => Err(error), + } + })(); + let _ = std::fs::remove_file(temporary); + result +} + +fn decode(body: &[u8], requested: Option, node_id: Option) -> io::Result { + let stored: StoredIdentity = serde_json::from_slice(body)?; + if stored.version != 1 + || stored.instance_id == 0 + || stored.node_id != node_id + || requested.is_some_and(|id| id != stored.instance_id) + { + return Err(io::Error::other( + "service registration identity does not match this node", + )); + } + Ok(KvServerIdentity { + instance_id: stored.instance_id, + node_id: stored.node_id, + }) +} diff --git a/app/crowdb-kv-server/src/background/keepalive.rs b/app/crowdb-kv-server/src/background/keepalive.rs index 9adb80f20..f297e680f 100644 --- a/app/crowdb-kv-server/src/background/keepalive.rs +++ b/app/crowdb-kv-server/src/background/keepalive.rs @@ -12,7 +12,7 @@ use std::sync::Arc; use crowdb_kv_client::{ClientConfig, CrowdbKvClient, ServiceRegistryClient}; use crowdb_protocol::common::HostedGroup; use tokio::task::JoinHandle; -use tracing::{info, info_span, warn, Instrument}; +use tracing::{debug, info, info_span, warn, Instrument}; use crate::store_registry::KvStoreRegistry; @@ -31,47 +31,85 @@ impl KeepAliveLoop { /// from the registry each tick so the record reflects live state. pub fn spawn( registry: Arc, - instance_id: u64, + identity: crowdb_protocol::common::KvServerIdentity, rpc_endpoint: String, group0_endpoint: &str, + group0_management_seeds: Vec, data_root: String, interval_secs: u64, ) -> Self { let (stop_tx, stop_rx) = tokio::sync::oneshot::channel(); + let instance_id = identity.instance_id; let ep = group0_endpoint.to_string(); - // The management endpoint (rpc_endpoint) is an HTTP URL suitable - // for /topology discovery seeds. The group0_endpoint is the - // crowdb-rpc endpoint for direct KV ops via seed_leader. - let mgmt_seeds = vec![rpc_endpoint.clone()]; + // Before Group 0 exists, this node can seed its own RPC endpoint. + // Once Group 0 exists, use its management seeds for discovery. + let bootstrap_local = group0_management_seeds.is_empty(); + let mgmt_seeds = if bootstrap_local { + vec![rpc_endpoint.clone()] + } else { + group0_management_seeds + }; let handle = tokio::spawn(async move { let kv_client = CrowdbKvClient::new(ClientConfig::new(mgmt_seeds)); - kv_client.seed_leader(0, 0, ep); + if bootstrap_local { + kv_client.seed_leader(0, 0, ep); + } let svc = ServiceRegistryClient::new(kv_client); + let mut discovered_seeds = Vec::new(); + + let discovery_ready = !bootstrap_local + || refresh_discovery(®istry, &svc, &mut discovered_seeds).await; // Initial registration. let (stores, groups) = hosted_summary(®istry); - if let Err(e) = svc - .register_kv_server(instance_id, &rpc_endpoint, &stores, &groups, "ok", &data_root) + let mut registered = if !discovery_ready { + false + } else if let Err(e) = svc + .register_kv_server(identity, &rpc_endpoint, &stores, &groups, "ok", &data_root) .await { warn!(error = %e, "keep-alive: initial register failed"); + false } else { info!(instance_id, "keep-alive: registered"); - } + true + }; - let mut ticker = tokio::time::interval(tokio::time::Duration::from_secs(interval_secs)); + let retry_interval = tokio::time::Duration::from_secs(1); + let regular_interval = tokio::time::Duration::from_secs(interval_secs); + let mut ticker = tokio::time::interval_at( + tokio::time::Instant::now() + if registered { regular_interval } else { retry_interval }, + if registered { regular_interval } else { retry_interval }, + ); ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); let mut stop_rx = stop_rx; loop { tokio::select! { _ = ticker.tick() => { + if bootstrap_local + && !refresh_discovery(®istry, &svc, &mut discovered_seeds).await + { + continue; + } let (stores, groups) = hosted_summary(®istry); if let Err(e) = svc - .heartbeat_kv_server(instance_id, &rpc_endpoint, &stores, &groups, "ok", &data_root) + .heartbeat_kv_server(identity, &rpc_endpoint, &stores, &groups, "ok", &data_root) .await { - warn!(error = %e, "keep-alive: heartbeat failed"); + if registered { + warn!(error = %e, "keep-alive: heartbeat failed"); + } else { + debug!(error = %e, "keep-alive: registration retry failed"); + } + } else if !registered { + registered = true; + info!(instance_id, "keep-alive: registered"); + ticker = tokio::time::interval_at( + tokio::time::Instant::now() + regular_interval, + regular_interval, + ); + ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); } } _ = &mut stop_rx => { @@ -117,6 +155,32 @@ impl KeepAliveLoop { } } +async fn refresh_discovery( + registry: &KvStoreRegistry, + service: &ServiceRegistryClient, + current: &mut Vec, +) -> bool { + match super::discovery::load(®istry.config.config_root).await { + Ok(Some(seeds)) if seeds != *current => { + service.kv().set_mgmt_seeds(seeds.clone()); + if let Err(error) = service.kv().refresh_topology().await { + debug!(%error, "keep-alive: discovery unavailable; deferring registration"); + return false; + } + *current = seeds; + true + } + Ok(Some(_)) => true, + Ok(None) => registry + .get_store(0) + .is_some_and(|store| store.get_group(0).is_some()), + Err(error) => { + warn!(%error, "keep-alive: discovery hints unreadable; deferring registration"); + false + } + } +} + impl Drop for KeepAliveLoop { fn drop(&mut self) { if let Some(tx) = self.stop_tx.take() { diff --git a/app/crowdb-kv-server/src/cli.rs b/app/crowdb-kv-server/src/cli.rs index 8c8f86c56..738738fd0 100644 --- a/app/crowdb-kv-server/src/cli.rs +++ b/app/crowdb-kv-server/src/cli.rs @@ -153,11 +153,19 @@ pub struct Cli { #[arg(long)] pub instance_id: Option, + /// Stable node identity published with the service-registry record. + #[arg(long, value_parser = clap::value_parser!(u64).range(1..))] + pub node_id: Option, + /// Keep-alive heartbeat interval in seconds. 0 disables the /// keep-alive loop. Default: 10. #[arg(long, default_value_t = 10)] pub keepalive_interval: u64, + /// HTTP management seeds for discovering Group 0 when this node does not host it. + #[arg(long = "group0-management-seed")] + pub group0_management_seeds: Vec, + /// chunkdb range binding monitor tick interval in seconds. 0 /// disables the monitor (the binding table is then operator-manual). /// Only the group-0 leader writes the table; followers run the tick diff --git a/app/crowdb-kv-server/src/main.rs b/app/crowdb-kv-server/src/main.rs index 644714e14..f1b57506b 100644 --- a/app/crowdb-kv-server/src/main.rs +++ b/app/crowdb-kv-server/src/main.rs @@ -140,6 +140,15 @@ async fn main() { args.apply_config_overrides(&mut config) .unwrap_or_else(|e| panic!("invalid config after CLI overrides: {e}")); + let service_identity = (args.keepalive_interval > 0).then(|| { + crowdb_kv_server::background::identity::load_or_create( + &config.config_root, + args.instance_id, + args.node_id, + ) + .unwrap_or_else(|error| panic!("failed to load service registration identity: {error}")) + }); + let registry = Arc::new( KvStoreRegistry::try_with_config(config.clone()) .unwrap_or_else(|error| panic!("failed to initialize WAL backend: {error}")) @@ -261,12 +270,7 @@ async fn main() { } // Start the keep-alive loop (registers under /srv/kv-server/). - let keepalive = if args.keepalive_interval > 0 { - let instance_id = args.instance_id.unwrap_or_else(|| { - let id = crowdb_kv_client::new_client_id(); - info!(instance_id = id, "keep-alive: generated instance id"); - id - }); + let keepalive = if let Some(identity) = service_identity { let mgmt_endpoint = format!("http://{display_addr}"); // The group-0 RPC endpoint is the first store's listen addr. // In first-boot mode (store 0 not created yet), derive it from @@ -282,9 +286,10 @@ async fn main() { .unwrap_or_else(|| format!("http://{display_addr}")); Some(crowdb_kv_server::background::keepalive::KeepAliveLoop::spawn( registry.clone(), - instance_id, + identity, mgmt_endpoint, &group0_ep, + args.group0_management_seeds.clone(), registry .config .node_root diff --git a/app/crowdb-kv-server/src/mgmt.rs b/app/crowdb-kv-server/src/mgmt.rs index a8144d7b7..596f0ce78 100644 --- a/app/crowdb-kv-server/src/mgmt.rs +++ b/app/crowdb-kv-server/src/mgmt.rs @@ -10,6 +10,7 @@ //! - [`system_init`] — system initialization + health-check endpoints //! - [`topology`] — topology export + metrics endpoints +mod discovery; mod group_ops; pub mod operation_registry; mod replica_ops; @@ -122,6 +123,7 @@ pub fn router(state: RegistryArc) -> Router { Router::new() .route("/health", get(system_init::health_check)) .route("/system/init", post(system_init::system_init)) + .route("/system/group0-discovery", post(discovery::update)) .route("/stores", get(store_ops::list_stores).post(store_ops::add_store)) .route( "/stores/:sid", @@ -190,6 +192,7 @@ pub fn router(state: RegistryArc) -> Router { replica_ops::remove_remote_replica, replica_ops::batch_add_remote_replicas, system_init::system_init, + discovery::update, topology::export_topology, topology::metrics ), diff --git a/app/crowdb-kv-server/src/mgmt/discovery.rs b/app/crowdb-kv-server/src/mgmt/discovery.rs new file mode 100644 index 000000000..08e98deb4 --- /dev/null +++ b/app/crowdb-kv-server/src/mgmt/discovery.rs @@ -0,0 +1,33 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use axum::extract::State; +use axum::http::StatusCode; +use axum::Json; +use crowdb_protocol::mgmt::Group0DiscoveryRequest; + +use crate::background::discovery; + +use super::{err_json, ErrorResponse, RegistryArc}; + +#[utoipa::path( + post, + path = "/system/group0-discovery", + tag = "management", + request_body = Group0DiscoveryRequest, + responses( + (status = 200, description = "Discovery hints persisted", body = Group0DiscoveryRequest), + (status = 400, description = "Invalid discovery hints", body = ErrorResponse), + (status = 500, description = "Persistence failed", body = ErrorResponse) + ) +)] +pub(super) async fn update( + State(state): State, + Json(request): Json, +) -> Result, (StatusCode, Json)> { + discovery::validate(&request).map_err(|error| err_json(StatusCode::BAD_REQUEST, error.to_string()))?; + discovery::save(&state.config.config_root, &request) + .await + .map_err(|error| err_json(StatusCode::INTERNAL_SERVER_ERROR, error.to_string()))?; + Ok(Json(request)) +} diff --git a/app/crowdb-kv-server/src/recovery/restore.rs b/app/crowdb-kv-server/src/recovery/restore.rs index 6101feb5e..a1e6f1971 100644 --- a/app/crowdb-kv-server/src/recovery/restore.rs +++ b/app/crowdb-kv-server/src/recovery/restore.rs @@ -122,9 +122,13 @@ pub async fn load_local_groups( } for (store_id, group_ids) in by_store { - let port = persisted_port_for_store(®istry.config.config_root, store_id) - .await - .unwrap_or_else(|| registry.next_port().unwrap_or(0)); + let port = if let Some(port) = persisted_port_for_store(®istry.config.config_root, store_id).await + { + registry.claim_port(port); + port + } else { + registry.next_port().unwrap_or(0) + }; let addr: SocketAddr = format!("0.0.0.0:{port}").parse().unwrap(); debug!(s = store_id, bind_addr = %addr, "restore: creating PxKvStore"); let mut store = PxKvStore::new(store_id, addr); diff --git a/app/crowdb-kv-server/src/store_registry.rs b/app/crowdb-kv-server/src/store_registry.rs index 329ed540e..cace9c01b 100644 --- a/app/crowdb-kv-server/src/store_registry.rs +++ b/app/crowdb-kv-server/src/store_registry.rs @@ -139,6 +139,15 @@ impl KvStoreRegistry { } } + /// Remove a restored store's persisted port from the allocation pool. + /// + /// # Panics + /// Panics if the internal mutex is poisoned. + pub fn claim_port(&self, port: u16) { + let mut pool = self.port_pool.lock().unwrap(); + pool.retain(|candidate| *candidate != port); + } + /// Peek at the first port in the pool without removing it. Used to /// derive the RPC endpoint for store 0 in first-boot mode (before the /// store is created via `/system/init`). diff --git a/app/crowdb-kv-server/tests/domain_monitor_test.rs b/app/crowdb-kv-server/tests/domain_monitor_test.rs index 55ef87e2f..a6d61825c 100644 --- a/app/crowdb-kv-server/tests/domain_monitor_test.rs +++ b/app/crowdb-kv-server/tests/domain_monitor_test.rs @@ -669,6 +669,13 @@ async fn assert_chunk_kv_split_plan(target_partitions_per_owner: u32, target_par let mut policy = descriptor(); policy.domain = "chunk-kv".into(); policy.service_registry_name = "chunk-kv".into(); + let driver = ChunkKvRangeMonitorDriver::new(); + driver.tick(&control, &policy).await.unwrap(); + assert!(control + .scan_all_prefix(Bytes::from(ChunkKvSplitKey::text_prefix_all()), 16) + .await + .unwrap() + .is_empty()); policy.chunk_kv_range_balance = Some(ChunkKvRangeBalancePolicy { target_partitions_per_owner, target_partition_bytes, @@ -676,7 +683,6 @@ async fn assert_chunk_kv_split_plan(target_partitions_per_owner: u32, target_par ..ChunkKvRangeBalancePolicy::default() }); - let driver = ChunkKvRangeMonitorDriver::new(); driver.tick(&control, &policy).await.unwrap(); driver.tick(&control, &policy).await.unwrap(); diff --git a/app/crowdb-kv-server/tests/group0_discovery_test.rs b/app/crowdb-kv-server/tests/group0_discovery_test.rs new file mode 100644 index 000000000..b098a376d --- /dev/null +++ b/app/crowdb-kv-server/tests/group0_discovery_test.rs @@ -0,0 +1,149 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +mod common; + +use std::time::Duration; + +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, ServiceRegistryClient}; +use crowdb_test_harness::test_dirs::TestDir; +use serde_json::json; + +use common::process::{start_test_server, start_test_server_at}; + +async fn wait_registered(service: &ServiceRegistryClient, endpoint: &str) { + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + loop { + if let Ok(instances) = service.read_all_kv_server_instances().await { + let matches: Vec<_> = instances + .iter() + .filter(|(_, value)| { + value + .extra + .as_ref() + .and_then(|extra| extra.kv_server.as_ref()) + .and_then(|identity| identity.node_id) + == Some(2) + }) + .collect(); + if matches.len() == 1 && matches[0].1.rpc_endpoint == endpoint { + return; + } + } + assert!( + tokio::time::Instant::now() < deadline, + "nonmember registration unavailable" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn prebootstrap_nonmember_discovers_group0_and_retains_hints_after_restart() { + let leader = start_test_server(&[ + "--instance-id", + "9001", + "--node-id", + "1", + "--keepalive-interval", + "1", + ]) + .await + .unwrap(); + let root = TestDir::new("nonmember-discovery").unwrap(); + let args = ["--node-id", "2", "--keepalive-interval", "1"]; + let nonmember = start_test_server_at(root.path(), &args, &[0]).await.unwrap(); + let http = reqwest::Client::new(); + let init = http + .post(format!("{}/system/init", leader.base_url())) + .json(&json!({"replica_id": 1, "start_election": true})) + .send() + .await + .unwrap(); + assert_eq!(init.status(), 201, "{}", init.text().await.unwrap()); + let service = ServiceRegistryClient::new(CrowdbKvClient::new(ClientConfig::new(vec![leader + .base_url() + .into()]))); + let endpoint = format!("{}/system/group0-discovery", nonmember.base_url()); + for invalid in [ + json!([]), + json!(["127.0.0.1:1"]), + json!(["http://user:password@localhost:1"]), + ] { + let response = http + .post(&endpoint) + .json(&json!({"management_seeds": invalid})) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 400); + } + let request = json!({"management_seeds": [leader.base_url()]}); + for _ in 0..2 { + let response = http.post(&endpoint).json(&request).send().await.unwrap(); + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + } + wait_registered(&service, nonmember.base_url()).await; + let initial = service.read_all_kv_server_instances().await.unwrap(); + let original_id = initial + .iter() + .find(|(_, value)| value.rpc_endpoint == nonmember.base_url()) + .unwrap() + .0; + let topology: serde_json::Value = http + .get(format!("{}/topology", nonmember.base_url())) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + assert!( + topology["stores"].as_array().unwrap().is_empty(), + "discovery must not create membership" + ); + drop(nonmember); + // Model a missed unregister while Group 0 was unavailable during shutdown. + service + .register_kv_server( + crowdb_protocol::common::KvServerIdentity { + instance_id: original_id, + node_id: Some(2), + }, + "http://127.0.0.1:1", + &[], + &[], + "ok", + &root.path().to_string_lossy(), + ) + .await + .unwrap(); + let restarted = start_test_server_at(root.path(), &args, &[0]).await.unwrap(); + wait_registered(&service, restarted.base_url()).await; + let instances = service.read_all_kv_server_instances().await.unwrap(); + let restarted_id = instances + .iter() + .find(|(_, value)| value.rpc_endpoint == restarted.base_url()) + .unwrap() + .0; + assert_eq!( + original_id, restarted_id, + "restart must preserve registration identity" + ); +} + +#[test] +fn persisted_identity_rejects_conflicting_configuration_and_corruption() { + use crowdb_kv_server::background::identity::load_or_create; + + let root = TestDir::new("service-identity").unwrap(); + let first = load_or_create(root.path(), Some(42), Some(2)).unwrap(); + assert_eq!(first.instance_id, 42); + assert_eq!(load_or_create(root.path(), None, Some(2)).unwrap(), first); + assert!(load_or_create(root.path(), Some(43), Some(2)).is_err()); + assert!(load_or_create(root.path(), None, Some(3)).is_err()); + let path = root.path().join("service-identity.json"); + std::fs::write(&path, b"partial").unwrap(); + assert!(load_or_create(root.path(), None, Some(2)).is_err()); + assert_eq!(std::fs::read(path).unwrap(), b"partial"); +} diff --git a/app/crowdb-kv-server/tests/restore_test.rs b/app/crowdb-kv-server/tests/restore_test.rs index 0a15fa9bd..0adf9ba29 100644 --- a/app/crowdb-kv-server/tests/restore_test.rs +++ b/app/crowdb-kv-server/tests/restore_test.rs @@ -168,6 +168,55 @@ async fn restart_restores_group0_from_disk() { ); } +#[tokio::test] +async fn restored_store_port_is_not_reallocated() { + let root = crowdb_test_harness::test_dirs::tempdir_in_test_data("restore-port"); + let root_path = root.path().to_path_buf(); + let server = start_test_server_at(&root_path, &[], &[0]) + .await + .expect("start first-boot server"); + let response: serde_json::Value = client() + .post(format!("{}/system/init", server.base_url())) + .json(&serde_json::json!({})) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + let persisted_port: u16 = response["listen_addr"] + .as_str() + .unwrap() + .rsplit(':') + .next() + .unwrap() + .parse() + .unwrap(); + drop(server); + + let next_port = + crowdb_protocol::port::alloc::alloc_test_port(crowdb_protocol::ServicePort::KvServerListen); + let server = start_test_server_at(&root_path, &[], &[persisted_port, next_port]) + .await + .expect("restart with restored port in pool"); + server + .wait_for_ready(std::time::Duration::from_secs(10)) + .await + .unwrap(); + let response = client() + .post(format!("{}/stores", server.base_url())) + .json(&serde_json::json!({"store_id": 7})) + .send() + .await + .unwrap(); + assert_eq!( + response.status().as_u16(), + 201, + "{}", + response.text().await.unwrap() + ); +} + // ── E2E: first boot with --root only (no toml) still works ─────── #[tokio::test] diff --git a/app/crowdb-web/Cargo.toml b/app/crowdb-web/Cargo.toml index 32ed6ec4f..70bfff35c 100644 --- a/app/crowdb-web/Cargo.toml +++ b/app/crowdb-web/Cargo.toml @@ -18,6 +18,7 @@ crowdb-common = { path = "../../lib/crowdb-common/rust" } crowdb-console-shared = { path = "../../lib/crowdb-console-shared" } crowdb-diskdb-client = { path = "../../lib/crowdb-diskdb-client" } crowdb-kv-client = { path = "../../lib/crowdb-kv-client" } +crowdb-monitor = { path = "../../container/crowdb-monitor" } crowdb-protocol = { path = "../../lib/crowdb-protocol" } crowdb-rpc-ffi = { path = "../../lib/crowdb-rpc/ffi" } clap = { version = "4", features = ["derive"] } @@ -29,6 +30,7 @@ tower-http = { version = "0.6", features = ["fs"] } tracing = { workspace = true } serde = { version = "1", features = ["derive"] } serde_json = "1" +subtle = "2" hex = "0.4" reqwest = { version = "0.12", default-features = false, features = ["rustls-tls", "json"] } futures = "0.3" @@ -41,3 +43,4 @@ tower = { version = "0.5", features = ["util"] } crowdb-kv = { path = "../../lib/crowdb-kv" } crowdb-rpc-ffi = { path = "../../lib/crowdb-rpc/ffi", features = ["test-util"] } crowdb-test-harness = { path = "../../lib/crowdb-test-harness", features = ["diskdb"] } +uuid = { version = "1", features = ["v4"] } diff --git a/app/crowdb-web/src/auth.rs b/app/crowdb-web/src/auth.rs new file mode 100644 index 000000000..2410360e1 --- /dev/null +++ b/app/crowdb-web/src/auth.rs @@ -0,0 +1,49 @@ +use axum::extract::Request; +use axum::extract::State; +use axum::http::header::AUTHORIZATION; +use axum::http::{HeaderMap, StatusCode}; +use axum::middleware::Next; +use axum::response::Response; +use subtle::ConstantTimeEq; + +use crate::state::AppState; + +fn valid_management_bearer(headers: &HeaderMap, state: &AppState) -> bool { + let Some(expected) = state.management_token.as_deref() else { + return false; + }; + let mut values = headers.get_all(AUTHORIZATION).iter(); + let (Some(value), None) = (values.next(), values.next()) else { + return false; + }; + let Some((scheme, token)) = value.to_str().ok().and_then(|value| value.split_once(' ')) else { + return false; + }; + scheme.eq_ignore_ascii_case("bearer") + && token.len() == expected.len() + && bool::from(token.as_bytes().ct_eq(expected.as_bytes())) +} + +pub(crate) async fn management_check(State(state): State, headers: HeaderMap) -> StatusCode { + if state.management_token.is_none() { + StatusCode::SERVICE_UNAVAILABLE + } else if valid_management_bearer(&headers, &state) { + StatusCode::NO_CONTENT + } else { + StatusCode::UNAUTHORIZED + } +} + +pub(crate) async fn require_management_bearer( + State(state): State, + request: Request, + next: Next, +) -> Result { + if state.management_token.is_none() { + return Err(StatusCode::SERVICE_UNAVAILABLE); + } + if !valid_management_bearer(request.headers(), &state) { + return Err(StatusCode::UNAUTHORIZED); + } + Ok(next.run(request).await) +} diff --git a/app/crowdb-web/src/health.rs b/app/crowdb-web/src/health.rs index d91796b7b..49e3f061d 100644 --- a/app/crowdb-web/src/health.rs +++ b/app/crowdb-web/src/health.rs @@ -12,3 +12,18 @@ pub async fn healthz() -> &'static str { "ok" } + +pub async fn mode( + axum::extract::State(state): axum::extract::State, +) -> axum::Json { + let mode = match state.web_mode { + Some(crowdb_console_shared::config::web::WebMode::Docker) => "docker", + Some(crowdb_console_shared::config::web::WebMode::BareMetal) => "bare-metal-pending", + None => "legacy", + }; + axum::Json(serde_json::json!({"mode": mode})) +} + +pub async fn managed_api_unavailable() -> axum::http::StatusCode { + axum::http::StatusCode::SERVICE_UNAVAILABLE +} diff --git a/app/crowdb-web/src/kv.rs b/app/crowdb-web/src/kv.rs index 0e93c2621..5b734997d 100644 --- a/app/crowdb-web/src/kv.rs +++ b/app/crowdb-web/src/kv.rs @@ -14,7 +14,7 @@ use crowdb_console_shared::ops; use crowdb_kv_client::{GetOutcome, ScanOutcome}; use hex; use serde::{Deserialize, Serialize}; -use std::collections::HashSet; +use std::collections::{HashMap, HashSet}; use tokio::time::{sleep, Duration}; #[derive(Debug, Deserialize)] @@ -84,132 +84,21 @@ fn decode_hex(s: &str) -> Result, (axum::http::StatusCode, Json Result)> { - for attempt in 0..5 { - if let Some(view) = state.monitor_cache.resolve_group(sid, gid).await { - // A degraded group (one node down in a 3-node cluster) can still - // make progress as long as a quorum and a leader exist. Route to - // the leader whenever we know one; only refuse if the group is - // unavailable (lost quorum) or has no leader at all. - if view.state != GroupHealth::Unavailable && view.state != GroupHealth::Unknown { - // Use strict_leader_for so we only return when a node - // self-reports as Leader and is Up — routing to a follower - // triggers "not leader" + slow retry in the KV client. - if let Some((_rid, node_id)) = state.monitor_cache.strict_leader_for(sid, gid).await { - let endpoint = kv_endpoint_for_node(state, sid, node_id).await?; - // For non-group-0 stores, verify the endpoint uses the - // per-store listen port (not the node's default rpc_url). - // If the monitor cache doesn't have listen_addr yet, keep - // polling — sending to the wrong port triggers a 4-5s - // retry cycle in the KV client. - if sid == 0 || endpoint_has_store_port(state, sid, node_id, &endpoint).await { - return Ok(endpoint); - } - } - } - } - if attempt == 4 { - break; - } - refresh_group_nodes(state, sid, gid).await; - sleep(Duration::from_millis(50 * (1 + attempt))).await; - } - - // Last-resort fallback: use leader_for (first-healthy fallback) so the - // caller can attempt the op rather than failing immediately. The KV - // client's retry loop will handle "not leader" if this is a follower. - if let Some((_rid, node_id)) = state.monitor_cache.leader_for(sid, gid).await { - let endpoint = kv_endpoint_for_node(state, sid, node_id).await?; - if sid == 0 || endpoint_has_store_port(state, sid, node_id, &endpoint).await { - return Ok(endpoint); - } - } - - Err(( - StatusCode::NOT_FOUND, - Json(ErrorBody { - error: format!("group {gid} in store {sid} not found or has no healthy leader"), - }), - )) -} - -/// Check whether `endpoint`'s port matches the store's `listen_addr` -/// port from the monitor cache. Returns `false` if the cache has no -/// `listen_addr` for this store (meaning the endpoint fell back to -/// the node's default `rpc_url`, which is wrong for non-group-0 stores). -async fn endpoint_has_store_port(state: &AppState, sid: u64, node_id: NodeId, endpoint: &str) -> bool { - let snap = state.monitor_cache.snapshot().await; - let Some(listen_addr) = snap - .get(&node_id) - .and_then(|rec| rec.stores.get(&sid)) - .and_then(|ns| ns.listen_addr.as_ref()) - else { - return false; - }; - let Some(listen_port) = port_of(listen_addr) else { - return false; - }; - port_of(endpoint) == Some(listen_port) -} - -async fn kv_endpoint_for_node( - state: &AppState, - sid: u64, - node_id: NodeId, -) -> Result)> { - // Each `PxKvStore` listens on its own crowdb-rpc port (ephemeral when created - // via the management API with `port: None`), reported as the store's - // `listen_addr`. KV requests must target that per-store endpoint — the - // node's configured `rpc_url` is a different listener and does not host - // this store's groups. Combine the node host (from `rpc_url`) with the - // store's listen port; fall back to `rpc_url` if the cache has no - // `listen_addr` yet. - let store_port = { - let snap = state.monitor_cache.snapshot().await; - snap.get(&node_id) - .and_then(|rec| rec.stores.get(&sid)) - .and_then(|ns| ns.listen_addr.as_ref()) - .and_then(|addr| port_of(addr)) - .filter(|p| *p != 0) - }; - - let cfg = state.config.read().unwrap(); - let rpc_url = cfg - .server_for_node(node_id) - .and_then(|s| s.rpc_url.clone()) - .ok_or_else(|| { - err_502(format!( - "leader node {node_id} has no crowdb-rpc endpoint configured" - )) - })?; - - match store_port { - Some(port) => Ok(format!("http://{}:{port}", host_of(&rpc_url))), - None => Ok(rpc_url), - } + let (_, nodes, registered) = group_discovery(state, sid, gid).await?; + authoritative_leader_hint(state, sid, gid, &nodes, ®istered) + .await + .ok_or_else(|| err_502(format!("group {gid} in store {sid} has no confirmed live leader"))) } #[derive(Debug, Serialize)] @@ -220,18 +109,17 @@ pub struct EndpointResponse { } /// `GET /api/stores/:sid/groups/:gid/endpoint`. Resolve the crowdb-rpc -/// endpoint of the group's leader via the monitor cache, so a direct +/// endpoint of the group's leader via Group 0, so a direct /// crowdb-rpc client (the CLI bench engine) can dial it without touching any /// registry. Same resolution as the KV data plane uses internally. /// /// # Errors -/// `404` if the group is unknown / has no replicas; `502` if the -/// leader's node has no crowdb-rpc endpoint configured. +/// `404` if the group has no replicas; `502` if discovery or leader +/// confirmation is unavailable. pub async fn http_kv_endpoint( State(state): State, Path((sid, gid)): Path<(u64, u64)>, ) -> Result, (StatusCode, Json)> { - refresh_group_nodes(&state, sid, gid).await; let rpc_url = resolve_kv_endpoint(&state, sid, gid).await?; Ok(Json(EndpointResponse { rpc_url })) } @@ -253,49 +141,45 @@ fn host_of(rpc_url: &str) -> String { } } -/// Node ids hosting a replica of `(sid, gid)`, per the monitor cache, or (if -/// the cache has no record for the group yet) the persisted config replica -/// list -- so a restarted web console can still find the nodes to query. -async fn group_node_ids(state: &AppState, sid: u64, gid: u64) -> Vec { - if let Some(view) = state.monitor_cache.resolve_group(sid, gid).await { - view.replicas.into_iter().map(|r| r.node_id).collect() - } else { - let cfg = state.config.read().unwrap(); - cfg.groups - .iter() - .find(|g| g.store_id == sid && g.group_id == gid) - .map(|g| g.replicas.iter().map(|r| r.node_id).collect()) - .unwrap_or_default() +/// Build an `OpContext` for a KV data-plane request on `(sid, gid)`. +/// +/// Uses Group 0 membership and live service registrations for discovery. +async fn kv_op_context( + state: &AppState, + sid: u64, + gid: u64, +) -> Result)> { + let (ctx, nodes, registered) = group_discovery(state, sid, gid).await?; + if let Some(endpoint) = authoritative_leader_hint(state, sid, gid, &nodes, ®istered).await { + ctx.kv().seed_leader(sid, gid, endpoint); } + let seeds = registered + .into_values() + .filter(|endpoints| endpoints.len() == 1) + .flatten() + .collect(); + ctx.kv().set_mgmt_seeds(seeds); + Ok(ctx) } -/// Refresh the monitor cache for every node hosting a replica of -/// `(sid, gid)`. Called on initial endpoint resolution so the next -/// `leader_for` call observes a post-election view. -async fn refresh_group_nodes(state: &AppState, sid: u64, gid: u64) { - let node_ids = group_node_ids(state, sid, gid).await; - futures::future::join_all(node_ids.iter().map(|&nid| refresh_node_cache(state, nid))).await; -} +type GroupDiscovery = ( + crowdb_console_shared::ops::OpContext, + HashSet, + HashMap>, +); -/// `crowdb-kv-server` management-API base URLs (`ServerEntry::url`, e.g. -/// `http://host:rest_port`) for every node hosting a replica of `(sid, -/// gid)`. This is [`CrowdbKvClient`]'s discovery input (`GET /topology` on -/// each seed): any one reachable replica's own `/topology` response -/// carries the real leader's endpoint via its `remotes` list, so seeding -/// with every known replica's mgmt URL is enough for `CrowdbKvClient` to -/// self-heal a stale/dead leader without this module doing any endpoint -/// bookkeeping itself (C1-C2). -/// -/// # Errors -/// `404` if the group is unknown / has no replicas; `502` if none of its -/// replica nodes have a configured management URL. -async fn mgmt_seeds_for_group( +async fn group_discovery( state: &AppState, sid: u64, gid: u64, -) -> Result, (StatusCode, Json)> { - let node_ids = group_node_ids(state, sid, gid).await; - if node_ids.is_empty() { +) -> Result)> { + let ctx = state.op_context().await.map_err(|e| err_502(format!("{e}")))?; + let replicas = ctx + .sysmd() + .list_replicas_in_group(sid, gid) + .await + .map_err(|error| err_502(format!("Group 0 replica lookup failed: {error}")))?; + if replicas.is_empty() { return Err(( StatusCode::NOT_FOUND, Json(ErrorBody { @@ -303,84 +187,74 @@ async fn mgmt_seeds_for_group( }), )); } - - let cfg = state.config.read().unwrap(); - let mut seen = HashSet::new(); - let mut seeds = Vec::new(); - for node_id in &node_ids { - // Skip stopped servers — no runtime pid means the server process - // is not running. Including its URL as a seed only wastes time - // (connection-refused) during topology refresh. - if state.runtime_pid(*node_id).is_none() { - continue; - } - if let Some(server) = cfg.server_for_node(*node_id) { - if seen.insert(server.url.clone()) { - seeds.push(server.url.clone()); - } + let nodes: HashSet<_> = replicas.into_iter().map(|replica| replica.node_id).collect(); + let instances = ctx + .sysmd() + .read_all_kv_server_instances() + .await + .map_err(|error| err_502(format!("Group 0 service lookup failed: {error}")))?; + let mut registered = HashMap::>::new(); + for (_, instance) in instances { + if let Some(node_id) = instance + .extra + .as_ref() + .and_then(|extra| extra.kv_server.as_ref()) + .and_then(|extra| extra.node_id) + { + registered.entry(node_id).or_default().push(instance.rpc_endpoint); } } - drop(cfg); - - if seeds.is_empty() { + if !nodes.iter().any(|node_id| { + registered + .get(node_id) + .is_some_and(|endpoints| endpoints.len() == 1) + }) { return Err(err_502(format!( - "group {gid} in store {sid} has no configured server management URL" + "group {gid} in store {sid} has no live KV registration in Group 0" ))); } - Ok(seeds) + Ok((ctx, nodes, registered)) } -/// Build an `OpContext` for a KV data-plane request on `(sid, gid)`. -/// -/// Fails fast with `502` if no KV servers are deployed — the shared -/// `CrowdbKvClient` would have no seeds for topology discovery and -/// every op would retry for ~5s before failing. Returning a clear -/// error immediately is better than a silent timeout. -/// -/// Seeds + leader hint are synced from the current config + monitor -/// cache so the shared client's topology cache is fresh for this call. -async fn kv_op_context( +async fn authoritative_leader_hint( state: &AppState, sid: u64, gid: u64, -) -> Result)> { - let t0 = std::time::Instant::now(); - // Fail fast: if no KV servers are deployed, the client cannot - // discover any leader. Don't let it retry for seconds. - let all_seeds: Vec = { - let cfg = state.config.read().unwrap(); - cfg.servers - .iter() - .filter(|s| s.service_type == crowdb_console_shared::config::ServiceType::Kv) - .map(|s| s.url.clone()) - .collect() - }; - if all_seeds.is_empty() { - tracing::warn!("kv_op_context: no KV servers deployed — fail-fast 502 (store={sid}, group={gid})"); - return Err(err_502( - "no KV servers deployed — cluster not initialized; run cluster init first", - )); - } - // Validate the target group exists + has replicas. - let _ = mgmt_seeds_for_group(state, sid, gid).await?; - let ctx = state.op_context().await.map_err(|e| err_502(format!("{e}")))?; - ctx.kv().set_mgmt_seeds(all_seeds); - let t_resolve = std::time::Instant::now(); - if let Ok(endpoint) = resolve_kv_endpoint(state, sid, gid).await { - tracing::debug!( - "kv_op_context: resolve_kv_endpoint store={sid} group={gid} endpoint={endpoint} in {}ms (total {}ms)", - t_resolve.elapsed().as_millis(), - t0.elapsed().as_millis() - ); - ctx.kv().seed_leader(sid, gid, endpoint); - } else { - tracing::warn!( - "kv_op_context: resolve_kv_endpoint failed for store={sid} group={gid} in {}ms (total {}ms)", - t_resolve.elapsed().as_millis(), - t0.elapsed().as_millis() - ); + nodes: &HashSet, + registered: &HashMap>, +) -> Option { + for attempt in 0..5 { + let group_healthy = state + .monitor_cache + .resolve_group(sid, gid) + .await + .is_some_and(|view| !matches!(view.state, GroupHealth::Unavailable | GroupHealth::Unknown)); + if let Some((_, node_id)) = state + .monitor_cache + .strict_leader_for(sid, gid) + .await + .filter(|_| group_healthy) + { + if nodes.contains(&node_id) { + if let Some(endpoints) = registered.get(&node_id).filter(|endpoints| endpoints.len() == 1) { + let snapshot = state.monitor_cache.snapshot().await; + let store_port = snapshot + .get(&node_id) + .and_then(|record| record.stores.get(&sid)) + .and_then(|store| store.listen_addr.as_deref()) + .and_then(port_of); + if let Some(port) = store_port { + return Some(format!("http://{}:{port}", host_of(&endpoints[0]))); + } + } + } + } + if attempt < 4 { + futures::future::join_all(nodes.iter().map(|node_id| refresh_node_cache(state, *node_id))).await; + sleep(Duration::from_millis(50 * (1 + attempt))).await; + } } - Ok(ctx) + None } /// Get a value from the KV store. diff --git a/app/crowdb-web/src/launch.rs b/app/crowdb-web/src/launch.rs new file mode 100644 index 000000000..b10918511 --- /dev/null +++ b/app/crowdb-web/src/launch.rs @@ -0,0 +1,116 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Bare-metal process controls; launch policy never supplies cluster topology. + +use axum::extract::{Path, State}; +use axum::http::StatusCode; +use axum::routing::{get, post}; +use axum::{Json, Router}; +use crowdb_console_shared::config::web::LaunchRegistry; +use crowdb_console_shared::error::{Error, Result}; +use crowdb_console_shared::launch::{LaunchRuntime, ProcessIdentity}; +use serde::Serialize; + +use crate::error::{map_err, ErrorBody}; +use crate::state::AppState; + +type HttpResult = std::result::Result)>; + +pub(crate) fn routes() -> Router { + Router::new() + .route("/api/launches", get(list)) + .route("/api/launches/:node/:service/start", post(start)) + .route("/api/launches/:node/:service/restart", post(restart)) + .route("/api/launches/:node/:service/stop", post(stop)) +} + +fn policy(state: &AppState) -> Result<(LaunchRegistry, LaunchRuntime)> { + let path = state + .launch_registry_path + .as_ref() + .ok_or_else(|| Error::NotFound { + kind: "launch registry".into(), + id: "bare-metal".into(), + })?; + Ok((LaunchRegistry::load(path)?, LaunchRuntime::for_registry(path)?)) +} + +#[derive(Serialize)] +struct LaunchView { + node_id: u64, + service_id: String, + host: String, + auto_start: bool, + process: Option, +} + +async fn list(State(state): State) -> HttpResult>> { + let (registry, runtime) = policy(&state).map_err(map_err)?; + let mut views = Vec::new(); + for launch in registry.launches { + let process = runtime.status(&launch).await.map_err(map_err)?; + views.push(LaunchView { + node_id: launch.node_id, + service_id: launch.service_id, + host: launch.host, + auto_start: launch.auto_start, + process, + }); + } + Ok(Json(views)) +} + +enum Action { + Start, + Restart, + Stop, +} + +async fn act(state: &AppState, node: u64, service: &str, action: Action) -> Result> { + let (registry, runtime) = policy(state)?; + let launch = registry + .launches + .iter() + .find(|launch| launch.node_id == node && launch.service_id == service) + .ok_or_else(|| Error::NotFound { + kind: "configured launch".into(), + id: format!("{node}/{service}"), + })?; + match action { + Action::Start => runtime.start(launch).await.map(Some), + Action::Restart => runtime.restart(launch).await.map(Some), + Action::Stop => { + runtime.stop(launch).await?; + Ok(None) + } + } +} + +async fn start( + State(state): State, + Path((node, service)): Path<(u64, String)>, +) -> HttpResult>> { + act(&state, node, &service, Action::Start) + .await + .map(Json) + .map_err(map_err) +} + +async fn restart( + State(state): State, + Path((node, service)): Path<(u64, String)>, +) -> HttpResult>> { + act(&state, node, &service, Action::Restart) + .await + .map(Json) + .map_err(map_err) +} + +async fn stop( + State(state): State, + Path((node, service)): Path<(u64, String)>, +) -> HttpResult { + act(&state, node, &service, Action::Stop).await.map_err(map_err)?; + Ok(StatusCode::NO_CONTENT) +} diff --git a/app/crowdb-web/src/lib.rs b/app/crowdb-web/src/lib.rs index 6b722969c..9a8986b93 100644 --- a/app/crowdb-web/src/lib.rs +++ b/app/crowdb-web/src/lib.rs @@ -9,13 +9,17 @@ //! KV data plane with leader resolution via the monitor cache and //! `NotLeader` retry (A8), Swagger UI (A9), React SPA shell. +mod auth; pub mod corr_id; pub mod diskdb; pub mod error; pub mod expand; pub mod health; pub mod kv; +mod launch; pub mod lifecycle; +mod managed; +mod managed_logical; pub mod mgmt; pub mod owner_assignment; pub mod physical; @@ -27,10 +31,62 @@ pub use state::AppState; /// Build the Axum router used by both the binary and integration tests. #[allow(clippy::too_many_lines)] pub fn router(state: AppState) -> axum::Router { - use axum::routing::{delete, get, post}; + use axum::routing::{any, delete, get, post}; + + if state.managed_mode { + let authorization = + axum::middleware::from_fn_with_state(state.clone(), auth::require_management_bearer); + let managed = axum::Router::new() + .route("/healthz", get(health::healthz)) + .route("/api/mode", get(health::mode)) + .route("/api/authority", get(managed::authority)) + .route("/api/preview", get(managed::snapshot)) + .route("/api/management/check", post(auth::management_check)) + .route( + "/api/stores", + get(managed_logical::list_stores) + .merge(post(managed_logical::add_store).route_layer(authorization.clone())), + ) + .route( + "/api/stores/:sid", + get(managed_logical::get_store) + .merge(delete(managed_logical::remove_store).route_layer(authorization.clone())), + ) + .route( + "/api/stores/:sid/groups", + get(managed_logical::list_groups) + .merge(post(managed_logical::add_group).route_layer(authorization.clone())), + ) + .route( + "/api/stores/:sid/groups/:gid", + get(managed_logical::get_group) + .merge(delete(managed_logical::remove_group).route_layer(authorization.clone())), + ) + .route( + "/api/stores/:sid/groups/:gid/replicas", + get(managed_logical::list_replicas) + .merge(post(managed_logical::add_replica).route_layer(authorization.clone())), + ) + .route( + "/api/stores/:sid/groups/:gid/replicas/:rid", + get(managed_logical::get_replica) + .merge(delete(managed_logical::remove_replica).route_layer(authorization.clone())), + ) + .route("/api/*path", any(health::managed_api_unavailable)) + .fallback(spa::spa_fallback); + let managed = if state.web_mode == Some(crowdb_console_shared::config::web::WebMode::BareMetal) { + managed.merge(launch::routes().route_layer(authorization)) + } else { + managed + }; + return managed + .with_state(state) + .layer(axum::middleware::from_fn(corr_id::corr_id_layer)); + } axum::Router::new() .route("/healthz", get(health::healthz)) + .route("/api/mode", get(health::mode)) // ── Physical tree (A3): rack + node lifecycle ──────────────── .route( "/api/racks", diff --git a/app/crowdb-web/src/lifecycle.rs b/app/crowdb-web/src/lifecycle.rs index 0d2ffac76..90c736684 100644 --- a/app/crowdb-web/src/lifecycle.rs +++ b/app/crowdb-web/src/lifecycle.rs @@ -708,6 +708,7 @@ pub async fn http_deploy_node_server( server_id: node_id.to_string(), rest_port: body.rest_port, rpc_port: body.rpc_port, + group0_management_seeds: (*state.authority_seeds).clone(), election_profile: body .election_profile .clone() @@ -782,48 +783,19 @@ pub async fn http_restart_node_server( State(state): State, Path(node_id): Path, ) -> Result, (StatusCode, Json)> { - use crowdb_console_shared::lifecycle; - - // Stop the running process first (using the in-memory runtime PID). - // ops::kv_server::restart also tries to stop via entry.pid, but the - // runtime PID is authoritative for web-deployed servers. - if let Some(pid) = state.runtime_pid(node_id) { - let node = { - let cfg = state.config.read().unwrap(); - cfg.node(node_id).cloned() - }; - let _sent = match node { - Some(n) if n.ssh_enabled() => crowdb_console_shared::ssh::stop_via_ssh(&n, pid) - .await - .map_err(|e| err_502(format!("ssh stop (restart): {e}")))?, - _ => { - let timeout = if state.test_mode { - std::time::Duration::from_secs(1) - } else { - std::time::Duration::from_secs(15) - }; - let sent = - tokio::task::spawn_blocking(move || lifecycle::stop_pid_with_timeout(pid, timeout)) - .await - .map_err(|e| err_500(format!("spawn_blocking (restart): {e}")))? - .unwrap_or(false); - if lifecycle::process_is_alive(pid) { - return Err(err_502(format!( - "process {pid} is still alive after restart stop" - ))); - } - sent - } - }; - } - let workspace_dir = state .prepare_node_workspace(node_id) .map_err(|e| err_500(e.to_string()))?; let ctx = state.op_context().await.map_err(|e| err_502(format!("{e}")))?; - let deployed = ops::kv_server::restart(&ctx, node_id, Some(&workspace_dir)) - .await - .map_err(map_config_err)?; + let deployed = ops::kv_server::restart( + &ctx, + node_id, + Some(&workspace_dir), + state.runtime_pid(node_id), + &state.authority_seeds, + ) + .await + .map_err(map_config_err)?; // Apply the updated server entry directly to state.config (avoid // commit_op_context snapshot-replace race). let new_entry = ctx diff --git a/app/crowdb-web/src/main.rs b/app/crowdb-web/src/main.rs index d183d38d7..567809290 100644 --- a/app/crowdb-web/src/main.rs +++ b/app/crowdb-web/src/main.rs @@ -7,154 +7,188 @@ use std::net::SocketAddr; use clap::Parser; use crowdb_common::logging::init_file_and_console_logging_split; +use crowdb_console_shared::config::web::{LaunchRegistry, WebMode, WebProcessConfig}; use crowdb_protocol::WEB_BASE; use tracing::info; +#[derive(Parser, Debug)] +#[command(name = "crowdb-web")] +struct Args { + /// Bind address for the web server (default: 0.0.0.0) + #[arg(long, conflicts_with = "config")] + bind: Option, + + /// Port for the web server (default: 14000) + #[arg(long, conflicts_with = "config", value_parser = clap::value_parser!(u16).range(1..))] + port: Option, + + /// Use an in-memory registry instead of the persisted console config. + #[arg(long, conflicts_with = "config")] + test_mode: bool, + + /// Versioned web process configuration. + #[arg(long, value_name = "PATH")] + config: Option, + + /// Optional launch-only registry for bare-metal deployments. + #[arg(long, value_name = "PATH", requires = "config")] + registry: Option, + + /// Load the registry without reconciling service processes at startup. + #[arg(long, conflicts_with = "config")] + skip_startup_restore: bool, + + /// Log directory. Default: ~/.crowdb-kv/log. + #[arg(long, conflicts_with = "config")] + log_dir: Option, + + /// Log level for both Rust and C++ stacks. Default: "info" + /// (or derived from `RUST_LOG`). + #[arg(long)] + log_level: Option, + + /// Max log file size in MiB before rotation. Default: 30. + #[arg(long, conflicts_with = "config")] + log_max_file_mb: Option, + + /// Number of rotated log files to keep. Default: 5. + #[arg(long, conflicts_with = "config")] + log_max_files: Option, + + /// Also print logs to console (in addition to file logging). + #[arg(short = 'l', long)] + log: bool, + + /// Mirror C++ log lines at this level or above to stderr. + /// Default: "warn" (mirrors warn+error to stderr). + #[arg(long)] + log_stderr: Option, +} + #[tokio::main] async fn main() -> Result<(), Box> { - #[derive(Parser, Debug)] - #[command(name = "crowdb-web")] - struct Args { - /// Bind address for the web server (default: 0.0.0.0) - #[arg(long, default_value = "0.0.0.0")] - bind: String, - - /// Port for the web server (default: 14000) - #[arg(long, default_value_t = WEB_BASE, value_parser = clap::value_parser!(u16).range(1..))] - port: u16, - - /// Use an in-memory registry instead of the persisted console config. - #[arg(long, conflicts_with = "config")] - test_mode: bool, - - /// Console registry to load and persist instead of the default path. - #[arg(long, value_name = "PATH")] - config: Option, - - /// Load the registry without reconciling service processes at startup. - #[arg(long)] - skip_startup_restore: bool, - - /// Log directory. Default: ~/.crowdb-kv/log. - #[arg(long)] - log_dir: Option, - - /// Log level for both Rust and C++ stacks. Default: "info" - /// (or derived from `RUST_LOG`). - #[arg(long)] - log_level: Option, - - /// Max log file size in MiB before rotation. Default: 30. - #[arg(long, default_value_t = crowdb_common::logging::DEFAULT_LOG_MAX_FILE_MB)] - log_max_file_mb: usize, - - /// Number of rotated log files to keep. Default: 5. - #[arg(long, default_value_t = crowdb_common::logging::DEFAULT_LOG_MAX_FILES)] - log_max_files: usize, - - /// Also print logs to console (in addition to file logging). - #[arg(short = 'l', long)] - log: bool, - - /// Mirror C++ log lines at this level or above to stderr. - /// Default: "warn" (mirrors warn+error to stderr). - #[arg(long)] - log_stderr: Option, + let args = Args::parse(); + let process_config = args.config.as_deref().map(WebProcessConfig::load).transpose()?; + if args.registry.is_some() + && process_config + .as_ref() + .is_some_and(|config| config.mode == WebMode::Docker) + { + return Err("docker web does not accept --registry".into()); } + let launch_registry = args.registry.as_deref().map(LaunchRegistry::load).transpose()?; + let _log_guards = init_logging(&args, process_config.as_ref())?; - let args = Args::parse(); + let bind = process_config.as_ref().map_or_else( + || args.bind.as_deref().unwrap_or("0.0.0.0"), + |config| config.bind.as_str(), + ); + let port = process_config + .as_ref() + .map_or_else(|| args.port.unwrap_or(WEB_BASE), |config| config.port); + let addr: SocketAddr = format!("{bind}:{port}").parse()?; + info!(%addr, "crowdb-web starting"); - // Layered logging: INFO+ to rotating file, WARN+ to console. - // RUST_LOG overrides both sinks for debugging. The file layer uses - // the persistent console namespace by default; the guard must outlive the process - // so the non-blocking appender flushes on exit. - let log_dir = args.log_dir.clone().unwrap_or_else(|| { - crowdb_protocol::port::namespace::runtime_root() - .join("persistent") - .join("console") - .join("log") - }); - let log_dir_str = log_dir.to_string_lossy().to_string(); - let cpp_level = args - .log_level - .clone() - .unwrap_or_else(|| crowdb_common::logging::cpp_level_from_rust_log("info")); + // Load the persisted registry; absence yields an empty default. + // Mutating handlers (rack/node/server CRUD) write back to this path. + let path = if args.test_mode || process_config.is_some() { + None + } else { + crowdb_console_shared::TomlFileEngine::default_path() + }; + let cfg = match path.as_ref() { + Some(p) => { + let engine = crowdb_console_shared::TomlFileEngine::new(p.clone()); + crowdb_console_shared::ConsoleConfig::load_with_engine(&engine)? + } + None => crowdb_console_shared::ConsoleConfig::default(), + }; + let server_count = cfg.servers.len(); + let mut state = crowdb_web::AppState::with_config(cfg, path).with_test_mode(args.test_mode); + if let Some(config) = process_config { + state = state.with_process_config(&config); + state = state.with_management_token(std::env::var("CROWDB_ICEBERG_MANAGE_TOKEN")?)?; + } + if let Some(path) = &args.registry { + state = state.with_launch_registry(path.clone())?; + let started = state.start_configured_services().await?; + info!(started, "reconciled configured service launches"); + } + tracing::info!( + servers = server_count, + launches = launch_registry + .as_ref() + .map_or(0, |registry| registry.launches.len()), + "loaded web startup configuration" + ); + if !args.skip_startup_restore && !state.managed_mode { + crowdb_web::mgmt::startup_topology_check(&state).await; + } + + let listener = tokio::net::TcpListener::bind(addr).await?; + axum::serve(listener, crowdb_web::router(state)).await?; + Ok(()) +} - let _log_guards = if args.log { +fn init_logging( + args: &Args, + process_config: Option<&WebProcessConfig>, +) -> Result { + let log_max_file_mb = process_config.map_or_else( + || { + args.log_max_file_mb + .unwrap_or(crowdb_common::logging::DEFAULT_LOG_MAX_FILE_MB) + }, + |config| config.log_max_file_mb, + ); + let log_max_files = process_config.map_or_else( + || { + args.log_max_files + .unwrap_or(crowdb_common::logging::DEFAULT_LOG_MAX_FILES) + }, + |config| config.log_max_files, + ); + let log_dir = process_config + .map(|config| config.log_dir.clone()) + .or(args.log_dir.clone()) + .unwrap_or_else(|| { + crowdb_protocol::port::namespace::runtime_root() + .join("persistent") + .join("console") + .join("log") + }); + let guards = if args.log { init_file_and_console_logging_split( &log_dir, "console-web", - args.log_max_file_mb, - args.log_max_files, + log_max_file_mb, + log_max_files, "info", "warn", - ) - .map_err(|e| { - eprintln!("failed to initialize logging: {e}"); - e - })? + )? } else { crowdb_common::logging::init_file_logging( &log_dir, "console-web", - args.log_max_file_mb, - args.log_max_files, + log_max_file_mb, + log_max_files, "info", - ) - .map_err(|e| { - eprintln!("failed to initialize logging: {e}"); - e - })? + )? }; - - // Initialize the crowdb-rpc C++ spdlog logger so transport info/debug - // messages go to rotating files instead of spdlog's default stderr - // logger. Uses the SAME log directory as the Rust tracing init — - // not the literal "log" (fixes the previous directory mismatch). - // No-op without spdlog. + let cpp_level = args + .log_level + .clone() + .unwrap_or_else(|| crowdb_common::logging::cpp_level_from_rust_log("info")); crowdb_rpc_ffi::init_logging( - &log_dir_str, + &log_dir.to_string_lossy(), &cpp_level, - args.log_max_file_mb, - args.log_max_files, + log_max_file_mb, + log_max_files, "crowdb-web-rpc", ); - - // Default: mirror warn+error to stderr (previous unconditional - // behavior). Override with --log-stderr or disable with - // --log-stderr off. let stderr_level = args.log_stderr.as_deref().unwrap_or("warn"); if stderr_level != "off" { crowdb_rpc_ffi::add_log_stderr(stderr_level); } - - let addr: SocketAddr = format!("{}:{}", args.bind, args.port).parse()?; - info!(%addr, "crowdb-web starting"); - - let listener = tokio::net::TcpListener::bind(addr).await?; - - // Load the persisted registry; absence yields an empty default. - // Mutating handlers (rack/node/server CRUD) write back to this path. - let path = if args.test_mode { - None - } else { - args.config - .or_else(crowdb_console_shared::TomlFileEngine::default_path) - }; - let cfg = match path.as_ref() { - Some(p) => { - let engine = crowdb_console_shared::TomlFileEngine::new(p.clone()); - crowdb_console_shared::ConsoleConfig::load_with_engine(&engine).unwrap_or_default() - } - None => crowdb_console_shared::ConsoleConfig::default(), - }; - let server_count = cfg.servers.len(); - let state = crowdb_web::AppState::with_config(cfg, path).with_test_mode(args.test_mode); - tracing::info!(servers = server_count, "loaded registry"); - if !args.skip_startup_restore { - crowdb_web::mgmt::startup_topology_check(&state).await; - } - - axum::serve(listener, crowdb_web::router(state)).await?; - Ok(()) + Ok(guards) } diff --git a/app/crowdb-web/src/managed.rs b/app/crowdb-web/src/managed.rs new file mode 100644 index 000000000..c13f5f04e --- /dev/null +++ b/app/crowdb-web/src/managed.rs @@ -0,0 +1,221 @@ +use std::collections::BTreeSet; +use std::path::PathBuf; +use std::time::Duration; + +use axum::extract::State; +use axum::http::StatusCode; +use axum::Json; +use crowdb_console_shared::config::web::WebMode; +use crowdb_kv_client::CrowdbSysmdClient; +use crowdb_monitor::{MonitorStatus, ServiceStatus, StatusStore}; +use crowdb_protocol::common::{ReplicaValue, StoreValue}; +use serde::Serialize; +use serde_json::{json, Value}; + +use crate::state::AppState; + +const SERVICE_TYPES: [(&str, &str); 5] = [ + ("kv-server", "kv"), + ("diskdb", "diskdb"), + ("diskio", "diskio"), + ("chunkdb", "chunkdb"), + ("chunk-kv", "chunk-kv"), +]; + +#[derive(Serialize)] +pub struct ManagedSnapshot { + source: &'static str, + racks: Vec, + nodes: Vec, + disk_groups: Vec, + disks: Vec, + stores: Vec, + groups: Vec, + replicas: Vec, + services: Vec, + monitor: Option, +} + +#[derive(Serialize)] +struct ServiceView { + kind: &'static str, + instance_id: String, + endpoint: String, + last_heartbeat_ms: u64, + monitor: Option, +} + +#[derive(Clone, Copy)] +enum SnapshotFailure { + Group0, + Monitor, +} + +impl SnapshotFailure { + fn reason(self) -> &'static str { + match self { + Self::Group0 => "group0_unavailable", + Self::Monitor => "monitor_unavailable", + } + } +} + +async fn monitor_status(path: PathBuf) -> Result { + tokio::task::spawn_blocking(move || { + StatusStore::open_file(&path) + .and_then(|store| store.read(Duration::from_secs(15))) + .map_err(|error| { + tracing::debug!(%error, "managed monitor status unavailable"); + SnapshotFailure::Monitor + }) + }) + .await + .map_err(|error| { + tracing::debug!(%error, "managed monitor status task failed"); + SnapshotFailure::Monitor + })? +} + +async fn validate_live_store_nodes( + sysmd: &CrowdbSysmdClient, + stores: &[StoreValue], + replicas: &[ReplicaValue], +) -> Result<(), crowdb_kv_client::Error> { + let mut live_nodes = BTreeSet::new(); + for (_, instance) in sysmd.read_all_kv_server_instances().await? { + let Some(node_id) = instance + .extra + .as_ref() + .and_then(|extra| extra.kv_server.as_ref()) + .and_then(|extra| extra.node_id) + else { + continue; + }; + if instance.rpc_endpoint.is_empty() || !live_nodes.insert(node_id) { + return Err(crowdb_kv_client::Error::Topology( + "live KV management registration is ambiguous".into(), + )); + } + } + if stores + .iter() + .flat_map(|store| &store.node_ids) + .chain(replicas.iter().map(|replica| &replica.node_id)) + .any(|node_id| !live_nodes.contains(node_id)) + { + return Err(crowdb_kv_client::Error::Topology( + "store node has no live KV management registration".into(), + )); + } + Ok(()) +} + +async fn load_snapshot(state: &AppState) -> Result { + let monitor = if state.web_mode == Some(WebMode::BareMetal) { + None + } else { + let path = state + .monitor_status_path + .as_ref() + .ok_or(SnapshotFailure::Monitor)?; + Some(monitor_status(path.as_ref().clone()).await?) + }; + if state.authority_seeds.is_empty() { + return Err(SnapshotFailure::Group0); + } + let client = state.kv_client().await; + let timeout = Duration::from_millis(state.authority_timeout_ms); + tokio::time::timeout(timeout, async { + client.refresh_topology().await?; + let sysmd = CrowdbSysmdClient::from_shared(client); + let racks = sysmd.list_racks().await?; + let nodes = sysmd.list_nodes().await?; + let disk_groups = sysmd.list_disk_groups().await?; + let disks = sysmd.list_all_disks().await?; + let stores = sysmd.list_stores().await?; + if racks.is_empty() || nodes.is_empty() || !stores.iter().any(|store| store.store_id == 0) { + return Err(crowdb_kv_client::Error::Topology("managed topology is incomplete".into())); + } + + let mut groups = Vec::new(); + let mut replicas = Vec::new(); + for store in &stores { + for group in sysmd.list_groups_in_store(store.store_id).await? { + replicas.extend(sysmd.list_replicas_in_group(store.store_id, group.group_id).await?); + groups.push(group); + } + } + validate_live_store_nodes(&sysmd, &stores, &replicas).await?; + let mut services = Vec::new(); + for (kind, monitor_id) in SERVICE_TYPES { + let instances = sysmd.read_service_instances(kind).await?; + let overlay = if instances.len() == 1 { + monitor.as_ref().and_then(|status| status.services.get(monitor_id)).cloned() + } else { + None + }; + services.extend(instances.into_iter().map(|(instance_id, record)| ServiceView { + kind, + instance_id: instance_id.to_string(), + endpoint: record.rpc_endpoint, + last_heartbeat_ms: record.last_heartbeat_ms, + monitor: overlay.clone(), + })); + } + Ok(ManagedSnapshot { + source: "group0", + racks: racks.into_iter().map(|(id, value)| json!({"id": id, "status": value.status, "node_ids": value.node_ids})).collect(), + nodes: nodes.into_iter().map(|(rack_id, id, value)| json!({"rack_id": rack_id, "id": id, "status": value.status, "disk_group_ids": value.disk_group_ids})).collect(), + disk_groups: disk_groups.into_iter().map(|group| json!(group)).collect(), + disks: disks.into_iter().map(|disk| json!({"rack_id": disk.rack_id, "node_id": disk.node_id, "disk_group_id": disk.disk_group_id, "disk_id": disk.disk_id, "value": disk.value})).collect(), + stores: stores.into_iter().map(|store| json!(store)).collect(), + groups: groups.into_iter().map(|group| json!(group)).collect(), + replicas: replicas.into_iter().map(|replica| json!(replica)).collect(), + services, + monitor, + }) + }) + .await + .map_err(|error| { + tracing::debug!(%error, "managed Group 0 snapshot timed out"); + SnapshotFailure::Group0 + })? + .map_err(|error| { + tracing::debug!(%error, "managed Group 0 snapshot failed"); + SnapshotFailure::Group0 + }) +} + +pub async fn authority(State(state): State) -> (StatusCode, Json) { + match load_snapshot(&state).await { + Ok(snapshot) => ( + StatusCode::OK, + Json( + json!({"source": "group0", "available": true, "monitor_revision": snapshot.monitor.map(|monitor| monitor.revision)}), + ), + ), + Err(error) => ( + StatusCode::SERVICE_UNAVAILABLE, + Json(json!({"source": "group0", "available": false, "reason": error.reason()})), + ), + } +} + +pub async fn snapshot( + State(state): State, +) -> Result, (StatusCode, Json)> { + match load_snapshot(&state).await { + Ok(snapshot) => Ok(Json(snapshot)), + Err(error) => { + let monitor = match state.monitor_status_path.as_ref() { + Some(path) => monitor_status(path.as_ref().clone()).await.ok(), + None => None, + }; + Err(( + StatusCode::SERVICE_UNAVAILABLE, + Json(json!({"source": "group0", "available": false, + "reason": error.reason(), "monitor": monitor})), + )) + } + } +} diff --git a/app/crowdb-web/src/managed_logical.rs b/app/crowdb-web/src/managed_logical.rs new file mode 100644 index 000000000..648d9bd37 --- /dev/null +++ b/app/crowdb-web/src/managed_logical.rs @@ -0,0 +1,243 @@ +use axum::extract::{Path, State}; +use axum::http::StatusCode; +use axum::Json; +use crowdb_console_shared::error::Error; +use crowdb_console_shared::ops; +use crowdb_protocol::common::{GroupValue, ReplicaValue, StoreValue}; +use serde::Deserialize; +use serde_json::{json, Value}; + +use crate::error::ErrorBody; +use crate::state::AppState; + +type ApiError = (StatusCode, Json); + +#[allow(clippy::needless_pass_by_value)] +fn api_error(error: Error) -> ApiError { + let status = match error { + Error::NotFound { .. } => StatusCode::NOT_FOUND, + Error::Conflict { .. } => StatusCode::CONFLICT, + Error::Validation { .. } => StatusCode::BAD_REQUEST, + _ => StatusCode::SERVICE_UNAVAILABLE, + }; + ( + status, + Json(ErrorBody { + error: error.to_string(), + }), + ) +} + +fn protected_record() -> ApiError { + ( + StatusCode::CONFLICT, + Json(ErrorBody { + error: "system store and group cannot be changed through the logical API".into(), + }), + ) +} + +#[derive(Deserialize)] +pub(crate) struct CreateStore { + store_id: u64, + #[serde(default)] + nodes: Vec, +} + +pub(crate) async fn list_stores(State(state): State) -> Result>, ApiError> { + let context = state.op_context().await.map_err(api_error)?; + ops::kv_logical::list_stores(&context) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn get_store( + State(state): State, + Path(store_id): Path, +) -> Result, ApiError> { + let context = state.op_context().await.map_err(api_error)?; + context + .sysmd() + .get_store(store_id) + .await + .map_err(|error| api_error(error.into()))? + .map(Json) + .ok_or_else(|| { + api_error(Error::NotFound { + kind: "store".into(), + id: store_id.to_string(), + }) + }) +} + +pub(crate) async fn add_store( + State(state): State, + Json(body): Json, +) -> Result<(StatusCode, Json), ApiError> { + if body.store_id == 0 { + return Err(protected_record()); + } + let context = state.op_context().await.map_err(api_error)?; + let nodes = ops::kv_logical::add_store(&context, body.store_id, &body.nodes) + .await + .map_err(api_error)?; + Ok(( + StatusCode::CREATED, + Json(json!({"store_id": body.store_id, "nodes": nodes})), + )) +} + +pub(crate) async fn remove_store( + State(state): State, + Path(store_id): Path, +) -> Result { + if store_id == 0 { + return Err(protected_record()); + } + let context = state.op_context().await.map_err(api_error)?; + ops::kv_logical::remove_store(&context, store_id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} + +#[derive(Deserialize)] +pub(crate) struct CreateGroup { + group_id: u64, + replica_id: u64, + nodes: Vec, +} + +pub(crate) async fn list_groups( + State(state): State, + Path(store_id): Path, +) -> Result>, ApiError> { + let context = state.op_context().await.map_err(api_error)?; + ops::kv_logical::list_groups(&context, store_id) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn get_group( + State(state): State, + Path((store_id, group_id)): Path<(u64, u64)>, +) -> Result, ApiError> { + let context = state.op_context().await.map_err(api_error)?; + context + .sysmd() + .get_group(store_id, group_id) + .await + .map_err(|error| api_error(error.into()))? + .map(Json) + .ok_or_else(|| { + api_error(Error::NotFound { + kind: "group".into(), + id: format!("{store_id}/{group_id}"), + }) + }) +} + +pub(crate) async fn add_group( + State(state): State, + Path(store_id): Path, + Json(body): Json, +) -> Result<(StatusCode, Json), ApiError> { + if store_id == 0 && body.group_id == 0 { + return Err(protected_record()); + } + let context = state.op_context().await.map_err(api_error)?; + ops::kv_logical::add_group(&context, store_id, body.group_id, body.replica_id, &body.nodes) + .await + .map_err(api_error)?; + Ok(( + StatusCode::CREATED, + Json(json!({"store_id": store_id, "group_id": body.group_id, "nodes": body.nodes})), + )) +} + +pub(crate) async fn remove_group( + State(state): State, + Path((store_id, group_id)): Path<(u64, u64)>, +) -> Result { + if store_id == 0 && group_id == 0 { + return Err(protected_record()); + } + let context = state.op_context().await.map_err(api_error)?; + ops::kv_logical::remove_group(&context, store_id, group_id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} + +#[derive(Deserialize)] +pub(crate) struct CreateReplica { + node_id: u64, + replica_id: Option, +} + +pub(crate) async fn list_replicas( + State(state): State, + Path((store_id, group_id)): Path<(u64, u64)>, +) -> Result>, ApiError> { + let context = state.op_context().await.map_err(api_error)?; + ops::kv_logical::list_replicas(&context, store_id, group_id) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn get_replica( + State(state): State, + Path((store_id, group_id, replica_id)): Path<(u64, u64, u64)>, +) -> Result, ApiError> { + let context = state.op_context().await.map_err(api_error)?; + context + .sysmd() + .get_replica(store_id, group_id, replica_id) + .await + .map_err(|error| api_error(error.into()))? + .map(Json) + .ok_or_else(|| { + api_error(Error::NotFound { + kind: "replica".into(), + id: format!("{store_id}/{group_id}/{replica_id}"), + }) + }) +} + +pub(crate) async fn add_replica( + State(state): State, + Path((store_id, group_id)): Path<(u64, u64)>, + Json(body): Json, +) -> Result<(StatusCode, Json), ApiError> { + if store_id == 0 && group_id == 0 { + return Err(protected_record()); + } + let context = state.op_context().await.map_err(api_error)?; + let replica_id = + ops::kv_logical::add_replica(&context, store_id, group_id, body.node_id, body.replica_id) + .await + .map_err(api_error)?; + Ok(( + StatusCode::CREATED, + Json( + json!({"store_id": store_id, "group_id": group_id, "replica_id": replica_id, "node_id": body.node_id}), + ), + )) +} + +pub(crate) async fn remove_replica( + State(state): State, + Path((store_id, group_id, replica_id)): Path<(u64, u64, u64)>, +) -> Result { + if store_id == 0 && group_id == 0 { + return Err(protected_record()); + } + let context = state.op_context().await.map_err(api_error)?; + ops::kv_logical::remove_replica(&context, store_id, group_id, replica_id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} diff --git a/app/crowdb-web/src/mgmt/group_ops.rs b/app/crowdb-web/src/mgmt/group_ops.rs index 9d6799dd4..ee73c4f44 100644 --- a/app/crowdb-web/src/mgmt/group_ops.rs +++ b/app/crowdb-web/src/mgmt/group_ops.rs @@ -2,36 +2,35 @@ // Licensed under the Apache License, Version 2.0. //! A6: Logical group plane — writes delegate to `ops::kv_logical`, -//! reads from the monitor cache (live role/leader info). +//! reads Group 0 topology with live role/leader overlays. -use crate::error::{err_502, map_config_err, map_persist_err, ErrorBody}; +use crate::error::{err_502, map_config_err, ErrorBody}; use crate::expand::Recursive; use crate::mgmt::{cluster_initialized, refresh_node_cache}; use crate::state::AppState; use axum::extract::{Path, State}; use axum::http::StatusCode; use axum::Json; -use crowdb_console_shared::cluster::{GroupSummary, GroupView, NodeId}; +use crowdb_console_shared::clients::http::ServerClient; +use crowdb_console_shared::cluster::{ + GroupHealth, GroupSummary, GroupView, NodeGroup, NodeId, ReplicaRole, ReplicaState, ReplicaView, +}; +use crowdb_console_shared::monitor::legacy_topology_to_node_stores; use crowdb_console_shared::ops; +use crowdb_protocol::common::ReplicaValue; use serde::Deserialize; +use std::collections::HashMap; -/// `GET /api/stores/:store_id/groups`. List groups from cache. +/// `GET /api/stores/:store_id/groups`. List Group 0 groups. /// /// # Errors -/// Returns `404` if the store is not found. +/// Returns `404` if the store is not found, or `502` if Group 0 is unavailable. pub(crate) async fn http_list_groups( State(state): State, Path(sid): Path, Recursive(_depth): Recursive, ) -> Result>, (StatusCode, Json)> { - let view = state.monitor_cache.resolve_store(sid).await.ok_or_else(|| { - ( - StatusCode::NOT_FOUND, - Json(ErrorBody { - error: format!("store {sid} not found"), - }), - ) - })?; + let view = super::store_ops::store_view(&state, sid).await?; Ok(Json(view.groups)) } @@ -73,7 +72,6 @@ pub(crate) async fn http_add_group( ops::kv_logical::add_group(&ctx, sid, body.group_id, body.replica_id, &body.nodes) .await .map_err(map_config_err)?; - state.commit_op_context(&ctx).map_err(map_persist_err)?; // Refresh the monitor cache for all target nodes so health badges // and RPC endpoint resolution reflect the new group. @@ -89,45 +87,145 @@ pub(crate) async fn http_add_group( )) } -/// `GET /api/stores/:store_id/groups/:group_id`. Aggregated group view -/// from cache. Refreshes all nodes hosting the store first so role / -/// leader info reflects the most recent election state. +/// `GET /api/stores/:store_id/groups/:group_id`. Group 0 membership with +/// observed per-replica runtime state. /// /// # Errors -/// Returns `404` if the group is not found. +/// Returns `404` if the group is not found, or `502` if Group 0 is unavailable. pub(crate) async fn http_get_group( State(state): State, Path((sid, gid)): Path<(u64, u64)>, Recursive(_depth): Recursive, ) -> Result, (StatusCode, Json)> { - // Refresh the cache for every node currently believed to host this - // store so role / leader info reflects the most recent topology. - let node_ids: Vec = { - let snap = state.monitor_cache.snapshot().await; - snap.iter() - .filter_map(|(nid, rec)| { - if rec.stores.contains_key(&sid) { - Some(*nid) - } else { - None - } - }) - .collect() - }; - futures::future::join_all(node_ids.iter().map(|&nid| refresh_node_cache(&state, nid))).await; - state - .monitor_cache - .resolve_group(sid, gid) + group_view(&state, sid, gid).await.map(Json) +} + +pub(super) async fn group_view( + state: &AppState, + sid: u64, + gid: u64, +) -> Result)> { + let ctx = state + .op_context() + .await + .map_err(|error| err_502(error.to_string()))?; + let group = ctx + .sysmd() + .get_group(sid, gid) .await - .map(Json) - .ok_or_else(|| { - ( - StatusCode::NOT_FOUND, - Json(ErrorBody { - error: format!("group {gid} in store {sid} not found"), - }), - ) - }) + .map_err(|error| err_502(format!("Group 0 group lookup failed: {error}")))?; + if group.is_none() { + return Err(( + StatusCode::NOT_FOUND, + Json(ErrorBody { + error: format!("group {gid} in store {sid} not found"), + }), + )); + } + let members = ctx + .sysmd() + .list_replicas_in_group(sid, gid) + .await + .map_err(|error| err_502(format!("Group 0 replica lookup failed: {error}")))?; + let instances = ctx + .sysmd() + .read_all_kv_server_instances() + .await + .map_err(|error| err_502(format!("Group 0 service lookup failed: {error}")))?; + let mut registered = HashMap::>::new(); + for (_, instance) in instances { + if let Some(node_id) = instance + .extra + .as_ref() + .and_then(|extra| extra.kv_server.as_ref()) + .and_then(|extra| extra.node_id) + { + registered.entry(node_id).or_default().push(instance.rpc_endpoint); + } + } + let reports = observe_replicas(sid, gid, &members, ®istered).await; + Ok(project_group(sid, gid, members, reports)) +} + +async fn observe_replicas( + sid: u64, + gid: u64, + members: &[ReplicaValue], + registered: &HashMap>, +) -> Vec> { + futures::future::join_all(members.iter().map(|member| async { + let endpoint = registered + .get(&member.node_id) + .filter(|endpoints| endpoints.len() == 1)?; + let client = ServerClient::new(&endpoint[0]).ok()?; + let stores = client.topology().await.ok()?; + let stores = legacy_topology_to_node_stores(member.node_id, &stores); + stores + .get(&sid)? + .groups + .iter() + .find(|group| group.group_id == gid && group.local.replica_id == member.replica_id) + .cloned() + })) + .await +} + +fn project_group( + sid: u64, + gid: u64, + members: Vec, + reports: Vec>, +) -> GroupView { + let mut replicas = Vec::with_capacity(members.len()); + let mut observed = 0usize; + let mut read_state = None; + let mut has_leader = false; + for (member, local) in members.into_iter().zip(reports) { + let replica = if let Some(local) = local { + observed += 1; + if local.local.role == ReplicaRole::Leader { + has_leader = true; + read_state = local.read_state; + } + ReplicaView { + replica_id: member.replica_id, + node_id: member.node_id, + role: local.local.role, + state: local.local.state, + engine_healthy: local.local.engine_healthy, + crowtree_stats: local.local.crowtree_stats, + election: local.local.election, + } + } else { + ReplicaView { + replica_id: member.replica_id, + node_id: member.node_id, + role: ReplicaRole::Unknown, + state: ReplicaState::Unknown, + engine_healthy: false, + crowtree_stats: None, + election: None, + } + }; + replicas.push(replica); + } + let total = replicas.len(); + let state = if observed == 0 { + GroupHealth::Unknown + } else if !has_leader || observed < total / 2 + 1 { + GroupHealth::Unavailable + } else if observed == total { + GroupHealth::Healthy + } else { + GroupHealth::Degraded + }; + GroupView { + store_id: sid, + group_id: gid, + replicas, + state, + read_state, + } } /// `DELETE /api/stores/:store_id/groups/:group_id`. Delete the group @@ -163,7 +261,6 @@ pub(crate) async fn http_remove_group( ops::kv_logical::remove_group(&ctx, sid, gid) .await .map_err(map_config_err)?; - state.commit_op_context(&ctx).map_err(map_persist_err)?; futures::future::join_all(hosting_nodes.iter().map(|&nid| refresh_node_cache(&state, nid))).await; Ok(StatusCode::NO_CONTENT) diff --git a/app/crowdb-web/src/mgmt/replica_ops.rs b/app/crowdb-web/src/mgmt/replica_ops.rs index e7857bd1d..c05c921fc 100644 --- a/app/crowdb-web/src/mgmt/replica_ops.rs +++ b/app/crowdb-web/src/mgmt/replica_ops.rs @@ -2,9 +2,9 @@ // Licensed under the Apache License, Version 2.0. //! A7: Logical replica plane — writes delegate to `ops::kv_logical`, -//! reads from the monitor cache (live role/leader info). +//! reads Group 0 membership with live role/leader overlays. -use crate::error::{err_502, map_config_err, map_persist_err, ErrorBody}; +use crate::error::{err_502, map_config_err, ErrorBody}; use crate::expand::Recursive; use crate::mgmt::refresh_node_cache; use crate::state::AppState; @@ -16,7 +16,7 @@ use crowdb_console_shared::ops; use serde::Deserialize; /// `GET /api/stores/:s/groups/:g/replicas`. Unified replica list from -/// the monitor cache. +/// Group 0 with runtime state from the monitor cache. /// /// # Errors /// Returns `404` if the group is not found. @@ -25,19 +25,12 @@ pub(crate) async fn http_list_replicas( Path((sid, gid)): Path<(u64, u64)>, Recursive(_depth): Recursive, ) -> Result>, (StatusCode, Json)> { - let view = state.monitor_cache.resolve_group(sid, gid).await.ok_or_else(|| { - ( - StatusCode::NOT_FOUND, - Json(ErrorBody { - error: format!("group {gid} in store {sid} not found"), - }), - ) - })?; + let view = super::group_ops::group_view(&state, sid, gid).await?; Ok(Json(view.replicas)) } /// `GET /api/stores/:s/groups/:g/replicas/:rid`. Single replica detail -/// (logical view) from the monitor cache. +/// (logical view) from Group 0 with runtime state overlay. /// /// # Errors /// Returns `404` if the group or replica is not found. @@ -46,14 +39,7 @@ pub(crate) async fn http_get_replica( Path((sid, gid, rid)): Path<(u64, u64, u64)>, Recursive(_depth): Recursive, ) -> Result, (StatusCode, Json)> { - let view = state.monitor_cache.resolve_group(sid, gid).await.ok_or_else(|| { - ( - StatusCode::NOT_FOUND, - Json(ErrorBody { - error: format!("group {gid} in store {sid} not found"), - }), - ) - })?; + let view = super::group_ops::group_view(&state, sid, gid).await?; let replica = view .replicas .iter() @@ -105,7 +91,6 @@ pub(crate) async fn http_add_replica( let new_rid = ops::kv_logical::add_replica(&ctx, sid, gid, body.node_id, body.replica_id) .await .map_err(map_config_err)?; - state.commit_op_context(&ctx).map_err(map_persist_err)?; // Refresh the monitor cache for the target node + all peers so // health badges and RPC endpoint resolution reflect the new replica. @@ -159,7 +144,6 @@ pub(crate) async fn http_remove_replica( ops::kv_logical::remove_replica(&ctx, sid, gid, rid) .await .map_err(map_config_err)?; - state.commit_op_context(&ctx).map_err(map_persist_err)?; let mut refresh_targets: Vec = peers.clone(); if let Some(target) = target_node { diff --git a/app/crowdb-web/src/mgmt/store_ops.rs b/app/crowdb-web/src/mgmt/store_ops.rs index f9cefea07..804574d95 100644 --- a/app/crowdb-web/src/mgmt/store_ops.rs +++ b/app/crowdb-web/src/mgmt/store_ops.rs @@ -2,9 +2,9 @@ // Licensed under the Apache License, Version 2.0. //! A5: Logical store plane — writes delegate to `ops::kv_logical`, -//! reads from the monitor cache (live role/leader info). +//! reads Group 0 topology with live leader hints from the monitor cache. -use crate::error::{err_502, map_config_err, map_persist_err, ErrorBody}; +use crate::error::{err_502, map_config_err, ErrorBody}; use crate::expand::Recursive; use crate::mgmt::{cluster_initialized, refresh_node_cache}; use crate::state::AppState; @@ -13,47 +13,94 @@ use axum::http::StatusCode; use axum::Json; use crowdb_console_shared::cluster::{GroupSummary, NodeId, StoreView}; use crowdb_console_shared::ops; +use crowdb_protocol::common::{GroupValue, ReplicaValue, StoreValue}; use serde::Deserialize; +use std::collections::{BTreeMap, HashMap}; -/// `GET /api/stores`. List stores aggregated from the monitor cache. +/// `GET /api/stores`. List Group 0 stores with runtime leader hints. /// -/// # Panics -/// Panics if the `RwLock` is poisoned (inside `snapshot()`). +/// # Errors +/// Returns `502` when Group 0 is unavailable. pub(crate) async fn http_list_stores( State(state): State, Recursive(_depth): Recursive, -) -> Json> { - let snap = state.monitor_cache.snapshot().await; - let mut seen: std::collections::BTreeMap = std::collections::BTreeMap::new(); - for (node_id, rec) in &snap { - for (sid, ns) in &rec.stores { - let entry = seen.entry(*sid).or_insert_with(|| StoreView { - store_id: *sid, - name: None, - nodes: Vec::new(), - groups: Vec::new(), +) -> Result>, (StatusCode, Json)> { + let ctx = state + .op_context() + .await + .map_err(|error| err_502(error.to_string()))?; + let (stores, groups, replicas) = tokio::try_join!( + ctx.sysmd().list_stores(), + ctx.sysmd().list_all_groups(), + ctx.sysmd().list_all_replicas() + ) + .map_err(|error| err_502(format!("Group 0 topology lookup failed: {error}")))?; + let mut groups_by_store = BTreeMap::>::new(); + let mut replicas_by_store = BTreeMap::>::new(); + for group in groups { + groups_by_store.entry(group.store_id).or_default().push(group); + } + for replica in replicas { + replicas_by_store + .entry(replica.store_id) + .or_default() + .push(replica); + } + let mut views = Vec::with_capacity(stores.len()); + for store in stores { + let groups = groups_by_store.remove(&store.store_id).unwrap_or_default(); + let replicas = replicas_by_store.remove(&store.store_id).unwrap_or_default(); + views.push(project_store(&state, store, groups, replicas).await); + } + views.sort_by_key(|store| store.store_id); + Ok(Json(views)) +} + +async fn project_store( + state: &AppState, + store: StoreValue, + groups: Vec, + replicas: Vec, +) -> StoreView { + let mut replicas_by_group = HashMap::>::new(); + for replica in replicas { + replicas_by_group + .entry(replica.group_id) + .or_default() + .push(replica); + } + let mut summaries = Vec::with_capacity(groups.len()); + for group in groups { + let members = replicas_by_group.remove(&group.group_id).unwrap_or_default(); + let leader = state + .monitor_cache + .resolve_group(store.store_id, group.group_id) + .await + .and_then(|view| { + view.leader().and_then(|leader| { + members + .iter() + .any(|member| { + member.replica_id == leader.replica_id && member.node_id == leader.node_id + }) + .then_some(leader.replica_id) + }) }); - entry.nodes.push(*node_id); - for g in &ns.groups { - if !entry.groups.iter().any(|gs| gs.group_id == g.group_id) { - entry.groups.push(GroupSummary { - group_id: g.group_id, - replica_count: 1, - leader: g.leader_hint, - }); - } else if let Some(gs) = entry.groups.iter_mut().find(|gs| gs.group_id == g.group_id) { - gs.replica_count += 1; - if gs.leader.is_none() { - gs.leader = g.leader_hint; - } - } - } - } + summaries.push(GroupSummary { + group_id: group.group_id, + replica_count: members.len(), + leader, + }); } - for entry in seen.values_mut() { - entry.groups.sort_by_key(|g| g.group_id); + summaries.sort_by_key(|group| group.group_id); + let mut nodes = store.node_ids; + nodes.sort_unstable(); + StoreView { + store_id: store.store_id, + name: None, + nodes, + groups: summaries, } - Json(seen.into_values().collect()) } #[derive(Debug, Deserialize)] @@ -88,7 +135,6 @@ pub(crate) async fn http_add_store( let succeeded = ops::kv_logical::add_store(&ctx, body.store_id, &body.nodes) .await .map_err(map_config_err)?; - state.commit_op_context(&ctx).map_err(map_persist_err)?; // Refresh the monitor cache for affected nodes so health badges // and RPC endpoint resolution reflect the new store. @@ -100,20 +146,31 @@ pub(crate) async fn http_add_store( )) } -/// `GET /api/stores/:store_id`. Aggregated store view from cache. +/// `GET /api/stores/:store_id`. Store view from Group 0, with runtime hints. /// /// # Errors -/// Returns `404` if the store is not found. +/// Returns `404` if the store is not found, or `502` if Group 0 is unavailable. pub(crate) async fn http_get_store( State(state): State, Path(sid): Path, Recursive(_depth): Recursive, ) -> Result, (StatusCode, Json)> { - state - .monitor_cache - .resolve_store(sid) + store_view(&state, sid).await.map(Json) +} + +pub(super) async fn store_view( + state: &AppState, + sid: u64, +) -> Result)> { + let ctx = state + .op_context() + .await + .map_err(|error| err_502(error.to_string()))?; + let store = ctx + .sysmd() + .get_store(sid) .await - .map(Json) + .map_err(|error| err_502(format!("Group 0 store lookup failed: {error}")))? .ok_or_else(|| { ( StatusCode::NOT_FOUND, @@ -121,7 +178,13 @@ pub(crate) async fn http_get_store( error: format!("store {sid} not found"), }), ) - }) + })?; + let (groups, replicas) = tokio::try_join!( + ctx.sysmd().list_groups_in_store(sid), + ctx.sysmd().list_replicas_in_store(sid) + ) + .map_err(|error| err_502(format!("Group 0 store topology lookup failed: {error}")))?; + Ok(project_store(state, store, groups, replicas).await) } /// `DELETE /api/stores/:store_id`. Delete the store across every hosting @@ -158,7 +221,6 @@ pub(crate) async fn http_remove_store( ops::kv_logical::remove_store(&ctx, sid) .await .map_err(map_config_err)?; - state.commit_op_context(&ctx).map_err(map_persist_err)?; futures::future::join_all(hosting_nodes.iter().map(|&nid| refresh_node_cache(&state, nid))).await; Ok(StatusCode::NO_CONTENT) diff --git a/app/crowdb-web/src/mgmt/topology.rs b/app/crowdb-web/src/mgmt/topology.rs index 3775fc93b..65f9522a9 100644 --- a/app/crowdb-web/src/mgmt/topology.rs +++ b/app/crowdb-web/src/mgmt/topology.rs @@ -4,14 +4,13 @@ //! Topology restore: startup three-way fallback + per-node restore. use crate::mgmt::{ - build_server_client, mgmt_url_for_node, port_of, refresh_node_cache, rpc_endpoint_for_node, - rpc_is_conflict, rpc_is_not_found, + build_server_client, mgmt_url_for_node, refresh_node_cache, rpc_endpoint_for_node, rpc_is_conflict, + rpc_is_not_found, }; use crate::state::AppState; use crowdb_console_shared::clients::http::ServerClient; use crowdb_console_shared::cluster::NodeId; -use crowdb_console_shared::config::{GroupEntry, NodeEntry, ServerEntry, ServiceType, StoreEntry}; -use crowdb_console_shared::lifecycle::{self, DeployRequest, DiskdbDeployRequest}; +use crowdb_console_shared::config::GroupEntry; use crowdb_console_shared::mgmt::{AddGroupInitialRole, AddGroupRequest, AddStoreRequest}; use tracing::{info, warn}; @@ -63,107 +62,14 @@ pub async fn startup_topology_check(state: &AppState) { info!("no nodes deployed; first-run scenario, skipping topology restore"); } Group0State::Missing => { - info!("group 0 not found on any node; TOML mode (phase 1)"); - restore_persisted_topology(state).await; + warn!("group 0 could not be confirmed; local topology restore is forbidden"); } Group0State::Ready => { - info!("group 0 is ready; loading topology from group 0 KV"); - restore_persisted_topology(state).await; + info!("group 0 is ready; local topology restore is skipped"); } } } -/// Restore persisted topology (servers, stores, groups, replicas) on startup. -/// -/// # Panics -/// Panics if the `RwLock` is poisoned. -pub(crate) async fn restore_persisted_topology(state: &AppState) { - let (nodes, servers, stores, groups) = { - let cfg = state.config.read().unwrap(); - ( - cfg.nodes.clone(), - cfg.servers.clone(), - cfg.stores.clone(), - cfg.groups.clone(), - ) - }; - for server in &servers { - let Some(node_id) = server.node_id else { - continue; - }; - let Some(node) = nodes.iter().find(|n| n.id == node_id) else { - warn!( - server_id = server.id, - node_id, "skipping restore for server with missing node" - ); - continue; - }; - let result = if server.service_type == ServiceType::Diskdb { - ensure_diskdb_running(state, node, server).await - } else { - ensure_server_running(state, node, server).await - }; - if let Err(err) = result { - warn!(server_id = server.id, node_id, error = %err, "failed to restore server process"); - } - } - for StoreEntry { store_id, nodes } in &stores { - for node_id in nodes { - if let Err(err) = ensure_store_on_node(state, *node_id, *store_id).await { - warn!(store_id, node_id, error = %err, "failed to restore store"); - } - } - } - for group in &groups { - let mut replicas = group.replicas.clone(); - replicas.sort_by_key(|r| r.replica_id); - // Defer the election driver for multi-replica groups until remotes are - // wired. - let start_election = Some(replicas.len() <= 1); - for (index, replica) in replicas.iter().enumerate() { - let initial_role = if index == 0 { - AddGroupInitialRole::Leader - } else { - AddGroupInitialRole::Follower - }; - if let Err(err) = ensure_group_local( - state, - replica.node_id, - group.store_id, - group.group_id, - replica.replica_id, - initial_role, - start_election, - ) - .await - { - warn!( - store_id = group.store_id, - group_id = group.group_id, - replica_id = replica.replica_id, - node_id = replica.node_id, - error = %err, - "failed to restore local group replica" - ); - } - } - if let Err(err) = ensure_group_remotes(state, group).await { - warn!(store_id = group.store_id, group_id = group.group_id, error = %err, "failed to restore group remotes"); - } - } - for server in &servers { - if let Some(node_id) = server.node_id { - refresh_node_cache(state, node_id).await; - } - } - info!( - servers = servers.len(), - stores = stores.len(), - groups = groups.len(), - "restore reconcile finished" - ); -} - /// Restores persisted topology (stores and groups) for a specific node. /// /// This function ensures that all stores and groups configured for the given node @@ -223,130 +129,6 @@ pub(crate) async fn restore_persisted_topology_for_node( Ok(()) } -async fn ensure_server_running( - state: &AppState, - node: &NodeEntry, - server: &ServerEntry, -) -> Result<(), String> { - let client = ServerClient::new(server.url.clone()).map_err(|e| e.to_string())?; - if client.health().await.is_ok() { - refresh_node_cache(state, node.id).await; - return Ok(()); - } - if !server.auto_start { - return Ok(()); - } - let rest_port = server - .rest_port - .ok_or_else(|| format!("server {} missing persisted rest_port", server.id))?; - let rpc_port = server - .rpc_port - .ok_or_else(|| format!("server {} missing persisted rpc_port", server.id))?; - let req = DeployRequest { - server_id: server.id.clone(), - rest_port, - rpc_port, - election_profile: server.election_profile.clone(), - binary: server.binary.clone().map(std::path::PathBuf::from), - ..Default::default() - }; - let deployed = if node.ssh_enabled() { - let server_bin = server.binary.clone().unwrap_or_else(|| { - std::env::var("CROWDB_KV_SERVER_BIN").unwrap_or_else(|_| "crowdb-kv-server".to_string()) - }); - crowdb_console_shared::ssh::deploy_via_ssh(&req, node, &server_bin) - .await - .map_err(|e| e.to_string())? - } else { - let workspace_dir = state - .prepare_node_workspace(node.id.to_string()) - .map_err(|e| e.to_string())?; - lifecycle::deploy_local_in_dir(&req, node, &workspace_dir) - .await - .map_err(|e| e.to_string())? - }; - state.set_runtime_pid(node.id, deployed.pid); - refresh_node_cache(state, node.id).await; - Ok(()) -} - -/// Restore a persisted `DiskDB` instance on startup. Mirrors -/// `ensure_server_running` but spawns `crowdb-diskdb` via -/// `deploy_diskdb_local` instead of the KV-server deploy path. -async fn ensure_diskdb_running( - state: &AppState, - node: &NodeEntry, - server: &ServerEntry, -) -> Result<(), String> { - // If the process is already alive, just refresh the cache. - if let Some(pid) = state.diskdb_runtime_pid(node.id) { - if lifecycle::process_is_alive(pid) { - refresh_node_cache(state, node.id).await; - return Ok(()); - } - state.clear_diskdb_runtime_pid(node.id); - } - if !server.auto_start { - return Ok(()); - } - let rpc_port = server - .rpc_port - .or_else(|| server.rpc_url.as_deref().and_then(port_of)) - .ok_or_else(|| format!("diskdb entry {} missing persisted rpc_port", server.id))?; - // Look up the kv-server management URL(s) on this node so the - // diskdb can discover group-0 after restart. - let kv_server_mgmt_seeds: Vec = { - let cfg = state.config.read().unwrap(); - cfg.servers - .iter() - .filter(|s| s.node_id == Some(node.id) && s.service_type == ServiceType::Kv) - .map(|s| s.url.clone()) - .collect() - }; - // Backward-compat: derive from the old paired-port scheme. - let listen_port = rpc_port; - let http_port = rpc_port.saturating_add(1); - let rpc_listen_port = rpc_port.saturating_add(2); - let req = DiskdbDeployRequest { - instance_id: None, - metrics_interval: None, - rpc_workers: None, - kv_connections: None, - kv_client_rpc_workers: None, - keepalive_interval_secs: state.test_mode.then_some(1), - free_batch_enabled: None, - free_flush_max_batch: None, - server_id: server.id.clone(), - listen_port, - http_port, - rpc_port: rpc_listen_port, - kv_server_mgmt_seeds, - }; - let workspace_dir = state - .prepare_node_workspace(node.id.to_string()) - .map_err(|e| e.to_string())?; - let deployed = lifecycle::deploy_diskdb_local(&req, node, &workspace_dir) - .await - .map_err(|e| e.to_string())?; - state.set_diskdb_runtime_pid(node.id, deployed.pid); - // Update the persisted entry with the fresh RPC endpoint. The HTTP - // readiness URL is intentionally lifecycle-local and is not persisted. - { - let mut cfg = state.config.write().unwrap(); - if let Some(entry) = cfg - .servers - .iter_mut() - .find(|s| s.node_id == Some(node.id) && s.service_type == ServiceType::Diskdb) - { - entry.url.clone_from(&deployed.endpoint); - entry.rpc_url = Some(deployed.endpoint.clone()); - } - } - state.persist().map_err(|e| e.to_string())?; - refresh_node_cache(state, node.id).await; - Ok(()) -} - async fn ensure_store_on_node(state: &AppState, node_id: NodeId, store_id: u64) -> Result<(), String> { let url = mgmt_url_for_node(state, node_id).map_err(|(_, body)| body.0.error.clone())?; let client = ServerClient::new(url).map_err(|e| e.to_string())?; diff --git a/app/crowdb-web/src/spa.rs b/app/crowdb-web/src/spa.rs index b8e3eea31..e49c959a9 100644 --- a/app/crowdb-web/src/spa.rs +++ b/app/crowdb-web/src/spa.rs @@ -1,8 +1,9 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crate::state::FRONTEND_DIST; +use crate::state::AppState; use axum::body::Body; +use axum::extract::State; use axum::http::{header, StatusCode, Uri}; use axum::response::{Html, IntoResponse, Response}; use std::path::{Path as StdPath, PathBuf}; @@ -16,8 +17,8 @@ use std::path::{Path as StdPath, PathBuf}; /// 3. Else, the build is missing — serve a static instructional page /// explaining how to run `make ui-build`. This keeps /// `cargo run` usable on machines without a Node toolchain. -pub async fn spa_fallback(uri: Uri) -> Response { - let dist = StdPath::new(FRONTEND_DIST); +pub async fn spa_fallback(State(state): State, uri: Uri) -> Response { + let dist = StdPath::new(state.ui_root.as_ref()); // Sanitize the request path: strip leading slash, refuse `..`. let req_path = uri.path().trim_start_matches('/'); diff --git a/app/crowdb-web/src/state.rs b/app/crowdb-web/src/state.rs index 1acba57f0..16c98abd0 100644 --- a/app/crowdb-web/src/state.rs +++ b/app/crowdb-web/src/state.rs @@ -5,7 +5,9 @@ use std::collections::HashMap; use std::path::PathBuf; use std::sync::{Arc, RwLock}; +use crowdb_console_shared::config::web::{LaunchRegistry, WebMode, WebProcessConfig}; use crowdb_console_shared::error::{Error, Result}; +use crowdb_console_shared::launch::LaunchRuntime; use crowdb_console_shared::monitor::MonitorCache; use crowdb_console_shared::ops::OpContext; use crowdb_console_shared::{ @@ -48,6 +50,14 @@ pub struct AppState { pub warn_dedup: Arc>>, /// Enables faster spawned-process intervals for E2E runs. pub test_mode: bool, + pub managed_mode: bool, + pub web_mode: Option, + pub ui_root: Arc, + pub authority_seeds: Arc>, + pub monitor_status_path: Option>, + pub authority_timeout_ms: u64, + pub(crate) management_token: Option>, + pub(crate) launch_registry_path: Option>, } impl Default for AppState { @@ -104,9 +114,74 @@ impl AppState { discovery_client: Arc::new(tokio::sync::RwLock::new(None)), warn_dedup: Arc::new(std::sync::Mutex::new(HashMap::new())), test_mode: false, + managed_mode: false, + web_mode: None, + ui_root: Arc::new(PathBuf::from(FRONTEND_DIST)), + authority_seeds: Arc::new(Vec::new()), + monitor_status_path: None, + authority_timeout_ms: 3_000, + management_token: None, + launch_registry_path: None, } } + #[must_use] + pub fn with_managed_ui(mut self, ui_root: PathBuf) -> Self { + self.managed_mode = true; + self.web_mode = Some(WebMode::Docker); + self.ui_root = Arc::new(ui_root); + self + } + + #[must_use] + pub fn with_process_config(mut self, config: &WebProcessConfig) -> Self { + self.managed_mode = true; + self.web_mode = Some(config.mode); + self.ui_root = Arc::new(config.ui_root.clone()); + self.authority_seeds = Arc::new(config.group0_management_seeds.clone()); + self.monitor_status_path = config.monitor_status.clone().map(Arc::new); + self.authority_timeout_ms = config.request_timeout_ms.unwrap_or(3_000); + self + } + + /// # Errors + /// Rejects launch policy outside bare-metal mode or invalid registry content. + pub fn with_launch_registry(mut self, path: PathBuf) -> Result { + if self.web_mode != Some(WebMode::BareMetal) { + return Err(Error::Config( + "only bare-metal Web accepts a launch registry".into(), + )); + } + LaunchRegistry::load(&path)?; + self.launch_registry_path = Some(Arc::new(std::fs::canonicalize(path)?)); + Ok(self) + } + + /// # Errors + /// Reports invalid policy or a failed configured service launch. + pub async fn start_configured_services(&self) -> Result { + let Some(path) = &self.launch_registry_path else { + return Ok(0); + }; + let registry = LaunchRegistry::load(path)?; + let runtime = LaunchRuntime::for_registry(path)?; + Ok(runtime.start_enabled(®istry).await?.len()) + } + + /// # Errors + /// Rejects a weak or malformed management credential. + pub fn with_management_token(mut self, token: String) -> std::result::Result { + if !(32..=256).contains(&token.len()) + || !token + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || b"-._~+/=".contains(&byte)) + { + return Err("management token is invalid"); + } + self.management_token = Some(Arc::from(token)); + Ok(self) + } + /// Enable or disable E2E test-mode behavior. #[must_use] pub fn with_test_mode(mut self, test_mode: bool) -> Self { @@ -319,7 +394,7 @@ impl AppState { } let transport = self.kv_rpc_transport().await; let c = Arc::new(crowdb_kv_client::CrowdbKvClient::new_with_rpc_transport( - crowdb_kv_client::ClientConfig::new(Vec::new()), + crowdb_kv_client::ClientConfig::new(self.authority_seeds.as_ref().clone()), transport, )); *guard = Some(Arc::clone(&c)); diff --git a/app/crowdb-web/tests/bare_metal_authority_test.rs b/app/crowdb-web/tests/bare_metal_authority_test.rs new file mode 100644 index 000000000..e7f7796a6 --- /dev/null +++ b/app/crowdb-web/tests/bare_metal_authority_test.rs @@ -0,0 +1,168 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use axum::body::Body; +use axum::http::{Request, StatusCode}; +use crowdb_console_shared::config::web::{WebMode, WebProcessConfig}; +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, CrowdbSysmdClient}; +use crowdb_protocol::common::{HwStatus, KvServerIdentity, NodeValue, RackValue, ReplicaValue}; +use crowdb_protocol::key::InstanceKey; +use crowdb_test_harness::cluster::KvCluster; +use crowdb_web::{router, AppState}; +use tower::ServiceExt; + +async fn snapshot(app: &axum::Router) -> (StatusCode, serde_json::Value) { + let response = app + .clone() + .oneshot( + Request::builder() + .uri("/api/preview") + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + let status = response.status(); + let body = axum::body::to_bytes(response.into_body(), 1024 * 1024) + .await + .unwrap(); + (status, serde_json::from_slice(&body).unwrap()) +} + +async fn register(sysmd: &CrowdbSysmdClient, node: u64, instance: u64, endpoint: &str) { + sysmd + .register_kv_server( + KvServerIdentity { + instance_id: instance, + node_id: Some(node), + }, + endpoint, + &[0], + &[], + "ok", + "/tmp/bare-metal-kv", + ) + .await + .unwrap(); +} + +async fn initialized_authority(cluster: &KvCluster) -> CrowdbSysmdClient { + let kv = CrowdbKvClient::new(ClientConfig::new(cluster.mgmt_endpoints.clone())); + kv.seed_leader(0, 0, cluster.group0_leader_endpoint.clone()); + let sysmd = CrowdbSysmdClient::new(kv); + sysmd + .add_rack( + 1, + &RackValue { + status: HwStatus::Up as i32, + node_ids: vec![1], + }, + ) + .await + .unwrap(); + sysmd + .add_node( + 1, + 1, + &NodeValue { + status: HwStatus::Up as i32, + ..Default::default() + }, + ) + .await + .unwrap(); + sysmd.add_store(0, &[1]).await.unwrap(); + register(&sysmd, 1, 7001, &cluster.mgmt_endpoints[0]).await; + sysmd +} + +fn application(cluster: &KvCluster) -> axum::Router { + let config = WebProcessConfig { + version: 1, + mode: WebMode::BareMetal, + bind: "127.0.0.1".into(), + port: 14000, + group0_management_seeds: cluster.mgmt_endpoints.clone(), + ui_root: "/tmp".into(), + monitor_status: None, + log_dir: "/tmp".into(), + log_max_file_mb: 30, + log_max_files: 5, + request_timeout_ms: Some(500), + }; + router(AppState::default().with_process_config(&config)) +} + +async fn unavailable(app: &axum::Router) { + let (code, body) = snapshot(app).await; + assert_eq!(code, StatusCode::SERVICE_UNAVAILABLE, "{body}"); + assert_eq!(body["reason"], "group0_unavailable"); + assert!(body.get("stores").is_none(), "stale topology: {body}"); + assert!(body["monitor"].is_null()); +} + +#[tokio::test] +async fn bare_metal_snapshot_requires_live_authority_without_a_docker_monitor() { + let cluster = KvCluster::start().await; + let sysmd = initialized_authority(&cluster).await; + let app = application(&cluster); + let (code, view) = snapshot(&app).await; + assert_eq!(code, StatusCode::OK, "{view}"); + assert_eq!(view["source"], "group0"); + assert_eq!(view["nodes"][0]["id"], 1); + assert!(view["monitor"].is_null()); + + register(&sysmd, 1, 7002, &cluster.mgmt_endpoints[0]).await; + unavailable(&app).await; + sysmd.unregister_service("kv-server", 7002).await.unwrap(); + sysmd.unregister_service("kv-server", 7001).await.unwrap(); + unavailable(&app).await; + register(&sysmd, 1, 7001, &cluster.mgmt_endpoints[0]).await; + + let (_, mut expired) = sysmd + .read_all_kv_server_instances() + .await + .unwrap() + .into_iter() + .find(|(id, _)| *id == 7001) + .unwrap(); + expired.last_heartbeat_ms = 1; + let key = InstanceKey { + service: "kv-server".into(), + instance_id: 7001, + } + .to_path(); + sysmd + .kv() + .put(0, 0, key.as_bytes(), &serde_json::to_vec(&expired).unwrap(), None) + .await + .unwrap(); + unavailable(&app).await; + register(&sysmd, 1, 7001, &cluster.mgmt_endpoints[0]).await; + assert_eq!(snapshot(&app).await.0, StatusCode::OK); + + drop(cluster); + unavailable(&app).await; +} + +#[tokio::test] +async fn snapshot_validates_later_replica_hosts_as_well_as_original_store_hosts() { + let cluster = KvCluster::start().await; + let sysmd = initialized_authority(&cluster).await; + let app = application(&cluster); + sysmd.add_group(0, 7).await.unwrap(); + sysmd + .add_replica(&ReplicaValue { + store_id: 0, + group_id: 7, + replica_id: 2, + node_id: 2, + voting: true, + ..Default::default() + }) + .await + .unwrap(); + unavailable(&app).await; + register(&sysmd, 2, 7002, &cluster.mgmt_endpoints[0]).await; + assert_eq!(snapshot(&app).await.0, StatusCode::OK); +} diff --git a/app/crowdb-web/tests/cluster_restart_incremental_test.rs b/app/crowdb-web/tests/cluster_restart_incremental_test.rs index 4dc9f8a15..e16897b5b 100644 --- a/app/crowdb-web/tests/cluster_restart_incremental_test.rs +++ b/app/crowdb-web/tests/cluster_restart_incremental_test.rs @@ -303,9 +303,11 @@ async fn wait_for_group_leader( timeout: Duration, ) -> Value { let deadline = Instant::now() + timeout; + let mut last_observation = String::new(); while Instant::now() < deadline { let (status, body) = json_get(client, &format!("{base}/api/stores/{store_id}/groups/{group_id}")).await; + last_observation = format!("{status}: {body}"); if status.is_success() { let replicas = body["replicas"].as_array().cloned().unwrap_or_default(); let leaders: Vec = replicas @@ -334,7 +336,7 @@ async fn wait_for_group_leader( } tokio::time::sleep(Duration::from_millis(10)).await; } - panic!("group {store_id}/{group_id} failed to converge to one leader within {timeout:?}"); + panic!("group {store_id}/{group_id} failed to converge to one leader within {timeout:?}; last observation: {last_observation}"); } async fn wait_for_store( @@ -345,8 +347,10 @@ async fn wait_for_store( timeout: Duration, ) { let deadline = Instant::now() + timeout; + let mut last_observation = String::new(); while Instant::now() < deadline { let (status, body) = json_get(client, &format!("{base}/api/stores/{store_id}")).await; + last_observation = format!("{status}: {body}"); if status.is_success() { let groups = body["groups"].as_array().cloned().unwrap_or_default(); if groups.len() == expected_groups { @@ -355,7 +359,7 @@ async fn wait_for_store( } tokio::time::sleep(Duration::from_millis(10)).await; } - panic!("store {store_id} failed to report {expected_groups} groups within {timeout:?}"); + panic!("store {store_id} failed to report {expected_groups} groups within {timeout:?}; last observation: {last_observation}"); } // ── WAL inspection helpers ───────────────────────────────────────────────── @@ -1000,7 +1004,8 @@ async fn restart_3node_1group() { n_puts: REPLAY_PUTS, deleted_keys: vec![1, 10, 20, 30, 40, 50], }], - "test", + // These are real processes; the paused-clock unit profile is not suitable. + "e2e", ) .await; } diff --git a/app/crowdb-web/tests/diskdb_auto_start_test.rs b/app/crowdb-web/tests/diskdb_auto_start_test.rs index f31c30f7a..3d7db0c9f 100644 --- a/app/crowdb-web/tests/diskdb_auto_start_test.rs +++ b/app/crowdb-web/tests/diskdb_auto_start_test.rs @@ -1,120 +1,49 @@ // Copyright 2026-present Gian -// Licensed under the Apache License, Version.0. - -//! Verifies that a persisted `DiskDB` entry with `auto_start: true` -//! is respawned by `startup_topology_check` after a console restart. -//! Regression: `restore_persisted_topology` only called -//! `ensure_server_running` (KV-only deploy path) for all servers, -//! so `DiskDB` entries were silently skipped on startup. - -use std::time::Duration; +// Licensed under the Apache License, Version 2.0. use crowdb_console_shared::config::{ConsoleConfig, NodeEntry, RackEntry, ServerEntry, ServiceType}; -use crowdb_console_shared::lifecycle::{crowdb_diskdb_bin, stop_pid_with_timeout}; -use crowdb_protocol::port::alloc as port_alloc; -use crowdb_protocol::ServicePort; use crowdb_web::mgmt::startup_topology_check; use crowdb_web::AppState; -struct PidGuard { - pids: Vec, -} - -impl Drop for PidGuard { - fn drop(&mut self) { - for pid in &self.pids { - let _ = stop_pid_with_timeout(*pid, Duration::from_secs(5)); - } - } -} - -/// Seed a config with one node + one `DiskDB` server entry (`auto_start`), -/// no KV server. `startup_topology_check` should detect `Missing` -/// (no reachable group-0) and call `restore_persisted_topology`, which -/// should spawn the diskdb process via `ensure_diskdb_running`. #[tokio::test] -async fn diskdb_auto_starts_on_console_restart() { - // Skip if crowdb-diskdb binary is not available. - if crowdb_diskdb_bin().is_none() { - eprintln!("skipping: crowdb-diskdb binary not found (set CROWDB_DISKDB_BIN)"); - return; - } - - let dir = crowdb_test_harness::test_dirs::test_data_dir().join(format!( - "crowdb-web-ddb-autostart-{}-{}", - std::process::id(), - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_nanos() - )); - std::fs::create_dir_all(&dir).unwrap(); - - let node_id: u64 = 7777; - // The auto-start path (`ensure_diskdb_running`) derives all three - // diskdb listener ports from a single base: listen = base, - // http = base+1, rpc = base+2. Allocate a consecutive range so all - // three are claimed and bind-probed free (individual allocations - // would leave base+1/base+2 unclaimed and vulnerable to collisions - // with leftover processes from prior runs). - let ports = port_alloc::alloc_test_port_range(ServicePort::DiskdbRpc, 3); - let http_port = ports[1]; - let rpc_port = ports[0]; - - let mut cfg = ConsoleConfig::default(); - cfg.add_rack(RackEntry { - id: 1, - name: "test-rack".into(), - }) - .unwrap(); - cfg.add_node(NodeEntry { - id: node_id, - rack_id: 1, - host: "127.0.0.1".into(), - ssh_port: 22, - ssh_user: String::new(), - ssh_key: None, - ssh_password: None, - }) - .unwrap(); - - let rpc_url = format!("http://127.0.0.1:{rpc_port}"); - cfg.add_server(ServerEntry { - id: format!("diskdb-{node_id}"), - url: format!("http://127.0.0.1:{http_port}"), - node_id: Some(node_id), - rpc_url: Some(rpc_url), - rest_port: None, - rpc_port: Some(rpc_port), - auto_start: true, - binary: None, - election_profile: None, - pid: None, - service_type: ServiceType::Diskdb, - rpc_workers: None, - no_fsync: false, - }) - .unwrap(); - - let cfg_path = dir.join("console.toml"); - let state = AppState::with_config(cfg, Some(cfg_path)); - - // Run the startup restore — should spawn the diskdb process. +async fn startup_does_not_replay_local_diskdb_launch_policy_without_group0() { + let mut config = ConsoleConfig::default(); + config + .add_rack(RackEntry { + id: 1, + name: "test-rack".into(), + }) + .unwrap(); + config + .add_node(NodeEntry { + id: 7777, + rack_id: 1, + host: "127.0.0.1".into(), + ssh_port: 22, + ssh_user: String::new(), + ssh_key: None, + ssh_password: None, + }) + .unwrap(); + config + .add_server(ServerEntry { + id: "diskdb-7777".into(), + url: "http://127.0.0.1:1".into(), + node_id: Some(7777), + rpc_url: None, + rest_port: None, + rpc_port: Some(1), + auto_start: true, + binary: None, + election_profile: None, + pid: None, + service_type: ServiceType::Diskdb, + rpc_workers: None, + no_fsync: false, + }) + .unwrap(); + + let state = AppState::with_config(config, None); startup_topology_check(&state).await; - - // The diskdb PID should now be tracked. - let pid = state - .diskdb_runtime_pid(node_id) - .expect("diskdb was not auto-started by startup_topology_check"); - - let _guard = PidGuard { pids: vec![pid] }; - - // Verify the process is actually alive. - assert!( - crowdb_console_shared::lifecycle::process_is_alive(pid), - "diskdb pid {pid} is not alive after auto-start" - ); - - // Clean up the workspace dir. - let _ = std::fs::remove_dir_all(&dir); + assert_eq!(state.diskdb_runtime_pid(7777), None); } diff --git a/app/crowdb-web/tests/kv_routes_test.rs b/app/crowdb-web/tests/kv_routes_test.rs index 3bbad1f5e..3fb2c45a1 100644 --- a/app/crowdb-web/tests/kv_routes_test.rs +++ b/app/crowdb-web/tests/kv_routes_test.rs @@ -10,7 +10,9 @@ use std::time::Duration; use crowdb_console_shared::clients::http::ServerClient; use crowdb_console_shared::cluster::NodeHealth; -use crowdb_console_shared::config::{NodeEntry, RackEntry, ServerEntry, ServiceType}; +use crowdb_console_shared::config::{ + GroupEntry, NodeEntry, RackEntry, ReplicaEntry, ServerEntry, ServiceType, +}; use crowdb_console_shared::lifecycle::{self, crowdb_kv_server_bin, DeployRequest}; use crowdb_console_shared::monitor::{legacy_topology_to_node_stores, NodeRecord}; use crowdb_console_shared::ConsoleConfig; @@ -48,7 +50,7 @@ async fn spawn_upstream() -> Option { ssh_password: None, }; let req = DeployRequest { - server_id: "n1".to_string(), + server_id: "1".to_string(), rest_port: pick_free_port(), rpc_port: pick_free_port(), election_profile: Some("e2e".into()), @@ -158,6 +160,18 @@ async fn kv_put_get_delete_through_web_routes() { .expect("add_group"); assert_eq!(group_resp.status(), 201, "add_group failed"); + let endpoint = http + .get(format!("{base}/api/stores/1/groups/1/endpoint")) + .send() + .await + .unwrap(); + assert_eq!( + endpoint.status(), + 200, + "endpoint: {:?}", + endpoint.text().await.ok() + ); + let url = format!("{base}/api/stores/1/groups/1/kv"); // PUT @@ -254,10 +268,18 @@ async fn kv_get_returns_502_when_leader_unreachable() { no_fsync: false, }) .unwrap(); + cfg.groups.push(GroupEntry { + store_id: 7, + group_id: 70, + replicas: vec![ReplicaEntry { + replica_id: 1, + node_id: 1, + }], + }); let state = AppState::with_config(cfg, None); - // Seed a fake group on n1 with a leader hint, so resolve_kv_endpoint - // returns Ok(rpc_url) and the handler proceeds to connect. + // Even with a locally persisted group and a cached leader, the KV + // request must not use either when Group 0 is unavailable. let mut stores = BTreeMap::new(); stores.insert( 7, @@ -306,4 +328,10 @@ async fn kv_get_returns_502_when_leader_unreachable() { "expected 502 when leader crowdb-rpc port is dead, got {}", resp.status() ); + let endpoint = http + .get(format!("http://{web}/api/stores/7/groups/70/endpoint")) + .send() + .await + .unwrap(); + assert_eq!(endpoint.status(), 502); } diff --git a/app/crowdb-web/tests/launch_registry_test.rs b/app/crowdb-web/tests/launch_registry_test.rs new file mode 100644 index 000000000..c711ebe9f --- /dev/null +++ b/app/crowdb-web/tests/launch_registry_test.rs @@ -0,0 +1,189 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +#![cfg(target_os = "linux")] + +use std::os::unix::fs::PermissionsExt; +use std::process::{Child, Command, Stdio}; +use std::time::Duration; + +use crowdb_console_shared::config::web::{LaunchRecord, LaunchRegistry}; +use crowdb_console_shared::launch::LaunchRuntime; +use crowdb_console_shared::lifecycle; +use crowdb_test_harness::test_dirs::tempdir_in_test_data; +use reqwest::StatusCode; + +struct TestProcesses { + web: Child, + services: Vec, +} +impl Drop for TestProcesses { + fn drop(&mut self) { + for pid in &self.services { + if lifecycle::process_is_alive(*pid) { + let _ = lifecycle::stop_pid_with_timeout(*pid, Duration::from_secs(2)); + } + } + let _ = self.web.kill(); + let _ = self.web.wait(); + } +} + +async fn start_web( + directory: &std::path::Path, + registry_path: &std::path::Path, +) -> (TestProcesses, String, String, reqwest::Client) { + let port = crowdb_protocol::port::alloc::alloc_test_port(crowdb_protocol::ServicePort::Web); + let config = directory.join("web.toml"); + std::fs::write(&config, format!( + "version = 1\nmode = 'bare-metal'\nbind = '127.0.0.1'\nport = {port}\ngroup0_management_seeds = ['http://127.0.0.1:9']\nui_root = {}\nlog_dir = {}\nlog_max_file_mb = 30\nlog_max_files = 5\nrequest_timeout_ms = 200\n", + serde_json::to_string(directory).unwrap(), serde_json::to_string(&directory.join("log")).unwrap())).unwrap(); + let token = "t".repeat(40); + let output = std::fs::File::create(directory.join("web-output.log")).unwrap(); + let web = Command::new(env!("CARGO_BIN_EXE_crowdb-web")) + .arg("--config") + .arg(&config) + .arg("--registry") + .arg(registry_path) + .env("CROWDB_ICEBERG_MANAGE_TOKEN", &token) + .stdout(Stdio::from(output.try_clone().unwrap())) + .stderr(Stdio::from(output)) + .spawn() + .unwrap(); + let guard = TestProcesses { + web, + services: Vec::new(), + }; + let base = format!("http://127.0.0.1:{port}"); + let http = reqwest::Client::builder() + .timeout(Duration::from_secs(3)) + .build() + .unwrap(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + loop { + if http + .get(format!("{base}/healthz")) + .send() + .await + .is_ok_and(|reply| reply.status().is_success()) + { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "Web startup failed: {}", + std::fs::read_to_string(directory.join("web-output.log")).unwrap() + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + (guard, base, token, http) +} + +#[tokio::test] +async fn bare_metal_web_loads_launch_policy_without_restoring_topology() { + let dir = tempdir_in_test_data("web-launch-registry"); + let binary = dir.path().join("service"); + std::fs::write(&binary, "#!/bin/sh\nexec sleep 60\n").unwrap(); + std::fs::set_permissions(&binary, std::fs::Permissions::from_mode(0o700)).unwrap(); + let service_config = dir.path().join("service.toml"); + std::fs::write(&service_config, "").unwrap(); + let record = LaunchRecord { + node_id: 701, + service_id: "kv".into(), + host: "localhost".into(), + ssh_credential_ref: None, + ssh_user: None, + ssh_port: 22, + binary_path: binary, + service_config_path: service_config, + workspace: dir.path().to_owned(), + auto_start: true, + args: Vec::new(), + readiness_url: None, + }; + let mut registry = LaunchRegistry { + version: 1, + launches: vec![record.clone()], + }; + let registry_path = dir.path().join("launches.toml"); + registry.save(®istry_path).unwrap(); + let (mut guard, base, token, http) = start_web(dir.path(), ®istry_path).await; + let runtime = LaunchRuntime::for_registry(®istry_path).unwrap(); + let first = runtime + .status(&record) + .await + .unwrap() + .expect("auto-start process"); + guard.services.push(first.pid); + assert_eq!( + http.get(format!("{base}/api/launches")) + .send() + .await + .unwrap() + .status(), + StatusCode::UNAUTHORIZED + ); + let view: serde_json::Value = http + .get(format!("{base}/api/launches")) + .bearer_auth(&token) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + assert_eq!(view[0]["process"]["pid"], first.pid); + assert_eq!( + http.get(format!("{base}/api/stores")) + .send() + .await + .unwrap() + .status(), + StatusCode::SERVICE_UNAVAILABLE + ); + let restarted: serde_json::Value = http + .post(format!("{base}/api/launches/701/kv/restart")) + .bearer_auth(&token) + .send() + .await + .unwrap() + .error_for_status() + .unwrap() + .json() + .await + .unwrap(); + let second = u32::try_from(restarted["pid"].as_u64().unwrap()).unwrap(); + guard.services.push(second); + assert_ne!(second, first.pid); + assert_eq!( + http.post(format!("{base}/api/launches/701/kv/stop")) + .bearer_auth(&token) + .send() + .await + .unwrap() + .status(), + StatusCode::NO_CONTENT + ); + assert!(runtime.status(&record).await.unwrap().is_none()); + registry.launches[0].auto_start = false; + registry.save(®istry_path).unwrap(); + let updated: serde_json::Value = http + .get(format!("{base}/api/launches")) + .bearer_auth(&token) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + assert_eq!(updated[0]["auto_start"], false); + assert!(!std::fs::read_to_string(registry_path).unwrap().contains("pid")); +} + +#[test] +fn docker_web_refuses_launch_policy() { + let state = crowdb_web::AppState::default().with_managed_ui("/tmp/ui".into()); + assert!(state + .with_launch_registry("/tmp/unused-registry.toml".into()) + .is_err()); +} diff --git a/app/crowdb-web/tests/managed_mode_test.rs b/app/crowdb-web/tests/managed_mode_test.rs new file mode 100644 index 000000000..a70b7a68a --- /dev/null +++ b/app/crowdb-web/tests/managed_mode_test.rs @@ -0,0 +1,429 @@ +use std::path::PathBuf; + +use axum::body::Body; +use axum::http::{Method, Request, StatusCode}; +use crowdb_console_shared::config::web::{WebMode, WebProcessConfig}; +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, CrowdbSysmdClient}; +use crowdb_monitor::{MonitorPhase, MonitorStatus, ServiceStatus, StatusStore}; +use crowdb_protocol::common::{HwStatus, KvServerIdentity, NodeValue, RackValue, ServiceExtra}; +use crowdb_web::{router, AppState}; +use tower::ServiceExt; +use uuid::Uuid; + +async fn get_json(app: axum::Router, path: &str) -> (StatusCode, serde_json::Value) { + let response = app + .oneshot(Request::builder().uri(path).body(Body::empty()).unwrap()) + .await + .unwrap(); + let status = response.status(); + let body = axum::body::to_bytes(response.into_body(), 1024 * 1024) + .await + .unwrap(); + (status, serde_json::from_slice(&body).unwrap()) +} + +async fn register_kv_node(sysmd: &CrowdbSysmdClient, endpoint: &str, hosted_stores: &[u64]) { + sysmd + .register_kv_server( + KvServerIdentity { + instance_id: 9, + node_id: Some(1), + }, + endpoint, + hosted_stores, + &[], + "ok", + "/tmp/managed-test-kv", + ) + .await + .unwrap(); +} + +async fn verify_managed_store_lifecycle( + app: &axum::Router, + sysmd: &CrowdbSysmdClient, + endpoint: &str, + token: &str, +) { + let response = app + .clone() + .oneshot( + Request::builder() + .method(Method::POST) + .uri("/api/stores") + .header("authorization", format!("Bearer {token}")) + .header("content-type", "application/json") + .body(Body::from(r#"{"store_id":7,"nodes":[1]}"#)) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), StatusCode::CREATED); + let (code, stores) = get_json(app.clone(), "/api/stores").await; + assert_eq!(code, StatusCode::OK); + assert!(stores + .as_array() + .unwrap() + .iter() + .any(|store| store["store_id"] == 7)); + + register_kv_node(sysmd, "http://127.0.0.1:1", &[0, 7]).await; + let response = app + .clone() + .oneshot( + Request::builder() + .method(Method::DELETE) + .uri("/api/stores/7") + .header("authorization", format!("Bearer {token}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), StatusCode::SERVICE_UNAVAILABLE); + assert!(sysmd.get_store(7).await.unwrap().is_some()); + + register_kv_node(sysmd, endpoint, &[0, 7]).await; + let response = app + .clone() + .oneshot( + Request::builder() + .method(Method::DELETE) + .uri("/api/stores/7") + .header("authorization", format!("Bearer {token}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), StatusCode::NO_CONTENT); + let (code, stores) = get_json(app.clone(), "/api/stores").await; + assert_eq!(code, StatusCode::OK); + assert!(!stores + .as_array() + .unwrap() + .iter() + .any(|store| store["store_id"] == 7)); + let response = app + .clone() + .oneshot( + Request::builder() + .method(Method::POST) + .uri("/api/stores") + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), StatusCode::UNAUTHORIZED); +} + +#[tokio::test] +async fn docker_management_uses_existing_bearer_without_unlocking_hardware() { + let token = "m".repeat(64); + let app = router( + AppState::default() + .with_managed_ui(PathBuf::from("/tmp/crowdb-ui")) + .with_management_token(token.clone()) + .unwrap(), + ); + for (authorization, expected) in [ + (None, StatusCode::UNAUTHORIZED), + (Some("Bearer wrong".to_owned()), StatusCode::UNAUTHORIZED), + (Some(format!("Bearer {token}")), StatusCode::NO_CONTENT), + ] { + let mut request = Request::builder() + .method(Method::POST) + .uri("/api/management/check"); + if let Some(value) = authorization { + request = request.header("authorization", value); + } + let response = app + .clone() + .oneshot(request.body(Body::empty()).unwrap()) + .await + .unwrap(); + assert_eq!(response.status(), expected); + } + let response = app + .clone() + .oneshot( + Request::builder() + .method(Method::POST) + .uri("/api/racks") + .header("authorization", format!("Bearer {token}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), StatusCode::SERVICE_UNAVAILABLE); + + for (authorization, expected) in [ + (None, StatusCode::UNAUTHORIZED), + (Some("Bearer wrong".to_owned()), StatusCode::UNAUTHORIZED), + (Some(format!("Bearer {token}")), StatusCode::CONFLICT), + ] { + let mut request = Request::builder().method(Method::POST).uri("/api/stores"); + if let Some(value) = authorization { + request = request.header("authorization", value); + } + let response = app + .clone() + .oneshot( + request + .header("content-type", "application/json") + .body(Body::from(r#"{"store_id":0}"#)) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), expected); + } +} + +#[tokio::test] +async fn managed_mode_does_not_expose_local_topology_or_mutations() { + let app = router(AppState::default().with_managed_ui(PathBuf::from("/tmp/crowdb-ui"))); + for (method, path) in [ + (Method::GET, "/api/racks"), + (Method::GET, "/api/stores"), + (Method::POST, "/api/racks"), + (Method::DELETE, "/api/nodes/1"), + ] { + let response = app + .clone() + .oneshot( + Request::builder() + .method(method) + .uri(path) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), StatusCode::SERVICE_UNAVAILABLE, "{path}"); + } + let response = app + .oneshot( + Request::builder() + .uri("/api/authority") + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), StatusCode::SERVICE_UNAVAILABLE); + let body = axum::body::to_bytes(response.into_body(), 16 * 1024) + .await + .unwrap(); + let status: serde_json::Value = serde_json::from_slice(&body).unwrap(); + assert_eq!(status["source"], "group0"); + assert_eq!(status["available"], false); + assert_eq!(status["reason"], "monitor_unavailable"); +} + +#[tokio::test] +async fn managed_snapshot_uses_group0_and_monitor_without_local_fallback() { + if crowdb_test_harness::cluster::crowdb_kv_server_bin().is_none() { + eprintln!("skipping real Group 0 test: crowdb-kv-server is unavailable"); + return; + } + let cluster = crowdb_test_harness::cluster::KvCluster::start().await; + let kv = CrowdbKvClient::new(ClientConfig::new(cluster.mgmt_endpoints.clone())); + kv.seed_leader(0, 0, cluster.group0_leader_endpoint.clone()); + let sysmd = CrowdbSysmdClient::new(kv); + sysmd + .add_rack( + 1, + &RackValue { + status: HwStatus::Up as i32, + node_ids: vec![1], + }, + ) + .await + .unwrap(); + sysmd + .add_node( + 1, + 1, + &NodeValue { + status: HwStatus::Up as i32, + ..Default::default() + }, + ) + .await + .unwrap(); + sysmd.add_store(0, &[1]).await.unwrap(); + sysmd + .register_service("diskio", 7, &cluster.mgmt_endpoints[0], &ServiceExtra::default()) + .await + .unwrap(); + register_kv_node(&sysmd, &cluster.mgmt_endpoints[0], &[0]).await; + + let run_root = + crowdb_test_harness::test_dirs::test_data_dir().join(format!("managed-web-{}", Uuid::new_v4())); + std::fs::create_dir_all(&run_root).unwrap(); + let store = StatusStore::new(&run_root).unwrap(); + let mut status = MonitorStatus::new(Uuid::new_v4(), MonitorPhase::Initializing); + status.services.insert( + "diskio".into(), + ServiceStatus { + pid: Some(123), + generation: 2, + healthy: true, + restart_attempts: 1, + }, + ); + store.publish(&mut status).unwrap(); + let config = WebProcessConfig { + version: 1, + mode: WebMode::Docker, + bind: "127.0.0.1".into(), + port: 8080, + group0_management_seeds: cluster.mgmt_endpoints.clone(), + ui_root: run_root.clone(), + monitor_status: Some(run_root.join("status/monitor.json")), + log_dir: run_root.join("log"), + log_max_file_mb: 30, + log_max_files: 5, + request_timeout_ms: Some(5_000), + }; + let token = "m".repeat(64); + let app = router( + AppState::default() + .with_process_config(&config) + .with_management_token(token.clone()) + .unwrap(), + ); + sysmd.unregister_service("kv-server", 9).await.unwrap(); + let (code, unavailable) = get_json(app.clone(), "/api/authority").await; + assert_eq!(code, StatusCode::SERVICE_UNAVAILABLE, "{unavailable}"); + assert_eq!(unavailable["reason"], "group0_unavailable"); + register_kv_node(&sysmd, &cluster.mgmt_endpoints[0], &[0]).await; + let (code, authority) = get_json(app.clone(), "/api/authority").await; + assert_eq!(code, StatusCode::OK, "{authority}"); + assert_eq!(authority["source"], "group0"); + assert_eq!(authority["available"], true); + let (code, snapshot) = get_json(app.clone(), "/api/preview").await; + assert_eq!(code, StatusCode::OK, "{snapshot}"); + assert_eq!(snapshot["racks"][0]["id"], 1); + assert_eq!(snapshot["nodes"][0]["id"], 1); + assert_eq!(snapshot["stores"][0]["store_id"], 0); + let diskio = snapshot["services"] + .as_array() + .unwrap() + .iter() + .find(|service| service["kind"] == "diskio") + .unwrap(); + assert_eq!(diskio["monitor"]["pid"], 123, "{snapshot}"); + assert_eq!(diskio["monitor"]["generation"], 2); + verify_managed_store_lifecycle(&app, &sysmd, &cluster.mgmt_endpoints[0], &token).await; + + drop(cluster); + verify_unavailable_snapshot(app, &store, &mut status, &run_root).await; + std::fs::remove_dir_all(run_root).unwrap(); +} + +async fn verify_unavailable_snapshot( + app: axum::Router, + store: &StatusStore, + status: &mut MonitorStatus, + run_root: &std::path::Path, +) { + store.publish(status).unwrap(); + let (code, unavailable) = get_json(app.clone(), "/api/preview").await; + assert_eq!(code, StatusCode::SERVICE_UNAVAILABLE, "{unavailable}"); + assert_eq!(unavailable["reason"], "group0_unavailable"); + assert_eq!(unavailable["monitor"]["services"]["diskio"]["pid"], 123); + assert!(unavailable.get("stores").is_none()); + std::fs::remove_file(run_root.join("status/monitor.json")).unwrap(); + let (code, unavailable) = get_json(app, "/api/preview").await; + assert_eq!(code, StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(unavailable["reason"], "monitor_unavailable"); + assert!(unavailable["monitor"].is_null()); +} + +#[test] +fn old_mixed_config_is_rejected_before_startup() { + let path = std::env::temp_dir().join(format!( + "crowdb-web-old-config-{}-{}", + std::process::id(), + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_nanos() + )); + std::fs::write(&path, "[[rack]]\nid = 1\n").unwrap(); + let output = std::process::Command::new(env!("CARGO_BIN_EXE_crowdb-web")) + .args(["--config", path.to_str().unwrap()]) + .output() + .unwrap(); + std::fs::remove_file(path).unwrap(); + assert!(!output.status.success()); + let error = String::from_utf8_lossy(&output.stderr); + assert!( + error.contains("version") || error.contains("unknown field"), + "{error}" + ); +} + +#[test] +fn malformed_bare_metal_registry_fails_before_listener_bind() { + let root = std::env::temp_dir().join(format!( + "crowdb-web-malformed-registry-{}-{}", + std::process::id(), + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_nanos() + )); + let registry_dir = root.join("persistent/console"); + std::fs::create_dir_all(®istry_dir).unwrap(); + std::fs::write(registry_dir.join("crowdb-kv.db.toml"), "[[rack]\n").unwrap(); + let output = std::process::Command::new(env!("CARGO_BIN_EXE_crowdb-web")) + .args(["--bind", "127.0.0.1", "--port", "14000"]) + .env("CROWDB_RUNTIME_ROOT", &root) + .output() + .unwrap(); + std::fs::remove_dir_all(root).unwrap(); + assert!(!output.status.success()); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!( + stderr.contains("Error: Config(") && stderr.contains("invalid table header"), + "{stderr}" + ); +} + +#[test] +fn managed_process_rejects_standalone_registry() { + let directory = std::env::temp_dir().join(format!( + "crowdb-web-registry-{}-{}", + std::process::id(), + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_nanos() + )); + std::fs::create_dir(&directory).unwrap(); + let config = directory.join("crowdb-web.toml"); + let registry = directory.join("registry.toml"); + std::fs::write( + &config, + "version = 1\nmode = 'docker'\nbind = '127.0.0.1'\nport = 14000\ngroup0_management_seeds = ['http://127.0.0.1:10000']\nui_root = '/tmp'\nmonitor_status = '/tmp/monitor.json'\nlog_dir = '/tmp'\nlog_max_file_mb = 30\nlog_max_files = 5\n", + ) + .unwrap(); + std::fs::write(®istry, "version = 1\n").unwrap(); + let output = std::process::Command::new(env!("CARGO_BIN_EXE_crowdb-web")) + .args([ + "--config", + config.to_str().unwrap(), + "--registry", + registry.to_str().unwrap(), + ]) + .output() + .unwrap(); + std::fs::remove_dir_all(directory).unwrap(); + assert!(!output.status.success()); + assert!(String::from_utf8_lossy(&output.stderr).contains("docker web does not accept --registry")); +} diff --git a/app/crowdb-web/tests/metrics_proxy_test.rs b/app/crowdb-web/tests/metrics_proxy_test.rs index 22722ad01..caec2564c 100644 --- a/app/crowdb-web/tests/metrics_proxy_test.rs +++ b/app/crowdb-web/tests/metrics_proxy_test.rs @@ -52,7 +52,7 @@ async fn spawn_upstream() -> Option { ssh_password: None, }; let req = DeployRequest { - server_id: "n1".into(), + server_id: "1".into(), rest_port: pick_free_port(), rpc_port: pick_free_port(), election_profile: Some("e2e".into()), diff --git a/app/crowdb-web/tests/mgmt_routes_test.rs b/app/crowdb-web/tests/mgmt_routes_test.rs index 64f8a7f98..f0b74a626 100644 --- a/app/crowdb-web/tests/mgmt_routes_test.rs +++ b/app/crowdb-web/tests/mgmt_routes_test.rs @@ -6,12 +6,15 @@ //! `/api/stores/:sid/groups` routes through HTTP. Skips silently when //! the `crowdb-kv-server` binary is not built. +use std::collections::BTreeMap; use std::net::SocketAddr; use std::path::PathBuf; use std::time::Duration; +use crowdb_console_shared::cluster::{NodeHealth, NodeStore}; use crowdb_console_shared::config::{NodeEntry, RackEntry, ServerEntry, ServiceType}; use crowdb_console_shared::lifecycle::{self, crowdb_kv_server_bin, stop_pid_with_timeout, DeployRequest}; +use crowdb_console_shared::monitor::NodeRecord; use crowdb_console_shared::ConsoleConfig; use crowdb_web::{router, AppState}; use serde_json::json; @@ -61,7 +64,7 @@ async fn spawn_upstream() -> Option { ssh_password: None, }; let req = DeployRequest { - server_id: "n1".to_string(), + server_id: "1".to_string(), rest_port: pick_free_port(), rpc_port: pick_free_port(), election_profile: Some("e2e".into()), @@ -266,3 +269,54 @@ async fn full_mgmt_cycle_through_web_routes() { let _ = lifecycle::stop_pid(upstream.pid); tokio::time::sleep(Duration::from_millis(50)).await; } + +#[tokio::test] +async fn store_reads_reject_cached_topology_without_group0() { + let listener = tokio::net::TcpListener::bind(SocketAddr::from(([127, 0, 0, 1], 0))) + .await + .unwrap(); + let address = listener.local_addr().unwrap(); + let state = AppState::with_config(ConsoleConfig::default(), None); + let mut stores = BTreeMap::new(); + stores.insert( + 7, + NodeStore { + node_id: 1, + store_id: 7, + listen_addr: None, + groups: Vec::new(), + }, + ); + state + .monitor_cache + .set_node_report( + 1, + NodeRecord { + health: NodeHealth::Up, + last_seen_ms: 1, + stores, + last_error: None, + recovering: false, + }, + ) + .await; + tokio::spawn(async move { + axum::serve(listener, router(state)).await.unwrap(); + }); + + let client = reqwest::Client::new(); + for path in [ + "/api/stores", + "/api/stores/7", + "/api/stores/7/groups", + "/api/stores/7/groups/70", + "/api/stores/7/groups/70/replicas", + ] { + let response = client + .get(format!("http://{address}{path}")) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 502, "{path}"); + } +} diff --git a/app/crowdb-web/tests/ops_migration_test.rs b/app/crowdb-web/tests/ops_migration_test.rs index 137325746..996d05673 100644 --- a/app/crowdb-web/tests/ops_migration_test.rs +++ b/app/crowdb-web/tests/ops_migration_test.rs @@ -54,7 +54,7 @@ async fn spawn_upstream() -> Option { ssh_password: None, }; let req = DeployRequest { - server_id: "n1".to_string(), + server_id: "1".to_string(), rest_port: pick_free_port(), rpc_port: pick_free_port(), election_profile: Some("e2e".into()), diff --git a/app/crowdb-web/tests/replica_leader_removal_test.rs b/app/crowdb-web/tests/replica_leader_removal_test.rs index 19fef13f3..666e788f1 100644 --- a/app/crowdb-web/tests/replica_leader_removal_test.rs +++ b/app/crowdb-web/tests/replica_leader_removal_test.rs @@ -445,9 +445,9 @@ async fn remove_leader_from_five_node_group_elects_new_leader() { #[tokio::test] #[allow(clippy::unused_async)] -async fn remove_unreachable_leader_from_five_node_group_uses_lease_fallback() { +async fn remove_unreachable_leader_retains_group0_membership() { let Some(mut cluster) = - spawn_five_node_cluster("remove_unreachable_leader_from_five_node_group_uses_lease_fallback").await + spawn_five_node_cluster("remove_unreachable_leader_retains_group0_membership").await else { return; }; @@ -488,23 +488,43 @@ async fn remove_unreachable_leader_from_five_node_group_uses_lease_fallback() { .unwrap(); assert_eq!( resp.status(), - 204, - "delete unreachable leader replica should succeed: {:?}", - resp.text().await.ok() + 500, + "delete must fail closed when the target is unreachable" + ); + let error = resp.text().await.unwrap(); + assert!( + error.contains("error sending request"), + "unexpected error: {error}" ); - let (new_leader_rid, new_leader_node) = - wait_for_leader_after_removal(&cluster, leader_rid, Duration::from_secs(15)) - .await - .expect("a new leader should be elected among survivors after lease expiry"); - assert_ne!(new_leader_rid, leader_rid); - assert!(cluster.nodes.contains_key(&new_leader_node)); - - // The console's monitor cache may still report the dead node as a - // replica because the node is unreachable and cannot be refreshed. - // The real safety check is that every survivor has removed the dead - // leader from its remote list and elected a new leader. - assert_removed_absent_from_all(&cluster, leader_rid).await; + let survivor = cluster + .nodes + .values() + .find(|node| node.node_id != leader_node) + .unwrap(); + let context = crowdb_console_shared::ops::OpContext::new( + survivor.rpc_url.trim_start_matches("http://").to_string(), + vec![survivor.mgmt_url.clone()], + ConsoleConfig::default(), + ); + let replicas = context.sysmd().list_replicas_in_group(sid, gid).await.unwrap(); + assert!(replicas.iter().any(|replica| replica.replica_id == leader_rid)); + + let response = http + .get(format!("{base}/api/stores/{sid}/groups/{gid}")) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 200); + let group: serde_json::Value = response.json().await.unwrap(); + let dead_replica = group["replicas"] + .as_array() + .unwrap() + .iter() + .find(|replica| replica["replica_id"] == leader_rid) + .expect("Group 0 member remains visible when unobserved"); + assert_eq!(dead_replica["role"], "unknown"); + assert_eq!(dead_replica["state"], "unknown"); cluster.stop(); } diff --git a/app/crowdb-web/ui/e2e/fixtures/crowClusterDeployer.ts b/app/crowdb-web/ui/e2e/fixtures/crowClusterDeployer.ts index f8770fbec..9b422158c 100644 --- a/app/crowdb-web/ui/e2e/fixtures/crowClusterDeployer.ts +++ b/app/crowdb-web/ui/e2e/fixtures/crowClusterDeployer.ts @@ -967,13 +967,14 @@ export async function setupCluster(baseURL: string, topo: TopologyDescriptor): P }), ); - await Promise.all(nodes.map((nodeId) => deployNodeServer(baseURL, nodeId, freePort('kv-mgmt'), freePort('kv-listen')))); - const stores: number[] = []; const groups: { storeId: number; groupId: number }[] = []; // Create all stores in parallel — stores are independent. const storeNodes = nodes.slice(0, Math.min(topo.replicasPerGroup, nodes.length)); + await Promise.all(storeNodes.map((nodeId) => deployNodeServer(baseURL, nodeId, freePort('kv-mgmt'), freePort('kv-listen')))); + await clusterInit(baseURL, storeNodes); + await Promise.all(nodes.slice(storeNodes.length).map((nodeId) => deployNodeServer(baseURL, nodeId, freePort('kv-mgmt'), freePort('kv-listen')))); for (let s = 0; s < topo.storeCount; s++) { stores.push(topo.storeBase + s); } diff --git a/app/crowdb-web/ui/e2e/flows/00-shell-embedding.spec.ts b/app/crowdb-web/ui/e2e/flows/00-shell-embedding.spec.ts index 62a7e2b66..543d69291 100644 --- a/app/crowdb-web/ui/e2e/flows/00-shell-embedding.spec.ts +++ b/app/crowdb-web/ui/e2e/flows/00-shell-embedding.spec.ts @@ -1,6 +1,6 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -// Baseline: 1.5s (2026-08-16) +// Baseline: four tests passed; embedding 2.4s, domain toggle 0.7s (2026-09-27) import { test, expect } from '../fixtures/realBackend'; import { @@ -27,9 +27,98 @@ test.describe('shell · embedding', () => { await step('shell: goto', () => page.goto('/')); - // Scope to the banner alert — a toast (also role=alert) may appear - // concurrently with "Failed to load server list:" text. - await expect(page.getByRole('alert').filter({ hasText: 'Backend unreachable' })).toBeVisible({ timeout: 3_000 }); + await expect(page.getByRole('alert').filter({ hasText: 'Console mode unavailable.' })).toBeVisible({ timeout: 3_000 }); + await expect(page.getByRole('button', { name: 'Add Rack' })).toHaveCount(0); + }); + + test('Docker mode shows Group 0 and monitor state without hardware controls', async ({ page }) => { + await page.route('**/api/mode', route => route.fulfill({ json: { mode: 'docker' } })); + await page.route('**/api/preview', route => route.fulfill({ json: { + source: 'group0', + racks: [{ id: 1, status: 1, node_ids: [1] }], + nodes: [{ id: 1, rack_id: 1, status: 1 }], + disk_groups: [], + disks: [], + stores: [{ store_id: 0, node_ids: [1] }, { store_id: 7, node_ids: [1] }], + groups: [{ store_id: 0, group_id: 0 }, { store_id: 7, group_id: 70 }], + replicas: [], + services: [], + monitor: { phase: 'Ready', revision: 1, updated_at_ms: 1, services: {} }, + } })); + await page.goto('/'); + await expect(page.getByTestId('managed-preview')).toBeVisible({ timeout: 3_000 }); + await expect(page.getByTestId('managed-source')).toHaveText('Source: Group 0'); + await expect(page.getByTestId('managed-readonly')).toHaveText('Hardware topology is read-only'); + await expect(page.getByTestId('managed-monitor-phase')).toContainText('Ready'); + await expect(page.getByRole('button', { name: 'Add Rack' })).toHaveCount(0); + await expect(page.getByRole('button', { name: 'Create store' })).toBeDisabled(); + const writes: Array<{ path: string; token: string | undefined; body: unknown }> = []; + await page.route('**/api/stores**', async (route) => { + const request = route.request(); + writes.push({ + path: new URL(request.url()).pathname, + token: request.headers().authorization, + body: request.postDataJSON(), + }); + await route.fulfill({ status: 201, json: {} }); + }); + await page.getByLabel('Management token').fill('m'.repeat(64)); + await page.getByLabel('Store ID').fill('8'); + await page.getByRole('combobox', { name: 'Store node' }).selectOption('1'); + await page.getByRole('button', { name: 'Create store' }).click(); + await expect.poll(() => writes.length, { intervals: [100] }).toBe(1); + expect(writes[0]).toEqual({ path: '/api/stores', token: `Bearer ${'m'.repeat(64)}`, body: { store_id: 8, nodes: [1] } }); + await page.getByRole('combobox', { name: 'Group store' }).selectOption('7'); + await page.getByLabel('Group ID').fill('71'); + await page.getByRole('combobox', { name: 'Group node' }).selectOption('1'); + await page.getByRole('button', { name: 'Create group' }).click(); + await expect.poll(() => writes.length, { intervals: [100] }).toBe(2); + expect(writes[1].path).toBe('/api/stores/7/groups'); + await page.getByRole('combobox', { name: 'Replica group' }).selectOption('7/70'); + await page.getByRole('combobox', { name: 'Replica node' }).selectOption('1'); + await page.getByRole('button', { name: 'Add replica' }).click(); + await expect.poll(() => writes.length, { intervals: [100] }).toBe(3); + expect(writes[2].path).toBe('/api/stores/7/groups/70/replicas'); + }); + + test('Docker mode separates unavailable topology from current monitor recovery', async ({ page }) => { + await page.clock.install(); + await page.route('**/api/mode', route => route.fulfill({ json: { mode: 'docker' } })); + let available = true; + let monitor: object | null = { + phase: 'ready', revision: 1, updated_at_ms: 1, + services: { kv: { pid: 100, generation: 1, restart_attempts: 0, healthy: true } }, + }; + await page.route('**/api/preview', route => route.fulfill({ + status: available ? 200 : 503, + json: available ? { + source: 'group0', racks: [], nodes: [], disks: [], disk_groups: [], + stores: [{ store_id: 7, node_ids: [1] }], groups: [], replicas: [], services: [], monitor, + } : { source: 'group0', available: false, + reason: monitor ? 'group0_unavailable' : 'monitor_unavailable', monitor }, + })); + await page.goto('/'); + await expect(page.getByTestId('managed-process-kv')).toContainText('PID 100'); + await expect(page.getByRole('list', { name: 'Logical stores' })).toContainText('Store 7'); + available = false; + monitor = { phase: 'restarting', revision: 2, updated_at_ms: 2, + services: { kv: { pid: null, generation: 1, restart_attempts: 1, healthy: false } } }; + await page.clock.runFor(3001); + await expect(page.getByTestId('managed-unavailable')).toContainText('Group 0 is unavailable'); + await expect(page.getByRole('list', { name: 'Logical stores' })).toHaveCount(0); + await expect(page.getByTestId('managed-process-kv')).toContainText('unhealthy'); + await expect(page.getByTestId('managed-monitor-phase')).toContainText('restarting'); + monitor = null; + await page.clock.runFor(3001); + await expect(page.getByTestId('managed-unavailable')).toContainText('missing or stale'); + await expect(page.getByTestId('managed-process-kv')).toHaveCount(0); + available = true; + monitor = { phase: 'ready', revision: 3, updated_at_ms: 3, + services: { kv: { pid: 200, generation: 2, restart_attempts: 1, healthy: true } } }; + await page.clock.runFor(3001); + await expect(page.getByTestId('managed-unavailable')).toHaveCount(0); + await expect(page.getByTestId('managed-process-kv')).toContainText('PID 200'); + await expect(page.getByTestId('managed-process-kv')).toContainText('generation 2'); }); test('embedding honors apiPrefix, readonly, and module opt-out', async ({ page, baseURL }) => { diff --git a/app/crowdb-web/ui/e2e/flows/20-kv-store-group.spec.ts b/app/crowdb-web/ui/e2e/flows/20-kv-store-group.spec.ts index 939d78e82..c0c68eff2 100644 --- a/app/crowdb-web/ui/e2e/flows/20-kv-store-group.spec.ts +++ b/app/crowdb-web/ui/e2e/flows/20-kv-store-group.spec.ts @@ -16,11 +16,12 @@ test.describe('kv cluster · store + group CRUD', () => { }); test('creates stores, groups and replicas through the UI against a real deployed server', async ({ page, baseURL }) => { - await step('store-group: setup servers', () => Promise.all([5, 171, 172].map(async (id) => { + await step('store-group: seed nodes', () => Promise.all([5, 171, 172].map(async (id) => { await seedRackAndNode(baseURL!, id, id); - await deployNodeServer(baseURL!, id, freePort(), freePort()); }))); + await deployNodeServer(baseURL!, 5, freePort(), freePort()); await clusterInit(baseURL!, [5]); + await step('store-group: deploy remaining servers', () => Promise.all([171, 172].map((id) => deployNodeServer(baseURL!, id, freePort(), freePort())))); // --- store + group creation chain (store 57, groups 570 / 580) --- const chainApi = await apiContext(baseURL!); @@ -136,11 +137,12 @@ test.describe('kv cluster · store + group CRUD', () => { test('deletes a replica and a group through the UI and verifies the real backend', async ({ page, baseURL }) => { // Keep node 7, which bootstraps group 0, alive through both deletion scenarios. - await step('del-replica-group: setup servers', () => Promise.all([7, 8].map(async (id) => { + await step('del-replica-group: seed nodes', () => Promise.all([7, 8].map(async (id) => { await seedRackAndNode(baseURL!, id, id); - await deployNodeServer(baseURL!, id, freePort(), freePort()); }))); + await deployNodeServer(baseURL!, 7, freePort(), freePort()); await clusterInit(baseURL!, [7]); + await deployNodeServer(baseURL!, 8, freePort(), freePort()); // --- delete a replica (store 77, group 770, replica 7700) --- await step('del-replica-group: setup replica', async () => { diff --git a/app/crowdb-web/ui/e2e/flows/21-kv-reconfig.spec.ts b/app/crowdb-web/ui/e2e/flows/21-kv-reconfig.spec.ts index 9712507e6..ded19f122 100644 --- a/app/crowdb-web/ui/e2e/flows/21-kv-reconfig.spec.ts +++ b/app/crowdb-web/ui/e2e/flows/21-kv-reconfig.spec.ts @@ -212,10 +212,11 @@ test.describe('kv cluster · reconfiguration', () => { const allNodes = [421, 422, 423, 441, 442, 443, 444, 445, 451, 452, 453, 454, 461, 462, 463, 464, 465]; await stepTime('setup: seedRackAndNode x17', () => Promise.all(allNodes.map((r) => seedRackAndNode(apiBase, r, r)))); - await stepTime('setup: deployNodeServer x17', () => Promise.all(allNodes.map((n) => deployNodeServer(apiBase, n, freePort(), freePort())))); + await stepTime('setup: deploy Group 0 nodes', () => Promise.all(allNodes.slice(0, 3).map((n) => deployNodeServer(apiBase, n, freePort(), freePort())))); // Bootstrap group-0 on the first 3 nodes. await stepTime('setup: clusterInit', () => clusterInit(apiBase, [421, 422, 423])); + await stepTime('setup: deploy remaining nodes', () => Promise.all(allNodes.slice(3).map((n) => deployNodeServer(apiBase, n, freePort(), freePort())))); // Create all stores (skip clusterInit — group-0 already exists). await stepTime('setup: createStore x5', () => Promise.all([ diff --git a/app/crowdb-web/ui/e2e/flows/22-kv-topology.spec.ts b/app/crowdb-web/ui/e2e/flows/22-kv-topology.spec.ts index 2ab239dfe..ee1890eb7 100644 --- a/app/crowdb-web/ui/e2e/flows/22-kv-topology.spec.ts +++ b/app/crowdb-web/ui/e2e/flows/22-kv-topology.spec.ts @@ -150,10 +150,11 @@ test.describe('kv cluster · multi-rack/multi-store/multi-group topology', () => ...[200, 201, 202, 203, 204, 205, 206, 207], ]; await step('topology: seedRackAndNode', () => Promise.all(allNodes.map((r) => seedRackAndNode(apiBase, r, r)))); - await step('topology: deployNodeServer', () => Promise.all(allNodes.map((n) => deployNodeServer(apiBase, n, freePort(), freePort())))); + await step('topology: deploy Group 0 nodes', () => Promise.all([191, 192, 193].map((n) => deployNodeServer(apiBase, n, freePort(), freePort())))); // Bootstrap group-0 on the first 3 nodes (191, 192, 193). await step('topology: clusterInit', () => clusterInit(apiBase, [191, 192, 193])); + await step('topology: deploy remaining nodes', () => Promise.all(allNodes.slice(3).map((n) => deployNodeServer(apiBase, n, freePort(), freePort())))); // Create all stores in parallel — stores are independent. This // replaces the per-test serial setup phases with a single fan-out, diff --git a/app/crowdb-web/ui/e2e/flows/41-canvas-fit-pan.spec.ts b/app/crowdb-web/ui/e2e/flows/41-canvas-fit-pan.spec.ts index 7085844c3..16a6b2a93 100644 --- a/app/crowdb-web/ui/e2e/flows/41-canvas-fit-pan.spec.ts +++ b/app/crowdb-web/ui/e2e/flows/41-canvas-fit-pan.spec.ts @@ -1,6 +1,6 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -// Baseline: 2s (2026-08-16) +// Baseline: 3.2s / 0.8s / 4.2s (2026-09-28) import { test, expect, consoleBaseURL } from '../fixtures/realBackend'; import { createRack, createNode, deployNodeServer, stopNodeServer, freePort, resetAll } from '../fixtures/consoleSetup'; @@ -38,9 +38,9 @@ test.describe('canvas · fit + pan', () => { await expect(page.locator('.react-flow__node').first()).toBeVisible({ timeout: 5_000 }); await expect(fitBtn).toBeVisible(); - // --- KV domain shows the operator panel (no canvas, no Fit All) --- + // Group 0 is not initialized: the KV panel must report unavailable. await page.getByTestId('domain-kv').click(); - await expect(page.getByText(/No stores available|Store/i)).toBeVisible({ timeout: 5_000 }); + await expect(page.getByRole('main').getByText('Backend unreachable — retrying', { exact: true })).toBeVisible({ timeout: 3_000 }); // --- Capacity view shows the CapacityPanel (no canvas) --- await page.getByTestId('domain-chunk').click(); @@ -130,9 +130,9 @@ test.describe('canvas · fit + pan', () => { const pannedTransform = await viewport.evaluate((el) => (el as HTMLElement).style.transform); expect(pannedTransform).not.toEqual(fittedTransform); - // Switch to KV view — no canvas here, just the operator panel. + // Switch to KV view, whose authority has not been initialized. await page.getByTestId('domain-kv').click(); - await expect(page.getByText(/No stores available|Store/i)).toBeVisible({ timeout: 5_000 }); + await expect(page.getByRole('main').getByText('Backend unreachable — retrying', { exact: true })).toBeVisible({ timeout: 3_000 }); // Switch back to Cluster — should fit to window, NOT restore the // panned viewport. The transform should match the fitted state diff --git a/app/crowdb-web/ui/package-lock.json b/app/crowdb-web/ui/package-lock.json index a7ba7bf3a..e867a4bce 100644 --- a/app/crowdb-web/ui/package-lock.json +++ b/app/crowdb-web/ui/package-lock.json @@ -1,12 +1,12 @@ { "name": "crowdb-console-frontend", - "version": "0.1.0", + "version": "0.1.0-dev", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "crowdb-console-frontend", - "version": "0.1.0", + "version": "0.1.0-dev", "dependencies": { "clsx": "^2.1.1", "lucide-react": "^0.456.0", diff --git a/app/crowdb-web/ui/package.json b/app/crowdb-web/ui/package.json index 19966ff9c..281432342 100644 --- a/app/crowdb-web/ui/package.json +++ b/app/crowdb-web/ui/package.json @@ -1,7 +1,7 @@ { "name": "crowdb-console-frontend", "private": true, - "version": "0.1.0", + "version": "0.1.0-dev", "type": "module", "description": "CrowDB Console SPA. Built with Vite + React + TypeScript + Tailwind. Compiled output in dist/ is served by crowdb-web (Axum) at runtime.", "scripts": { diff --git a/app/crowdb-web/ui/src/App.tsx b/app/crowdb-web/ui/src/App.tsx index 38f284786..686daa7c8 100644 --- a/app/crowdb-web/ui/src/App.tsx +++ b/app/crowdb-web/ui/src/App.tsx @@ -65,6 +65,7 @@ import { toUiHealth, HW_STATUS_NAMES } from './utils/entityDisplay'; import { ClusterView } from './views/ClusterView'; import { KvView } from './views/KvView'; import { ChunkView } from './views/ChunkView'; +import { ManagedPreview } from './managed/ManagedPreview'; const Inspector = lazy(() => import('./shell/Inspector').then((m) => ({ default: m.Inspector }))); @@ -1250,6 +1251,27 @@ function AppContent({ apiPrefix = '/api', readonly = false, modules, onEvent }: } export default function App(props: CrowdbConsoleProps = {}) { + const apiPrefix = props.apiPrefix ?? '/api'; + const [mode, setMode] = useState<'loading' | 'legacy' | 'docker' | 'bare-metal-pending' | 'unavailable'>('loading'); + useEffect(() => { + let active = true; + fetch(`${apiPrefix}/mode`) + .then(async (response) => { + if (!response.ok) throw new Error('Console mode unavailable'); + return response.json(); + }) + .then((body) => { + if (active) setMode(body?.mode === 'docker' || body?.mode === 'bare-metal-pending' || body?.mode === 'legacy' ? body.mode : 'unavailable'); + }) + .catch(() => { + if (active) setMode('unavailable'); + }); + return () => { active = false; }; + }, [apiPrefix]); + if (mode === 'loading') return
Loading console…
; + if (mode === 'unavailable') return
Console mode unavailable.
; + if (mode === 'bare-metal-pending') return
Bare-metal deployment management is not available yet.
; + if (mode === 'docker') return ; return ( diff --git a/app/crowdb-web/ui/src/components/Tree.tsx b/app/crowdb-web/ui/src/components/Tree.tsx index 620fb05da..1608cf84a 100644 --- a/app/crowdb-web/ui/src/components/Tree.tsx +++ b/app/crowdb-web/ui/src/components/Tree.tsx @@ -18,7 +18,7 @@ export interface TreeNode { icon?: React.ReactNode; children?: TreeNode[]; health?: 'Healthy' | 'Degraded' | 'Failed' | 'Unknown'; - role?: 'Leader' | 'Follower' | 'Remote'; + role?: 'Leader' | 'Follower' | 'Remote' | 'Unknown'; parentIds?: Record; /** Service flavor for `Server` nodes: KV vs DiskDB. */ serviceType?: 'kv' | 'diskdb'; diff --git a/app/crowdb-web/ui/src/components/ui/Badge.test.tsx b/app/crowdb-web/ui/src/components/ui/Badge.test.tsx index c1e5e1348..2221a3fa6 100644 --- a/app/crowdb-web/ui/src/components/ui/Badge.test.tsx +++ b/app/crowdb-web/ui/src/components/ui/Badge.test.tsx @@ -3,9 +3,16 @@ import { describe, it, expect } from 'vitest'; import { render } from '@testing-library/react'; -import { HwStatusBadge } from './Badge'; +import { HwStatusBadge, RoleBadge } from './Badge'; import { hwStatusLabel, hwStatusValue, HW_STATUS_NAMES, hwStatusToUiHealth } from '../../utils/entityDisplay'; +describe('RoleBadge', () => { + it('shows an unobserved replica as unknown, not remote', () => { + const { getByTitle } = render(); + expect(getByTitle('Unknown').textContent).toBe('?'); + }); +}); + describe('HwStatusBadge', () => { it('renders the correct label for each status', () => { for (let s = 0; s < HW_STATUS_NAMES.length; s++) { diff --git a/app/crowdb-web/ui/src/components/ui/Badge.tsx b/app/crowdb-web/ui/src/components/ui/Badge.tsx index 5dca8bf58..196b206f5 100644 --- a/app/crowdb-web/ui/src/components/ui/Badge.tsx +++ b/app/crowdb-web/ui/src/components/ui/Badge.tsx @@ -14,7 +14,7 @@ interface BadgeProps extends React.HTMLAttributes { variant?: BadgeVariant; size?: BadgeSize; healthStatus?: 'Healthy' | 'Degraded' | 'Failed' | 'Unknown'; - role?: 'Leader' | 'Follower' | 'Remote'; + role?: 'Leader' | 'Follower' | 'Remote' | 'Unknown'; icon?: React.ReactNode; compact?: boolean; } @@ -51,18 +51,21 @@ const roleColors = { Leader: 'tw-bg-amber-400/15 tw-text-amber-300 tw-border tw-border-amber-300/40', Follower: 'tw-bg-blue-500/10 tw-text-blue-500 tw-border tw-border-blue-500/30', Remote: 'tw-bg-purple-500/10 tw-text-purple-500 tw-border tw-border-purple-500/30', + Unknown: 'tw-bg-gray-500/10 tw-text-gray-500 tw-border tw-border-gray-500/30', }; const roleIcons = { Leader: , Follower: , Remote: , + Unknown: , }; const roleCompactLabel: Record = { Leader: 'L', Follower: 'F', Remote: 'R', + Unknown: '?', }; export const Badge = React.forwardRef( @@ -118,7 +121,7 @@ export function HealthBadge({ ); } -export function RoleBadge({ role, size = 'sm', compact = false }: { role: ReplicaRole | 'Leader' | 'Follower' | 'Remote'; size?: BadgeSize; compact?: boolean }) { +export function RoleBadge({ role, size = 'sm', compact = false }: { role: ReplicaRole | 'Leader' | 'Follower' | 'Remote' | 'Unknown'; size?: BadgeSize; compact?: boolean }) { const normalizedRole = toUiRole(role.toString()); return ( diff --git a/app/crowdb-web/ui/src/data/useLogicalTree.test.ts b/app/crowdb-web/ui/src/data/useLogicalTree.test.ts new file mode 100644 index 000000000..7c6d23fc7 --- /dev/null +++ b/app/crowdb-web/ui/src/data/useLogicalTree.test.ts @@ -0,0 +1,34 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +import { act, renderHook, waitFor } from '@testing-library/react'; +import { afterEach, expect, it, vi } from 'vitest'; +import { getGroup, listStores } from '../api'; +import { GroupHealth, ReplicaRole, ReplicaState } from '../types'; +import { useLogicalTree } from './useLogicalTree'; + +vi.mock('../api', () => ({ listStores: vi.fn(), getGroup: vi.fn(), listGroups: vi.fn() })); +afterEach(() => { vi.resetAllMocks(); }); + +it('removes previously confirmed topology during an outage and restores confirmed reads', async () => { + vi.mocked(listStores).mockResolvedValue([{ + store_id: '7', nodes: [1], groups: [{ group_id: '70', replica_count: 1 }], + }]); + vi.mocked(getGroup).mockResolvedValue({ + store_id: '7', group_id: '70', state: GroupHealth.Healthy, + replicas: [{ store_id: '7', group_id: '70', replica_id: '700', node_id: 1, + role: ReplicaRole.Leader, state: ReplicaState.Running, engine_healthy: true }], + }); + const { result, unmount } = renderHook(() => useLogicalTree()); + await waitFor(() => expect(result.current.replicas).toHaveLength(1)); + vi.mocked(listStores).mockRejectedValueOnce(new Error('Group 0 unavailable')); + await act(() => result.current.refresh()); + expect(result.current.error?.message).toBe('Group 0 unavailable'); + expect(result.current.stores).toEqual([]); + expect(result.current.groups).toEqual([]); + expect(result.current.replicas).toEqual([]); + await act(() => result.current.refresh()); + expect(result.current.error).toBeNull(); + expect(result.current.replicas).toHaveLength(1); + unmount(); +}); diff --git a/app/crowdb-web/ui/src/data/useLogicalTree.ts b/app/crowdb-web/ui/src/data/useLogicalTree.ts index 81410cfc1..b105eee41 100644 --- a/app/crowdb-web/ui/src/data/useLogicalTree.ts +++ b/app/crowdb-web/ui/src/data/useLogicalTree.ts @@ -135,7 +135,9 @@ export function useLogicalTree({ setReplicas(allReplicas); setError(null); } catch (err) { - console.error('Failed to fetch logical tree:', err); + setStores([]); + setGroups([]); + setReplicas([]); setError(err instanceof Error ? err : new Error('Unknown error fetching logical tree')); } finally { hasLoadedRef.current = true; diff --git a/app/crowdb-web/ui/src/managed/ManagedPreview.tsx b/app/crowdb-web/ui/src/managed/ManagedPreview.tsx new file mode 100644 index 000000000..ce493c041 --- /dev/null +++ b/app/crowdb-web/ui/src/managed/ManagedPreview.tsx @@ -0,0 +1,296 @@ +import { useEffect, useState } from 'react'; + +interface ServiceStatus { + pid: number | null; + generation: number; + healthy: boolean; + restart_attempts: number; +} + +interface ServiceView { + kind: string; + instance_id: string; + endpoint: string; + monitor: ServiceStatus | null; +} + +interface ManagedSnapshot { + source: string; + racks: Array<{ id: number; status: number; node_ids: number[] }>; + nodes: Array<{ id: number; rack_id: number; status: number }>; + disk_groups: Array<{ dg_id: number; node_id: number }>; + disks: Array<{ disk_group_id: number; disk_id: unknown }>; + stores: Array<{ store_id: number; node_ids: number[] }>; + groups: Array<{ store_id: number; group_id: number }>; + replicas: Array<{ store_id: number; group_id: number; replica_id: number }>; + services: ServiceView[]; + monitor: { + phase: string; + revision: number; + updated_at_ms: number; + services: Record; + }; +} + +const reasonLabel: Record = { + group0_unavailable: 'Group 0 is unavailable or its topology is incomplete.', + monitor_unavailable: 'The monitor status is missing or stale.', +}; + +export function ManagedPreview({ apiPrefix }: { apiPrefix: string }) { + const [snapshot, setSnapshot] = useState(null); + const [reason, setReason] = useState(null); + const [monitor, setMonitor] = useState(null); + const [managementToken, setManagementToken] = useState(''); + const [storeId, setStoreId] = useState(''); + const [groupStoreId, setGroupStoreId] = useState(''); + const [groupId, setGroupId] = useState(''); + const [replicaId, setReplicaId] = useState('1'); + const [nodeId, setNodeId] = useState(''); + const [groupNodeId, setGroupNodeId] = useState(''); + const [replicaGroup, setReplicaGroup] = useState(''); + const [replicaNodeId, setReplicaNodeId] = useState(''); + const [newReplicaId, setNewReplicaId] = useState(''); + const [writeError, setWriteError] = useState(null); + const [writeBusy, setWriteBusy] = useState(false); + const [refreshRevision, setRefreshRevision] = useState(0); + + useEffect(() => { + let disposed = false; + let timer: ReturnType | undefined; + let controller: AbortController | undefined; + const refresh = async () => { + controller = new AbortController(); + try { + const response = await fetch(`${apiPrefix}/preview`, { + cache: 'no-store', + signal: controller.signal, + }); + const body = await response.json(); + if (!disposed) { + const available = response.ok && body.source === 'group0'; + setSnapshot(available ? body as ManagedSnapshot : null); + setMonitor(body.source === 'group0' ? body.monitor ?? null : null); + setReason(available ? null : body.reason || 'group0_unavailable'); + } + } catch (error) { + if (!disposed) { + setSnapshot(null); + setMonitor(null); + setReason(error instanceof Error ? error.message : 'group0_unavailable'); + } + } finally { + if (!disposed) timer = setTimeout(refresh, 3000); + } + }; + void refresh(); + return () => { + disposed = true; + if (timer) clearTimeout(timer); + controller?.abort(); + }; + }, [apiPrefix, refreshRevision]); + + const write = async (path: string, method: 'POST' | 'DELETE', body?: object) => { + setWriteBusy(true); + setWriteError(null); + try { + const response = await fetch(`${apiPrefix}${path}`, { + method, + headers: { + Authorization: `Bearer ${managementToken}`, + ...(body ? { 'Content-Type': 'application/json' } : {}), + }, + ...(body ? { body: JSON.stringify(body) } : {}), + }); + if (!response.ok) { + const error = await response.json().catch(() => null); + throw new Error(error?.error || `Request failed (${response.status})`); + } + setRefreshRevision((revision) => revision + 1); + } catch (error) { + setWriteError(error instanceof Error ? error.message : 'Request failed'); + } finally { + setWriteBusy(false); + } + }; + + return ( +
+
+
+
+

CROWDB Single-Node Container

+

Non-production preview · one host, no fault tolerance or upgrade guarantee. Space may remain unreclaimed.

+

Live topology from Group 0 and process state from crowdb-monitor.

+
+
+ Source: Group 0 + Hardware topology is read-only +
+
+ + {!snapshot && ( +
+

Live status unavailable

+

{reason ? (reasonLabel[reason] || 'The live authority cannot be reached.') : 'Loading live status…'}

+

No cached topology is displayed while the source is unavailable.

+
+ )} + + {monitor && ( +
+

Monitor status

+

Phase: {monitor.phase} · revision {monitor.revision}

+
+ {Object.entries(monitor.services).map(([name, service]) => ( +
+ {name} + PID {service.pid ?? '—'} · generation {service.generation} · restarts {service.restart_attempts} · {service.healthy ? 'healthy' : 'unhealthy'} +
+ ))} +
+
+ )} + + {snapshot && ( + <> +
+ {([ + ['Racks', snapshot.racks.length], + ['Nodes', snapshot.nodes.length], + ['Disks', snapshot.disks.length], + ['Stores', snapshot.stores.length], + ] as const).map(([label, count]) => ( +
+
{label}
+
{count}
+
+ ))} +
+ + + +
+

Group 0 topology

+

{snapshot.groups.length} groups · {snapshot.replicas.length} replicas · {snapshot.disk_groups.length} disk groups

+
+
+

Nodes

+
    + {snapshot.nodes.map((node) =>
  • Node {node.id} · rack {node.rack_id} · status {node.status}
  • )} +
+
+
+

Stores

+
    + {snapshot.stores.map((store) =>
  • Store {store.store_id} · nodes {store.node_ids.join(', ')}
  • )} +
+
+
+
+ +
+

Logical topology management

+

Store, group, and replica changes use Group 0. Hardware and process controls remain disabled.

+ + {writeError &&

{writeError}

} +
{ + event.preventDefault(); + void write('/stores', 'POST', { store_id: Number(storeId), nodes: [Number(nodeId)] }); + }} + > + + + +
+
{ + event.preventDefault(); + void write(`/stores/${groupStoreId}/groups`, 'POST', { group_id: Number(groupId), replica_id: Number(replicaId), nodes: [Number(groupNodeId)] }); + }} + > + + + + + +
+
{ + event.preventDefault(); + const [selectedStore, selectedGroup] = replicaGroup.split('/'); + void write(`/stores/${selectedStore}/groups/${selectedGroup}/replicas`, 'POST', { + node_id: Number(replicaNodeId), + ...(newReplicaId ? { replica_id: Number(newReplicaId) } : {}), + }); + }} + > + + + + +
+
    + {snapshot.stores.filter((store) => store.store_id !== 0).map((store) => ( +
  • + Store {store.store_id} · nodes {store.node_ids.join(', ')} + +
  • + ))} +
+
    + {snapshot.groups.map((group) => ( +
  • + Store {group.store_id} · group {group.group_id} + {!(group.store_id === 0 && group.group_id === 0) && ( + + )} +
  • + ))} +
+
    + {snapshot.replicas.filter((replica) => replica.store_id !== 0 || replica.group_id !== 0).map((replica) => ( +
  • + Store {replica.store_id} · group {replica.group_id} · replica {replica.replica_id} + +
  • + ))} +
+
+ +
+

Group 0 services

+
    + {snapshot.services.map((service) => ( +
  • + {service.kind} #{service.instance_id} + {service.endpoint} +
    + {service.monitor + ? `Monitor PID ${service.monitor.pid ?? '—'} · generation ${service.monitor.generation} · restarts ${service.monitor.restart_attempts}` + : 'Monitor mapping unavailable'} +
    +
  • + ))} +
+
+ + )} +
+
+ ); +} diff --git a/app/crowdb-web/ui/src/types/index.ts b/app/crowdb-web/ui/src/types/index.ts index 096ab448b..cbe4b3f7a 100644 --- a/app/crowdb-web/ui/src/types/index.ts +++ b/app/crowdb-web/ui/src/types/index.ts @@ -180,7 +180,8 @@ export interface ReplicaView { // Common Enums export enum ReplicaRole { Leader = 'leader', - Follower = 'follower' + Follower = 'follower', + Unknown = 'unknown' } export enum ReplicaState { diff --git a/app/crowdb-web/ui/src/utils/entityDisplay.ts b/app/crowdb-web/ui/src/utils/entityDisplay.ts index 47b1e7be2..2cb70c0d0 100644 --- a/app/crowdb-web/ui/src/utils/entityDisplay.ts +++ b/app/crowdb-web/ui/src/utils/entityDisplay.ts @@ -2,7 +2,7 @@ // Licensed under the Apache License, Version 2.0. export type UiHealth = 'Healthy' | 'Degraded' | 'Failed' | 'Unknown'; -export type UiRole = 'Leader' | 'Follower' | 'Remote'; +export type UiRole = 'Leader' | 'Follower' | 'Remote' | 'Unknown'; function normalize(value?: string | null): string { return String(value || '').trim().toLowerCase(); @@ -70,7 +70,8 @@ export function toUiRole(value?: string | null): UiRole { const raw = normalize(value); if (raw === 'leader') return 'Leader'; if (raw === 'follower') return 'Follower'; - return 'Remote'; + if (raw === 'remote') return 'Remote'; + return 'Unknown'; } export function toUiReplicaRole(value?: string | null, state?: string | null): UiRole | undefined { diff --git a/container/crowdb-monitor/Cargo.toml b/container/crowdb-monitor/Cargo.toml new file mode 100644 index 000000000..1ae82c36f --- /dev/null +++ b/container/crowdb-monitor/Cargo.toml @@ -0,0 +1,33 @@ +[package] +name = "crowdb-monitor" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +description = "Profile-driven deployment monitor for CROWDB." +publish = false + +[lints] +workspace = true + +[dependencies] +clap = { version = "4", features = ["derive"] } +crowdb-kv-client = { path = "../../lib/crowdb-kv-client" } +crowdb-diskio-client = { path = "../../lib/crowdb-diskio-client" } +crowdb-protocol = { path = "../../lib/crowdb-protocol" } +crowdb-rpc-ffi = { path = "../../lib/crowdb-rpc/ffi" } +flatbuffers = { workspace = true } +rand = "0.8" +reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls"] } +rustix = { version = "1", features = ["process"] } +serde = { version = "1", features = ["derive"] } +serde_json = "1" +sha2 = "0.10" +thiserror.workspace = true +tokio = { workspace = true, features = ["fs", "io-std", "io-util", "macros", "net", "process", "rt-multi-thread", "signal", "sync", "time"] } +toml = "0.8" +uuid = { version = "1", features = ["v4", "v7", "serde"] } + +[dev-dependencies] +crowdb-test-harness = { path = "../../lib/crowdb-test-harness", features = ["kv-client"] } diff --git a/container/crowdb-monitor/src/bootstrap.rs b/container/crowdb-monitor/src/bootstrap.rs new file mode 100644 index 000000000..e9a5d1021 --- /dev/null +++ b/container/crowdb-monitor/src/bootstrap.rs @@ -0,0 +1,17 @@ +mod chunk; +mod disk_files; +mod hardware; +mod iceberg; +mod kv; +mod logical; +mod s3; +mod storage_probe; + +pub use chunk::{verify_chunk_services, ChunkBootstrapError}; +pub use disk_files::{disk_step_names, ensure_disk_files, DiskBootstrapError}; +pub use hardware::{hardware_step_names, HardwareBootstrap, HardwareBootstrapError}; +pub use iceberg::{iceberg_step_names, IcebergBootstrap, IcebergBootstrapError}; +pub use kv::{kv_step_names, KvBootstrap, KvBootstrapError}; +pub use logical::{logical_step_names, LogicalBootstrap, LogicalBootstrapError}; +pub use s3::{s3_step_names, S3Bootstrap, S3BootstrapError}; +pub use storage_probe::{verify_diskio_disks, StorageProbeError}; diff --git a/container/crowdb-monitor/src/bootstrap/chunk.rs b/container/crowdb-monitor/src/bootstrap/chunk.rs new file mode 100644 index 000000000..9b02aaf4f --- /dev/null +++ b/container/crowdb-monitor/src/bootstrap/chunk.rs @@ -0,0 +1,129 @@ +use std::fs; +use std::time::Duration; + +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, ServiceRegistryClient}; +use crowdb_protocol::chunk_kv::Id128; +use serde::Deserialize; +use thiserror::Error; +use tokio::time::{sleep, Instant}; + +use crate::DeploymentProfile; + +const READY_DEADLINE: Duration = Duration::from_secs(30); +const PROBE_INTERVAL: Duration = Duration::from_millis(200); + +#[derive(Debug, Error)] +pub enum ChunkBootstrapError { + #[error("chunk service configuration failed: {0}")] + Io(#[from] std::io::Error), + #[error("chunk service registry lookup failed: {0}")] + Registry(#[from] crowdb_kv_client::Error), + #[error("chunk service configuration is invalid: {0}")] + Invalid(&'static str), + #[error("chunk service registration did not become ready")] + Deadline, +} + +#[derive(Deserialize)] +struct ChunkdbConfig { + server: ChunkdbServer, +} + +#[derive(Deserialize)] +struct ChunkdbServer { + instance_id: String, + rpc_listen_addr: String, +} + +#[derive(Deserialize)] +struct ChunkKvConfig { + instance_id: u64, + rpc_advertise_addr: String, + bootstrap_partition: BootstrapPartition, +} + +#[derive(Deserialize)] +struct BootstrapPartition { + partition_id: Id128, + owner_epoch: u64, +} + +/// # Errors +/// Requires the live `ChunkDB` and `Chunk-KV` registrations to match rendered configuration. +pub async fn verify_chunk_services( + management_seed: &str, + profile: &DeploymentProfile, +) -> Result<(), ChunkBootstrapError> { + let config_root = profile.paths.run_root.join("config"); + let chunkdb: ChunkdbConfig = toml::from_str(&fs::read_to_string(config_root.join("chunkdb.toml"))?) + .map_err(|_| ChunkBootstrapError::Invalid("ChunkDB configuration cannot be parsed"))?; + let chunk_kv: ChunkKvConfig = toml::from_str(&fs::read_to_string(config_root.join("chunk-kv.toml"))?) + .map_err(|_| ChunkBootstrapError::Invalid("Chunk-KV configuration cannot be parsed"))?; + let chunkdb_id = chunkdb + .server + .instance_id + .parse::() + .map_err(|_| ChunkBootstrapError::Invalid("ChunkDB instance ID is invalid"))?; + if chunkdb_id == 0 || chunk_kv.instance_id == 0 || chunk_kv.bootstrap_partition.owner_epoch == 0 { + return Err(ChunkBootstrapError::Invalid( + "chunk identity or owner epoch is zero", + )); + } + let registry = ServiceRegistryClient::new(CrowdbKvClient::new(ClientConfig::new(vec![ + management_seed.to_owned() + ]))); + registry.kv().refresh_topology().await?; + let deadline = Instant::now() + READY_DEADLINE; + loop { + let chunkdb_instances = registry.read_all_instances("chunkdb").await?; + let chunk_kv_instances = registry.read_all_instances("chunk-kv").await?; + if chunkdb_instances.len() > 1 || chunk_kv_instances.len() > 1 { + return Err(ChunkBootstrapError::Invalid( + "unexpected live chunk service instance", + )); + } + if let Some((id, instance)) = chunkdb_instances.first() { + if *id != chunkdb_id + || instance.instance_id != chunkdb_id + || instance.rpc_endpoint != format!("http://{}", chunkdb.server.rpc_listen_addr) + { + return Err(ChunkBootstrapError::Invalid( + "ChunkDB registration conflicts with configuration", + )); + } + } + if let Some((id, instance)) = chunk_kv_instances.first() { + if *id != chunk_kv.instance_id + || instance.instance_id != chunk_kv.instance_id + || instance.rpc_endpoint != chunk_kv.rpc_advertise_addr + || instance + .extra + .as_ref() + .and_then(|extra| extra.chunk_kv.as_ref()) + .is_none() + { + return Err(ChunkBootstrapError::Invalid( + "Chunk-KV registration conflicts with configuration", + )); + } + } + let partition_ready = chunk_kv_instances + .first() + .and_then(|(_, instance)| instance.extra.as_ref()) + .and_then(|extra| extra.chunk_kv.as_ref()) + .is_some_and(|extra| { + extra.hosted.iter().any(|hosted| { + hosted.partition_id == chunk_kv.bootstrap_partition.partition_id + && hosted.owner_epoch == chunk_kv.bootstrap_partition.owner_epoch + && !hosted.recovering + }) + }); + if !chunkdb_instances.is_empty() && partition_ready { + return Ok(()); + } + if Instant::now() >= deadline { + return Err(ChunkBootstrapError::Deadline); + } + sleep(PROBE_INTERVAL).await; + } +} diff --git a/container/crowdb-monitor/src/bootstrap/disk_files.rs b/container/crowdb-monitor/src/bootstrap/disk_files.rs new file mode 100644 index 000000000..bc2dc52fa --- /dev/null +++ b/container/crowdb-monitor/src/bootstrap/disk_files.rs @@ -0,0 +1,144 @@ +use std::fs::{self, File, OpenOptions}; +use std::io; +use std::os::unix::fs::OpenOptionsExt; +use std::path::Path; + +use thiserror::Error; + +use crate::{ + BootstrapSession, DeploymentProfile, DiskProfile, ManifestError, ManifestState, MonitorEvent, + MonitorEventKind, MonitorLog, MonitorLogError, +}; + +#[derive(Debug, Error)] +pub enum DiskBootstrapError { + #[error("disk bootstrap I/O failed: {0}")] + Io(#[from] io::Error), + #[error("bootstrap manifest failed: {0}")] + Manifest(#[from] ManifestError), + #[error("monitor lifecycle log failed: {0}")] + MonitorLog(#[from] MonitorLogError), + #[error("disk bootstrap state is invalid: {0}")] + Invalid(&'static str), +} + +#[must_use] +pub fn disk_step_names(profile: &DeploymentProfile) -> Vec { + let mut names = profile + .disks + .iter() + .map(|disk| format!("disk-file-{}", disk.disk_id)) + .collect::>(); + names.sort(); + names +} + +/// # Errors +/// Rejects missing or changed files on Ready restart, and never truncates an existing disk. +pub async fn ensure_disk_files( + session: &mut BootstrapSession, + profile: &DeploymentProfile, + events: &mut MonitorLog, +) -> Result<(), DiskBootstrapError> { + let result = ensure_disk_files_inner(session, profile, events).await; + if result.is_err() { + events + .record(&MonitorEvent { + kind: MonitorEventKind::BootstrapFailed, + service: session.manifest().next_step().or(Some("disk-files")), + pid: None, + attempt: None, + }) + .await?; + } + result +} + +async fn ensure_disk_files_inner( + session: &mut BootstrapSession, + profile: &DeploymentProfile, + events: &mut MonitorLog, +) -> Result<(), DiskBootstrapError> { + profile + .validate() + .map_err(|_| DiskBootstrapError::Invalid("deployment profile is invalid"))?; + let disk_root = profile.paths.data_root.join("disks"); + match fs::symlink_metadata(&disk_root) { + Ok(metadata) if !metadata.file_type().is_dir() => { + return Err(DiskBootstrapError::Invalid("disk root is not a directory")); + } + Ok(_) => {} + Err(error) if error.kind() == io::ErrorKind::NotFound => { + if session.manifest().state() == ManifestState::Ready { + return Err(DiskBootstrapError::Invalid("ready disk root is missing")); + } + fs::create_dir(&disk_root)?; + File::open(&profile.paths.data_root)?.sync_all()?; + } + Err(error) => return Err(error.into()), + } + let mut disks = profile.disks.iter().collect::>(); + disks.sort_by(|left, right| left.disk_id.cmp(&right.disk_id)); + for disk in disks { + let step = format!("disk-file-{}", disk.disk_id); + let complete = session + .manifest() + .step_complete(&step) + .ok_or(DiskBootstrapError::Invalid("disk step is absent from manifest"))?; + if !complete { + events + .record(&MonitorEvent { + kind: MonitorEventKind::BootstrapStepStarted, + service: Some(&step), + pid: None, + attempt: None, + }) + .await?; + } + ensure_one_disk(&disk_root, disk, complete, session.manifest().state())?; + if !complete { + session.complete_step(&step)?; + events + .record(&MonitorEvent { + kind: MonitorEventKind::BootstrapStepCompleted, + service: Some(&step), + pid: None, + attempt: None, + }) + .await?; + } + } + Ok(()) +} + +fn ensure_one_disk( + disk_root: &Path, + disk: &DiskProfile, + complete: bool, + state: ManifestState, +) -> Result<(), DiskBootstrapError> { + match fs::symlink_metadata(&disk.path) { + Ok(metadata) => { + if !metadata.file_type().is_file() || metadata.len() != disk.capacity_bytes { + return Err(DiskBootstrapError::Invalid( + "disk file type or capacity conflicts", + )); + } + } + Err(error) if error.kind() == io::ErrorKind::NotFound => { + if complete || state == ManifestState::Ready { + return Err(DiskBootstrapError::Invalid("completed disk file is missing")); + } + let file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&disk.path)?; + file.set_len(disk.capacity_bytes)?; + file.sync_all()?; + File::open(disk_root)?.sync_all()?; + } + Err(error) => return Err(error.into()), + } + Ok(()) +} diff --git a/container/crowdb-monitor/src/bootstrap/hardware.rs b/container/crowdb-monitor/src/bootstrap/hardware.rs new file mode 100644 index 000000000..81535fdb3 --- /dev/null +++ b/container/crowdb-monitor/src/bootstrap/hardware.rs @@ -0,0 +1,476 @@ +use std::collections::BTreeSet; + +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, HardwareClient}; +use crowdb_protocol::common::{DiskId, HwStatus, NodeValue, RackValue}; +use crowdb_protocol::diskdb::rpc::{DiskGroupValue, DiskType, DiskValue}; +use thiserror::Error; + +use crate::{ + BootstrapSession, DeploymentProfile, GroupRole, ManifestError, MonitorEvent, MonitorEventKind, + MonitorLog, MonitorLogError, +}; + +const STEP: &str = "hardware-topology"; +const UNIT_BYTES: u64 = 1024 * 1024; + +#[derive(Debug, Error)] +pub enum HardwareBootstrapError { + #[error("Group 0 hardware request failed: {0}")] + Client(#[from] crowdb_kv_client::Error), + #[error("bootstrap manifest failed: {0}")] + Manifest(#[from] ManifestError), + #[error("monitor lifecycle log failed: {0}")] + MonitorLog(#[from] MonitorLogError), + #[error("hardware bootstrap state is invalid: {0}")] + Invalid(&'static str), +} + +struct ExpectedHardware { + rack_id: u64, + node_id: u64, + group_id: u64, + bind_store_id: u64, + bind_group_id: u64, + rack: RackValue, + node: NodeValue, + group: DiskGroupValue, + disks: Vec<(DiskId, DiskValue)>, +} + +#[derive(Clone, Copy, Eq, Ord, PartialEq, PartialOrd)] +enum HardwarePart { + Rack, + Node, + Group, + Owner, + Bind, +} + +#[derive(Default)] +struct ExistingHardware { + parts: BTreeSet, + disks: BTreeSet<(u64, u64)>, +} + +impl ExistingHardware { + fn has(&self, part: HardwarePart) -> bool { + self.parts.contains(&part) + } + + fn complete(&self, disk_count: usize) -> bool { + self.parts.len() == 5 && self.disks.len() == disk_count + } +} + +pub struct HardwareBootstrap { + client: HardwareClient, +} + +impl HardwareBootstrap { + #[must_use] + pub fn new(management_seed: String) -> Self { + let kv = CrowdbKvClient::new(ClientConfig::new(vec![management_seed])); + Self { + client: HardwareClient::new(kv), + } + } + + /// # Errors + /// Refuses unknown or conflicting Group 0 hardware before any mutation. + pub async fn reconcile( + &self, + session: &mut BootstrapSession, + profile: &DeploymentProfile, + events: &mut MonitorLog, + ) -> Result<(), HardwareBootstrapError> { + let result = self.reconcile_inner(session, profile, events).await; + if result.is_err() { + events + .record(&MonitorEvent { + kind: MonitorEventKind::BootstrapFailed, + service: Some(STEP), + pid: None, + attempt: None, + }) + .await?; + } + result + } + + async fn reconcile_inner( + &self, + session: &mut BootstrapSession, + profile: &DeploymentProfile, + events: &mut MonitorLog, + ) -> Result<(), HardwareBootstrapError> { + let expected = expected(profile)?; + let complete = session + .manifest() + .step_complete(STEP) + .ok_or(HardwareBootstrapError::Invalid( + "hardware step is absent from manifest", + ))?; + self.client.kv().refresh_topology().await?; + let found = self.preflight(&expected).await?; + if complete && !found.complete(expected.disks.len()) { + return Err(HardwareBootstrapError::Invalid( + "completed hardware topology is incomplete", + )); + } + if complete { + return Ok(()); + } + if session.manifest().next_step() != Some(STEP) { + return Err(HardwareBootstrapError::Invalid("hardware step is out of order")); + } + record(events, MonitorEventKind::BootstrapStepStarted).await?; + self.write_missing(&expected, &found).await?; + let final_state = self.preflight(&expected).await?; + if !final_state.complete(expected.disks.len()) { + return Err(HardwareBootstrapError::Invalid("hardware topology is incomplete")); + } + session.complete_step(STEP)?; + record(events, MonitorEventKind::BootstrapStepCompleted).await?; + Ok(()) + } + + async fn write_missing( + &self, + expected: &ExpectedHardware, + found: &ExistingHardware, + ) -> Result<(), HardwareBootstrapError> { + if !found.has(HardwarePart::Rack) { + let write = self.client.add_rack(expected.rack_id, &expected.rack).await; + let actual = self.client.get_rack(expected.rack_id).await?; + verify_written(write, actual.is_some_and(|value| rack_matches(&value, expected)))?; + } + if !found.has(HardwarePart::Node) { + let write = self + .client + .add_node(expected.rack_id, expected.node_id, &expected.node) + .await; + let actual = self.client.get_node(expected.rack_id, expected.node_id).await?; + verify_written(write, actual.is_some_and(|value| node_matches(&value, expected)))?; + } + if !found.has(HardwarePart::Group) { + let write = self + .client + .add_disk_group_with_owner( + expected.rack_id, + expected.node_id, + expected.group_id, + &expected.group, + expected.node_id, + u64::MAX, + ) + .await; + let actual = self + .client + .get_disk_group(expected.rack_id, expected.node_id, expected.group_id) + .await?; + let owner = self + .client + .get_owner(expected.rack_id, expected.node_id, expected.group_id) + .await?; + verify_written( + write, + actual.is_some_and(|value| group_matches(&value.value, expected)) + && owner.is_some_and(|value| value.instance_id == expected.node_id), + )?; + } + if !found.has(HardwarePart::Owner) && found.has(HardwarePart::Group) { + let write = self + .client + .set_owner( + expected.rack_id, + expected.node_id, + expected.group_id, + expected.node_id, + u64::MAX, + ) + .await; + let actual = self + .client + .get_owner(expected.rack_id, expected.node_id, expected.group_id) + .await?; + verify_written( + write, + actual.is_some_and(|value| value.instance_id == expected.node_id), + )?; + } + for (disk_id, disk_value) in &expected.disks { + if found.disks.contains(&(disk_id.high, disk_id.low)) { + continue; + } + let write = self + .client + .add_disk( + expected.rack_id, + expected.node_id, + expected.group_id, + disk_id, + disk_value, + ) + .await; + let actual = self + .client + .get_disk(expected.rack_id, expected.node_id, expected.group_id, disk_id) + .await?; + verify_written( + write, + actual.is_some_and(|value| disk_matches(&value, disk_value)), + )?; + } + self.ensure_bind(expected, found).await?; + Ok(()) + } + + async fn ensure_bind( + &self, + expected: &ExpectedHardware, + found: &ExistingHardware, + ) -> Result<(), HardwareBootstrapError> { + if found.has(HardwarePart::Bind) { + return Ok(()); + } + let write = self + .client + .set_bind( + expected.rack_id, + expected.node_id, + expected.group_id, + expected.bind_store_id, + expected.bind_group_id, + ) + .await; + let actual = self + .client + .get_bind(expected.rack_id, expected.node_id, expected.group_id) + .await?; + verify_written( + write, + actual.is_some_and(|value| { + value.store_id == expected.bind_store_id && value.group_id == expected.bind_group_id + }), + ) + } + + async fn preflight( + &self, + expected: &ExpectedHardware, + ) -> Result { + let mut found = ExistingHardware::default(); + for (rack_id, value) in self.client.list_racks().await? { + if rack_id != expected.rack_id || !rack_matches(&value, expected) { + return Err(HardwareBootstrapError::Invalid( + "Group 0 rack conflicts with profile", + )); + } + found.parts.insert(HardwarePart::Rack); + } + for (rack_id, node_id, value) in self.client.list_nodes().await? { + if rack_id != expected.rack_id || node_id != expected.node_id || !node_matches(&value, expected) { + return Err(HardwareBootstrapError::Invalid( + "Group 0 node conflicts with profile", + )); + } + found.parts.insert(HardwarePart::Node); + } + for entry in self.client.list_disk_groups().await? { + if entry.rack_id != expected.rack_id + || entry.node_id != expected.node_id + || entry.dg_id != expected.group_id + || !group_matches(&entry.value, expected) + { + return Err(HardwareBootstrapError::Invalid( + "Group 0 disk group conflicts with profile", + )); + } + found.parts.insert(HardwarePart::Group); + } + for owner in self.client.list_owners().await? { + if owner.rack_id != expected.rack_id + || owner.node_id != expected.node_id + || owner.dg_id != expected.group_id + || owner.instance_id != expected.node_id + { + return Err(HardwareBootstrapError::Invalid( + "Group 0 owner conflicts with profile", + )); + } + found.parts.insert(HardwarePart::Owner); + } + for bind in self.client.list_binds().await? { + if bind.rack_id != expected.rack_id + || bind.node_id != expected.node_id + || bind.dg_id != expected.group_id + || bind.store_id != expected.bind_store_id + || bind.group_id != expected.bind_group_id + { + return Err(HardwareBootstrapError::Invalid( + "Group 0 bind conflicts with profile", + )); + } + found.parts.insert(HardwarePart::Bind); + } + for entry in self.client.list_all_disks().await? { + let matching = expected + .disks + .iter() + .find(|(disk_id, _)| *disk_id == entry.disk_id); + if entry.rack_id != expected.rack_id + || entry.node_id != expected.node_id + || entry.disk_group_id != expected.group_id + || !matching.is_some_and(|(_, value)| disk_matches(&entry.value, value)) + { + return Err(HardwareBootstrapError::Invalid( + "Group 0 disk conflicts with profile", + )); + } + found.disks.insert((entry.disk_id.high, entry.disk_id.low)); + } + Ok(found) + } +} + +#[must_use] +pub fn hardware_step_names() -> Vec { + vec![STEP.to_owned()] +} + +fn expected(profile: &DeploymentProfile) -> Result { + profile + .validate() + .map_err(|_| HardwareBootstrapError::Invalid("deployment profile is invalid"))?; + let [node] = profile.nodes.as_slice() else { + return Err(HardwareBootstrapError::Invalid( + "preview requires one hardware node", + )); + }; + let Some(first_disk) = profile.disks.first() else { + return Err(HardwareBootstrapError::Invalid("preview has no disks")); + }; + if profile.disks.iter().any(|disk| { + disk.node_id != node.node_id + || disk.disk_group_id != first_disk.disk_group_id + || disk.capacity_bytes != disk.zone_size_bytes + || disk.capacity_bytes % UNIT_BYTES != 0 + }) { + return Err(HardwareBootstrapError::Invalid( + "preview disk layout is incompatible", + )); + } + let mut data_groups = profile + .groups + .iter() + .filter(|group| group.role == GroupRole::Data); + let data_group = data_groups + .next() + .ok_or(HardwareBootstrapError::Invalid("preview has no data group"))?; + if data_groups.next().is_some() { + return Err(HardwareBootstrapError::Invalid("preview requires one data group")); + } + let mut disks = profile + .disks + .iter() + .map(|disk| { + let path = disk + .path + .to_str() + .ok_or(HardwareBootstrapError::Invalid("disk path is not UTF-8"))?; + Ok(( + parse_disk_id(&disk.disk_id)?, + DiskValue { + disk_type: DiskType::BlockSsd as i32, + capacity_units: disk.capacity_bytes / UNIT_BYTES, + zone_size_units: disk.zone_size_bytes / UNIT_BYTES, + unit_size_bytes: u32::try_from(UNIT_BYTES).expect("1 MiB fits into u32"), + zone_count: 1, + status: HwStatus::Up as i32, + device_path: path.to_owned(), + }, + )) + }) + .collect::, HardwareBootstrapError>>()?; + disks.sort_by_key(|(id, _)| (id.high, id.low)); + let disk_ids = disks.iter().map(|(id, _)| *id).collect(); + Ok(ExpectedHardware { + rack_id: node.rack_id, + node_id: node.node_id, + group_id: first_disk.disk_group_id, + bind_store_id: data_group.store_id, + bind_group_id: data_group.group_id, + rack: RackValue { + status: HwStatus::Up as i32, + node_ids: vec![node.node_id], + }, + node: NodeValue { + status: HwStatus::Up as i32, + last_used_dg_id: 0, + disk_group_ids: vec![first_disk.disk_group_id], + status_changed_at_ms: 0, + temp_failure_since_ms: None, + }, + group: DiskGroupValue { + status: HwStatus::Up as i32, + disk_ids, + }, + disks, + }) +} + +fn rack_matches(actual: &RackValue, expected: &ExpectedHardware) -> bool { + actual.node_ids == expected.rack.node_ids +} + +fn node_matches(actual: &NodeValue, expected: &ExpectedHardware) -> bool { + actual.disk_group_ids == expected.node.disk_group_ids +} + +fn group_matches(actual: &DiskGroupValue, expected: &ExpectedHardware) -> bool { + actual.disk_ids == expected.group.disk_ids +} + +fn disk_matches(actual: &DiskValue, expected: &DiskValue) -> bool { + actual.disk_type == expected.disk_type + && actual.capacity_units == expected.capacity_units + && actual.zone_size_units == expected.zone_size_units + && actual.unit_size_bytes == expected.unit_size_bytes + && actual.zone_count == expected.zone_count + && actual.device_path == expected.device_path +} + +fn verify_written( + write: Result<(), crowdb_kv_client::Error>, + visible: bool, +) -> Result<(), HardwareBootstrapError> { + if visible { + return Ok(()); + } + write?; + Err(HardwareBootstrapError::Invalid( + "Group 0 hardware write is not visible", + )) +} + +async fn record(events: &mut MonitorLog, kind: MonitorEventKind) -> Result<(), MonitorLogError> { + events + .record(&MonitorEvent { + kind, + service: Some(STEP), + pid: None, + attempt: None, + }) + .await +} + +fn parse_disk_id(value: &str) -> Result { + if value.len() != 32 || !value.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(HardwareBootstrapError::Invalid("disk ID is not 32 hex digits")); + } + let high = u64::from_str_radix(&value[..16], 16) + .map_err(|_| HardwareBootstrapError::Invalid("disk ID is invalid"))?; + let low = u64::from_str_radix(&value[16..], 16) + .map_err(|_| HardwareBootstrapError::Invalid("disk ID is invalid"))?; + Ok(DiskId { high, low }) +} diff --git a/container/crowdb-monitor/src/bootstrap/iceberg.rs b/container/crowdb-monitor/src/bootstrap/iceberg.rs new file mode 100644 index 000000000..b6dab5689 --- /dev/null +++ b/container/crowdb-monitor/src/bootstrap/iceberg.rs @@ -0,0 +1,306 @@ +use std::time::Duration; + +use serde::Deserialize; +use thiserror::Error; +use tokio::process::Command; +use tokio::time::{sleep, Instant}; +use uuid::Uuid; + +use crate::{ + BootstrapSession, DeploymentProfile, ManifestError, MonitorEvent, MonitorEventKind, MonitorLog, + MonitorLogError, ServerCredentials, +}; + +const INITIALIZE: &str = "iceberg-initialize"; +const ACTIVATE: &str = "iceberg-activate"; +const CAPABILITIES: &str = "0x3fff"; +const COMMAND_TIMEOUT: Duration = Duration::from_secs(30); +const SETTLE_TIMEOUT: Duration = Duration::from_secs(30); + +#[derive(Debug, Error)] +pub enum IcebergBootstrapError { + #[error("Iceberg bootstrap profile is invalid: {0}")] + Profile(&'static str), + #[error("Iceberg catalog state conflicts with the preview: {0}")] + Conflict(&'static str), + #[error("Iceberg management command failed: {0}")] + Command(&'static str), + #[error("Iceberg management process failed: {0}")] + Io(#[from] std::io::Error), + #[error("Iceberg bootstrap manifest failed: {0}")] + Manifest(#[from] ManifestError), + #[error("monitor lifecycle log failed: {0}")] + MonitorLog(#[from] MonitorLogError), +} + +#[derive(Deserialize)] +struct CatalogInspection { + initialized: bool, + catalog_id: Option, + display_name: Option, + activation_epoch: Option, + state: Option, + capability_bits: Option, + root_operation_id: Option, +} + +#[must_use] +pub fn iceberg_step_names() -> [&'static str; 2] { + [INITIALIZE, ACTIVATE] +} + +pub struct IcebergBootstrap; + +impl IcebergBootstrap { + /// # Errors + /// Refuses a foreign catalog, uncertain identity, or incomplete prior step. + pub async fn reconcile( + session: &mut BootstrapSession, + profile: &DeploymentProfile, + credentials: &ServerCredentials, + events: &mut MonitorLog, + ) -> Result<(), IcebergBootstrapError> { + let result = Self::reconcile_inner(session, profile, credentials, events).await; + if result.is_err() { + record(events, MonitorEventKind::BootstrapFailed, "iceberg").await?; + } + result + } + + async fn reconcile_inner( + session: &mut BootstrapSession, + profile: &DeploymentProfile, + credentials: &ServerCredentials, + events: &mut MonitorLog, + ) -> Result<(), IcebergBootstrapError> { + let service = profile + .services + .iter() + .find(|service| service.id == "iceberg") + .ok_or(IcebergBootstrapError::Profile("Iceberg service is absent"))?; + let seeds = service + .env + .get("CROWDB_MANAGEMENT_SEEDS") + .ok_or(IcebergBootstrapError::Profile("management seeds are absent"))?; + let command = ManagementCommand { + program: &service.program, + seeds, + credentials, + }; + let name = &profile.iceberg_catalog; + let initial = command.inspect().await?; + let catalog_id = Self::initialize(session, &command, name, initial, events).await?; + Self::activate(session, &command, name, catalog_id, events).await + } + + async fn initialize( + session: &mut BootstrapSession, + command: &ManagementCommand<'_>, + name: &str, + mut inspection: CatalogInspection, + events: &mut MonitorLog, + ) -> Result { + let complete = session + .manifest() + .step_complete(INITIALIZE) + .ok_or(IcebergBootstrapError::Profile("initialize step is absent"))?; + if complete { + let expected = + session + .manifest() + .step_catalog(INITIALIZE) + .ok_or(IcebergBootstrapError::Conflict( + "completed catalog identity is absent", + ))?; + verify_identity(&inspection, name, expected)?; + return Ok(expected); + } + if session.manifest().next_step() != Some(INITIALIZE) { + return Err(IcebergBootstrapError::Profile("initialize step is out of order")); + } + record(events, MonitorEventKind::BootstrapStepStarted, INITIALIZE).await?; + if !inspection.initialized || inspection.state.as_deref() != Some("Ready") { + let operation = session.reserve_operation(INITIALIZE)?; + let arguments = ["initialize", &operation.to_string(), name]; + let _ = command.execute(&arguments).await; + inspection = command.wait_initialized().await?; + } + verify_name_and_epoch(&inspection, name)?; + let operation = session + .manifest() + .step_operation(INITIALIZE) + .ok_or(IcebergBootstrapError::Conflict("foreign initialized catalog"))?; + verify_operation(&inspection, operation)?; + let catalog_id = inspection + .catalog_id + .ok_or(IcebergBootstrapError::Conflict("catalog identity is absent"))?; + session.complete_catalog_step(INITIALIZE, catalog_id)?; + record(events, MonitorEventKind::BootstrapStepCompleted, INITIALIZE).await?; + Ok(catalog_id) + } + + async fn activate( + session: &mut BootstrapSession, + command: &ManagementCommand<'_>, + name: &str, + catalog_id: Uuid, + events: &mut MonitorLog, + ) -> Result<(), IcebergBootstrapError> { + let complete = session + .manifest() + .step_complete(ACTIVATE) + .ok_or(IcebergBootstrapError::Profile("activate step is absent"))?; + let mut inspection = command.inspect().await?; + verify_identity(&inspection, name, catalog_id)?; + if complete { + return verify_capabilities(&inspection); + } + if session.manifest().next_step() != Some(ACTIVATE) { + return Err(IcebergBootstrapError::Profile("activate step is out of order")); + } + record(events, MonitorEventKind::BootstrapStepStarted, ACTIVATE).await?; + if inspection.capability_bits.as_deref() == Some("0x0000") { + let operation = session.reserve_operation(ACTIVATE)?; + let arguments = ["activate", &operation.to_string(), name, "1", CAPABILITIES]; + let _ = command.execute(&arguments).await; + inspection = command.wait_capabilities().await?; + } + verify_identity(&inspection, name, catalog_id)?; + verify_capabilities(&inspection)?; + let operation = session + .manifest() + .step_operation(ACTIVATE) + .ok_or(IcebergBootstrapError::Conflict("foreign catalog activation"))?; + verify_operation(&inspection, operation)?; + session.complete_step(ACTIVATE)?; + record(events, MonitorEventKind::BootstrapStepCompleted, ACTIVATE).await?; + Ok(()) + } +} + +fn verify_name_and_epoch(inspection: &CatalogInspection, name: &str) -> Result<(), IcebergBootstrapError> { + if !inspection.initialized + || inspection.display_name.as_deref() != Some(name) + || inspection.activation_epoch != Some(1) + || inspection.state.as_deref() != Some("Ready") + { + return Err(IcebergBootstrapError::Conflict( + "catalog name, epoch, or state differs", + )); + } + Ok(()) +} + +fn verify_identity( + inspection: &CatalogInspection, + name: &str, + catalog_id: Uuid, +) -> Result<(), IcebergBootstrapError> { + verify_name_and_epoch(inspection, name)?; + if inspection.catalog_id != Some(catalog_id) { + return Err(IcebergBootstrapError::Conflict("catalog identity differs")); + } + Ok(()) +} + +fn verify_operation(inspection: &CatalogInspection, operation: Uuid) -> Result<(), IcebergBootstrapError> { + if inspection.root_operation_id.as_deref() != Some(operation.simple().to_string().as_str()) { + return Err(IcebergBootstrapError::Conflict( + "management operation identity differs", + )); + } + Ok(()) +} + +fn verify_capabilities(inspection: &CatalogInspection) -> Result<(), IcebergBootstrapError> { + if inspection.capability_bits.as_deref() != Some(CAPABILITIES) { + return Err(IcebergBootstrapError::Conflict("catalog capabilities differ")); + } + Ok(()) +} + +struct ManagementCommand<'a> { + program: &'a std::path::Path, + seeds: &'a str, + credentials: &'a ServerCredentials, +} + +impl ManagementCommand<'_> { + async fn inspect(&self) -> Result { + let output = self.execute(&["inspect"]).await?; + let body = std::str::from_utf8(&output) + .map_err(|_| IcebergBootstrapError::Command("inspection output is invalid"))?; + let mut states = body + .lines() + .filter_map(|line| serde_json::from_str::(line).ok()); + let state = states + .next() + .ok_or(IcebergBootstrapError::Command("inspection output is absent"))?; + if states.next().is_some() { + return Err(IcebergBootstrapError::Command("inspection output is ambiguous")); + } + Ok(state) + } + + async fn execute(&self, arguments: &[&str]) -> Result, IcebergBootstrapError> { + let server_env = self.credentials.server_env(); + let output = tokio::time::timeout( + COMMAND_TIMEOUT, + Command::new(self.program) + .args(arguments) + .env("CROWDB_MANAGEMENT_SEEDS", self.seeds) + .env("CROWDB_ICEBERG_TOKEN", self.credentials.iceberg_manage_token()) + .envs(server_env.lines().filter_map(|line| line.split_once('='))) + .kill_on_drop(true) + .output(), + ) + .await + .map_err(|_| IcebergBootstrapError::Command("management command timed out"))??; + if !output.status.success() { + return Err(IcebergBootstrapError::Command( + "management command exited unsuccessfully", + )); + } + Ok(output.stdout) + } + + async fn wait_initialized(&self) -> Result { + self.wait_for(|inspection| inspection.initialized && inspection.state.as_deref() == Some("Ready")) + .await + } + + async fn wait_capabilities(&self) -> Result { + self.wait_for(|inspection| inspection.capability_bits.as_deref() == Some(CAPABILITIES)) + .await + } + + async fn wait_for( + &self, + ready: impl Fn(&CatalogInspection) -> bool, + ) -> Result { + let deadline = Instant::now() + SETTLE_TIMEOUT; + loop { + let inspection = self.inspect().await?; + if ready(&inspection) { + return Ok(inspection); + } + if Instant::now() >= deadline { + return Err(IcebergBootstrapError::Command( + "management result is not yet visible", + )); + } + sleep(Duration::from_millis(200)).await; + } + } +} + +async fn record(events: &mut MonitorLog, kind: MonitorEventKind, step: &str) -> Result<(), MonitorLogError> { + events + .record(&MonitorEvent { + kind, + service: Some(step), + pid: None, + attempt: None, + }) + .await +} diff --git a/container/crowdb-monitor/src/bootstrap/kv.rs b/container/crowdb-monitor/src/bootstrap/kv.rs new file mode 100644 index 000000000..7a7a798ca --- /dev/null +++ b/container/crowdb-monitor/src/bootstrap/kv.rs @@ -0,0 +1,290 @@ +use std::collections::BTreeSet; +use std::time::Duration; + +use crowdb_protocol::mgmt::{AddGroupRequest, GroupSummary, StoreListResponse, SystemInitRequest}; +use serde::Deserialize; +use thiserror::Error; +use tokio::time::{sleep, Instant}; + +use crate::{ + BootstrapSession, DeploymentProfile, GroupProfile, GroupRole, ManifestError, MonitorEvent, + MonitorEventKind, MonitorLog, MonitorLogError, +}; + +const READY_DEADLINE: Duration = Duration::from_secs(30); +const POLL_INTERVAL: Duration = Duration::from_millis(100); + +#[derive(Debug, Error)] +pub enum KvBootstrapError { + #[error("KV management request failed: {0}")] + Http(#[from] reqwest::Error), + #[error("bootstrap manifest failed: {0}")] + Manifest(#[from] ManifestError), + #[error("monitor lifecycle log failed: {0}")] + MonitorLog(#[from] MonitorLogError), + #[error("KV bootstrap state is invalid: {0}")] + Invalid(&'static str), +} + +#[derive(Deserialize)] +struct GroupReadiness { + ready: bool, + leader_id: u64, + voting_replicas: u32, + reachable_replicas: u32, +} + +pub struct KvBootstrap { + client: reqwest::Client, + base_url: reqwest::Url, +} + +impl KvBootstrap { + /// # Errors + /// Rejects an invalid management endpoint or HTTP client configuration. + pub fn new(base_url: &str) -> Result { + let base_url = reqwest::Url::parse(base_url) + .map_err(|_| KvBootstrapError::Invalid("KV management URI is invalid"))?; + if base_url.scheme() != "http" || base_url.path() != "/" || base_url.query().is_some() { + return Err(KvBootstrapError::Invalid( + "KV management URI must be an HTTP origin", + )); + } + let client = reqwest::Client::builder() + .no_proxy() + .timeout(Duration::from_secs(5)) + .redirect(reqwest::redirect::Policy::none()) + .build()?; + Ok(Self { client, base_url }) + } + + /// # Errors + /// Rejects unknown topology, conflicting identities, or uncertain creation results. + pub async fn reconcile( + &self, + session: &mut BootstrapSession, + profile: &DeploymentProfile, + events: &mut MonitorLog, + ) -> Result<(), KvBootstrapError> { + let result = self.reconcile_inner(session, profile, events).await; + if result.is_err() { + events + .record(&MonitorEvent { + kind: MonitorEventKind::BootstrapFailed, + service: session.manifest().next_step().or(Some("kv")), + pid: None, + attempt: None, + }) + .await?; + } + result + } + + async fn reconcile_inner( + &self, + session: &mut BootstrapSession, + profile: &DeploymentProfile, + events: &mut MonitorLog, + ) -> Result<(), KvBootstrapError> { + let ordered = ordered_groups(profile)?; + self.reject_unknown_stores().await?; + let known = self.list_groups().await?; + let expected = ordered + .iter() + .map(|group| group.group_id) + .collect::>(); + if known.iter().any(|group| !expected.contains(&group.group_id)) { + return Err(KvBootstrapError::Invalid("KV store contains an unknown group")); + } + for group in ordered { + let name = step_name(group); + let complete = session + .manifest() + .step_complete(&name) + .ok_or(KvBootstrapError::Invalid("KV step is absent from manifest"))?; + if !complete { + events + .record(&MonitorEvent { + kind: MonitorEventKind::BootstrapStepStarted, + service: Some(&name), + pid: None, + attempt: None, + }) + .await?; + } + if let Some(existing) = self + .list_groups() + .await? + .iter() + .find(|entry| entry.group_id == group.group_id) + { + verify_group(existing, group)?; + self.wait_ready(group).await?; + if !complete { + session.complete_step(&name)?; + record_step_completed(events, &name).await?; + } + continue; + } + if complete || session.manifest().next_step() != Some(name.as_str()) { + return Err(KvBootstrapError::Invalid("completed KV group is missing")); + } + self.create_group(group).await?; + let existing = self.wait_present(group).await?; + verify_group(&existing, group)?; + self.wait_ready(group).await?; + session.complete_step(&name)?; + record_step_completed(events, &name).await?; + } + Ok(()) + } + + async fn reject_unknown_stores(&self) -> Result<(), KvBootstrapError> { + let response = self + .client + .get(self.url("stores")) + .send() + .await? + .error_for_status()?; + let stores: StoreListResponse = response.json().await?; + if stores.stores.iter().any(|store| store.store_id != 0) { + return Err(KvBootstrapError::Invalid("KV server contains an unknown store")); + } + Ok(()) + } + + async fn list_groups(&self) -> Result, KvBootstrapError> { + let response = self.client.get(self.url("stores/0/groups")).send().await?; + if response.status() == reqwest::StatusCode::NOT_FOUND { + return Ok(Vec::new()); + } + Ok(response.error_for_status()?.json().await?) + } + + async fn create_group(&self, group: &GroupProfile) -> Result<(), KvBootstrapError> { + let response = if group.role == GroupRole::System { + self.client + .post(self.url("system/init")) + .json(&SystemInitRequest { + replica_id: group.replica_id, + start_election: true, + }) + .send() + .await + } else { + self.client + .post(self.url("stores/0/groups")) + .json(&AddGroupRequest { + group_id: group.group_id, + replica_id: group.replica_id, + initial_role: None, + start_election: Some(true), + }) + .send() + .await + }; + match response { + Ok(response) + if response.status().is_client_error() + && response.status() != reqwest::StatusCode::CONFLICT => + { + Err(KvBootstrapError::Invalid("KV rejected group creation")) + } + Ok(_) | Err(_) => Ok(()), + } + } + + async fn wait_present(&self, group: &GroupProfile) -> Result { + let deadline = Instant::now() + READY_DEADLINE; + loop { + if let Some(existing) = self + .list_groups() + .await? + .into_iter() + .find(|entry| entry.group_id == group.group_id) + { + return Ok(existing); + } + if Instant::now() >= deadline { + return Err(KvBootstrapError::Invalid("created KV group is not visible")); + } + sleep(POLL_INTERVAL).await; + } + } + + async fn wait_ready(&self, group: &GroupProfile) -> Result<(), KvBootstrapError> { + let deadline = Instant::now() + READY_DEADLINE; + let path = format!("stores/{}/groups/{}/ready", group.store_id, group.group_id); + loop { + let response = self.client.get(self.url(&path)).send().await?; + if response.status() == reqwest::StatusCode::OK { + let readiness: GroupReadiness = response.json().await?; + if readiness.ready + && readiness.leader_id == group.replica_id + && readiness.voting_replicas == 1 + && readiness.reachable_replicas == 1 + { + return Ok(()); + } + } else if response.status() != reqwest::StatusCode::SERVICE_UNAVAILABLE { + return Err(KvBootstrapError::Invalid( + "KV group readiness endpoint is incompatible", + )); + } + if Instant::now() >= deadline { + return Err(KvBootstrapError::Invalid("KV group leadership deadline expired")); + } + sleep(POLL_INTERVAL).await; + } + } + + fn url(&self, path: &str) -> reqwest::Url { + self.base_url + .join(path) + .expect("validated origin accepts relative paths") + } +} + +async fn record_step_completed(events: &mut MonitorLog, name: &str) -> Result<(), MonitorLogError> { + events + .record(&MonitorEvent { + kind: MonitorEventKind::BootstrapStepCompleted, + service: Some(name), + pid: None, + attempt: None, + }) + .await +} + +/// # Errors +/// Rejects group layouts unsupported by the current one-store KV bootstrap. +pub fn kv_step_names(profile: &DeploymentProfile) -> Result, KvBootstrapError> { + Ok(ordered_groups(profile)?.into_iter().map(step_name).collect()) +} + +fn ordered_groups(profile: &DeploymentProfile) -> Result, KvBootstrapError> { + let mut groups = profile.groups.iter().collect::>(); + groups.sort_by_key(|group| group.group_id); + if groups.first().map_or(true, |group| { + group.store_id != 0 || group.group_id != 0 || group.role != GroupRole::System + }) || groups.iter().any(|group| group.store_id != 0) + { + return Err(KvBootstrapError::Invalid( + "KV bootstrap requires system group 0 in store 0", + )); + } + Ok(groups) +} + +fn step_name(group: &GroupProfile) -> String { + format!("kv-group-{}-{}", group.store_id, group.group_id) +} + +fn verify_group(actual: &GroupSummary, expected: &GroupProfile) -> Result<(), KvBootstrapError> { + if actual.local_replica_id != expected.replica_id || actual.remote_count != 0 { + return Err(KvBootstrapError::Invalid( + "KV group identity or membership conflicts with manifest", + )); + } + Ok(()) +} diff --git a/container/crowdb-monitor/src/bootstrap/logical.rs b/container/crowdb-monitor/src/bootstrap/logical.rs new file mode 100644 index 000000000..ac6b595e9 --- /dev/null +++ b/container/crowdb-monitor/src/bootstrap/logical.rs @@ -0,0 +1,248 @@ +use std::collections::{BTreeMap, BTreeSet}; + +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, CrowdbSysmdClient}; +use crowdb_protocol::common::{GroupValue, ReplicaValue, StoreValue}; +use thiserror::Error; + +use crate::{ + BootstrapSession, DeploymentProfile, GroupRole, ManifestError, MonitorEvent, MonitorEventKind, + MonitorLog, MonitorLogError, +}; + +const STEP: &str = "logical-topology"; + +#[derive(Debug, Error)] +pub enum LogicalBootstrapError { + #[error("Group 0 logical topology request failed: {0}")] + Client(#[from] crowdb_kv_client::Error), + #[error("bootstrap manifest failed: {0}")] + Manifest(#[from] ManifestError), + #[error("monitor lifecycle log failed: {0}")] + MonitorLog(#[from] MonitorLogError), + #[error("logical topology conflicts with the deployment profile")] + Conflict, + #[error("logical topology bootstrap state is invalid: {0}")] + Invalid(&'static str), +} + +#[derive(Default)] +struct Observed { + stores: BTreeSet, + groups: BTreeSet<(u64, u64)>, + replicas: BTreeSet<(u64, u64, u64)>, +} + +pub struct LogicalBootstrap { + client: CrowdbSysmdClient, +} + +impl LogicalBootstrap { + #[must_use] + pub fn new(management_seed: String) -> Self { + Self { + client: CrowdbSysmdClient::new(CrowdbKvClient::new(ClientConfig::new(vec![management_seed]))), + } + } + + /// # Errors + /// Reconciles only matching, profile-owned records after the runtime KV groups exist. + pub async fn reconcile( + &self, + session: &mut BootstrapSession, + profile: &DeploymentProfile, + events: &mut MonitorLog, + ) -> Result<(), LogicalBootstrapError> { + let result = self.reconcile_inner(session, profile, events).await; + if result.is_err() { + record(events, MonitorEventKind::BootstrapFailed).await?; + } + result + } + + async fn reconcile_inner( + &self, + session: &mut BootstrapSession, + profile: &DeploymentProfile, + events: &mut MonitorLog, + ) -> Result<(), LogicalBootstrapError> { + profile + .validate() + .map_err(|_| LogicalBootstrapError::Invalid("deployment profile is invalid"))?; + self.client.kv().refresh_topology().await?; + let expected = expected(profile); + let found = self.preflight(&expected).await?; + let complete = session + .manifest() + .step_complete(STEP) + .ok_or(LogicalBootstrapError::Invalid( + "logical step is absent from manifest", + ))?; + if complete { + return if found.complete(&expected) { + Ok(()) + } else { + Err(LogicalBootstrapError::Invalid( + "completed logical topology is incomplete", + )) + }; + } + if session.manifest().next_step() != Some(STEP) { + return Err(LogicalBootstrapError::Invalid("logical step is out of order")); + } + record(events, MonitorEventKind::BootstrapStepStarted).await?; + self.write_missing(&expected, &found).await?; + if !self.preflight(&expected).await?.complete(&expected) { + return Err(LogicalBootstrapError::Invalid("logical topology is incomplete")); + } + session.complete_step(STEP)?; + record(events, MonitorEventKind::BootstrapStepCompleted).await?; + Ok(()) + } + + async fn preflight(&self, expected: &Expected) -> Result { + let mut found = Observed::default(); + for store in self.client.list_stores().await? { + if expected.stores.get(&store.store_id) != Some(&store) { + return Err(LogicalBootstrapError::Conflict); + } + found.stores.insert(store.store_id); + } + for store_id in expected.stores.keys() { + for group in self.client.list_groups_in_store(*store_id).await? { + let key = (group.store_id, group.group_id); + if expected.groups.get(&key) != Some(&group) { + return Err(LogicalBootstrapError::Conflict); + } + found.groups.insert(key); + } + } + for (store_id, group_id) in expected.groups.keys() { + for replica in self.client.list_replicas_in_group(*store_id, *group_id).await? { + let key = (replica.store_id, replica.group_id, replica.replica_id); + if expected.replicas.get(&key) != Some(&replica) { + return Err(LogicalBootstrapError::Conflict); + } + found.replicas.insert(key); + } + } + Ok(found) + } + + async fn write_missing( + &self, + expected: &Expected, + found: &Observed, + ) -> Result<(), LogicalBootstrapError> { + for (store_id, store) in &expected.stores { + if !found.stores.contains(store_id) { + let write = self.client.add_store(*store_id, &store.node_ids).await; + let actual = self.client.get_store(*store_id).await?; + verify_write(write, actual.as_ref() == Some(store))?; + } + } + for ((store_id, group_id), group) in &expected.groups { + if !found.groups.contains(&(*store_id, *group_id)) { + let write = self.client.add_group(*store_id, *group_id).await; + let actual = self.client.get_group(*store_id, *group_id).await?; + verify_write(write, actual.as_ref() == Some(group))?; + } + } + for ((store_id, group_id, replica_id), replica) in &expected.replicas { + if !found.replicas.contains(&(*store_id, *group_id, *replica_id)) { + let write = self.client.add_replica(replica).await; + let actual = self.client.get_replica(*store_id, *group_id, *replica_id).await?; + verify_write(write, actual.as_ref() == Some(replica))?; + } + } + Ok(()) + } +} + +struct Expected { + stores: BTreeMap, + groups: BTreeMap<(u64, u64), GroupValue>, + replicas: BTreeMap<(u64, u64, u64), ReplicaValue>, +} + +impl Observed { + fn complete(&self, expected: &Expected) -> bool { + self.stores.len() == expected.stores.len() + && self.groups.len() == expected.groups.len() + && self.replicas.len() == expected.replicas.len() + } +} + +fn expected(profile: &DeploymentProfile) -> Expected { + let mut node_ids = BTreeMap::>::new(); + let mut groups = BTreeMap::new(); + let mut replicas = BTreeMap::new(); + for group in &profile.groups { + node_ids.entry(group.store_id).or_default().insert(group.node_id); + groups.insert( + (group.store_id, group.group_id), + GroupValue { + store_id: group.store_id, + group_id: group.group_id, + }, + ); + replicas.insert( + (group.store_id, group.group_id, group.replica_id), + ReplicaValue { + store_id: group.store_id, + group_id: group.group_id, + replica_id: group.replica_id, + node_id: group.node_id, + role: match group.role { + GroupRole::System => "system", + GroupRole::Data => "data", + } + .to_owned(), + voting: true, + endpoint: group.rpc_endpoint.clone(), + }, + ); + } + Expected { + stores: node_ids + .into_iter() + .map(|(store_id, node_ids)| { + ( + store_id, + StoreValue { + store_id, + node_ids: node_ids.into_iter().collect(), + }, + ) + }) + .collect(), + groups, + replicas, + } +} + +fn verify_write( + write: Result<(), crowdb_kv_client::Error>, + matches: bool, +) -> Result<(), LogicalBootstrapError> { + if matches { + Ok(()) + } else { + write?; + Err(LogicalBootstrapError::Invalid("Group 0 write did not persist")) + } +} + +async fn record(events: &mut MonitorLog, kind: MonitorEventKind) -> Result<(), MonitorLogError> { + events + .record(&MonitorEvent { + kind, + service: Some(STEP), + pid: None, + attempt: None, + }) + .await +} + +pub fn logical_step_names() -> impl Iterator { + [STEP].into_iter() +} diff --git a/container/crowdb-monitor/src/bootstrap/s3.rs b/container/crowdb-monitor/src/bootstrap/s3.rs new file mode 100644 index 000000000..2bc2b6197 --- /dev/null +++ b/container/crowdb-monitor/src/bootstrap/s3.rs @@ -0,0 +1,171 @@ +use std::time::Duration; + +use thiserror::Error; +use tokio::process::Command; + +use crate::{ + BootstrapSession, ClientCredentials, CredentialError, DeploymentProfile, ManifestError, MonitorEvent, + MonitorEventKind, MonitorLog, MonitorLogError, ServerCredentials, +}; + +const STEP: &str = "s3-user"; +const COMMAND_TIMEOUT: Duration = Duration::from_secs(30); + +#[derive(Debug, Error)] +pub enum S3BootstrapError { + #[error("S3 bootstrap profile is invalid: {0}")] + Profile(&'static str), + #[error("S3 credential command failed: {0}")] + Command(&'static str), + #[error("S3 credential process failed: {0}")] + Io(#[from] std::io::Error), + #[error("S3 client credentials failed: {0}")] + Credentials(#[from] CredentialError), + #[error("S3 bootstrap manifest failed: {0}")] + Manifest(#[from] ManifestError), + #[error("monitor lifecycle log failed: {0}")] + MonitorLog(#[from] MonitorLogError), +} + +#[must_use] +pub fn s3_step_names() -> [&'static str; 1] { + [STEP] +} + +pub struct S3Bootstrap; + +impl S3Bootstrap { + /// # Errors + /// Rejects absent or conflicting durable users, malformed command output, + /// and a client credential file that differs from the Group 0 record. + pub async fn reconcile( + session: &mut BootstrapSession, + profile: &DeploymentProfile, + credentials: &ServerCredentials, + events: &mut MonitorLog, + ) -> Result<(), S3BootstrapError> { + let complete = session + .manifest() + .step_complete(STEP) + .ok_or(S3BootstrapError::Profile("S3 step is absent from manifest"))?; + if !complete && session.manifest().next_step() != Some(STEP) { + return Err(S3BootstrapError::Profile("S3 step is out of order")); + } + if !complete { + record(events, MonitorEventKind::BootstrapStepStarted).await?; + } + let result = Self::reconcile_inner(session, profile, credentials, complete).await; + match result { + Ok(()) => { + if !complete { + record(events, MonitorEventKind::BootstrapStepCompleted).await?; + } + Ok(()) + } + Err(error) => { + record(events, MonitorEventKind::BootstrapFailed).await?; + Err(error) + } + } + } + + async fn reconcile_inner( + session: &mut BootstrapSession, + profile: &DeploymentProfile, + credentials: &ServerCredentials, + complete: bool, + ) -> Result<(), S3BootstrapError> { + let service = profile + .services + .iter() + .find(|service| service.id == "s3") + .ok_or(S3BootstrapError::Profile("S3 service is missing"))?; + let seeds = service + .env + .get("CROWDB_MANAGEMENT_SEEDS") + .ok_or(S3BootstrapError::Profile("S3 management seeds are missing"))?; + let s3_endpoint = service + .env + .get("CROWDB_S3_PUBLIC_URI") + .cloned() + .ok_or(S3BootstrapError::Profile("S3 public URI is missing"))?; + let iceberg_endpoint = profile + .services + .iter() + .find(|service| service.id == "iceberg") + .and_then(|service| service.env.get("CROWDB_ICEBERG_PUBLIC_URI")) + .cloned() + .ok_or(S3BootstrapError::Profile("Iceberg public URI is missing"))?; + let user = format!("preview-{}", session.manifest().deployment_id()); + let action = if complete { "lookup-user" } else { "ensure-user" }; + let output = tokio::time::timeout( + COMMAND_TIMEOUT, + Command::new(&service.program) + .args([action, &user]) + .env("CROWDB_MANAGEMENT_SEEDS", seeds) + .env("CROWDB_S3_MASTER_KEY", credentials.s3_master_key()) + .kill_on_drop(true) + .output(), + ) + .await + .map_err(|_| S3BootstrapError::Command("credential command timed out"))??; + if !output.status.success() { + return Err(S3BootstrapError::Command( + "credential command exited unsuccessfully", + )); + } + let (access_key_id, secret_access_key) = parse_token(&output.stdout)?; + let client = ClientCredentials { + s3_endpoint, + iceberg_endpoint, + region: service + .env + .get("CROWDB_S3_REGION") + .cloned() + .ok_or(S3BootstrapError::Profile("S3 region is missing"))?, + access_key_id, + secret_access_key, + }; + if complete { + credentials.verify_client(&client)?; + } else { + credentials.persist_client(&client)?; + session.complete_step(STEP)?; + } + Ok(()) + } +} + +fn parse_token(output: &[u8]) -> Result<(String, String), S3BootstrapError> { + let body = + std::str::from_utf8(output).map_err(|_| S3BootstrapError::Command("credential output is invalid"))?; + let mut access_key_id = None; + let mut secret_access_key = None; + for line in body.lines() { + if let Some(value) = line.strip_prefix("AWS_ACCESS_KEY_ID=") { + if value.is_empty() || access_key_id.replace(value).is_some() { + return Err(S3BootstrapError::Command("credential output is invalid")); + } + } + if let Some(value) = line.strip_prefix("AWS_SECRET_ACCESS_KEY=") { + if value.is_empty() || secret_access_key.replace(value).is_some() { + return Err(S3BootstrapError::Command("credential output is invalid")); + } + } + } + let (Some(access_key_id), Some(secret_access_key)) = (access_key_id, secret_access_key) else { + return Err(S3BootstrapError::Command("credential output is invalid")); + }; + Ok((access_key_id.to_owned(), secret_access_key.to_owned())) +} + +async fn record(events: &mut MonitorLog, kind: MonitorEventKind) -> Result<(), MonitorLogError> { + events + .record(&MonitorEvent { + kind, + service: Some(STEP), + pid: None, + attempt: None, + }) + .await +} diff --git a/container/crowdb-monitor/src/bootstrap/storage_probe.rs b/container/crowdb-monitor/src/bootstrap/storage_probe.rs new file mode 100644 index 000000000..eae7d7e26 --- /dev/null +++ b/container/crowdb-monitor/src/bootstrap/storage_probe.rs @@ -0,0 +1,123 @@ +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use crowdb_diskio_client::{DiskId, DiskioClient, DiskioClientConfig, DiskioError, OperationOptions}; +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, ServiceRegistryClient}; +use thiserror::Error; +use tokio::time::{sleep, Instant}; + +use crate::DeploymentProfile; + +const READY_DEADLINE: Duration = Duration::from_secs(30); +const PROBE_INTERVAL: Duration = Duration::from_millis(200); + +#[derive(Debug, Error)] +pub enum StorageProbeError { + #[error("DiskIO probe failed: {0}")] + Diskio(#[from] DiskioError), + #[error("DiskIO registration lookup failed: {0}")] + Registry(#[from] crowdb_kv_client::Error), + #[error("DiskIO profile is invalid: {0}")] + Invalid(&'static str), + #[error("DiskIO registration did not become ready")] + Deadline, +} + +/// # Errors +/// Requires every profile disk to accept a read-only fsync through its Group 0 route. +pub async fn verify_diskio_disks( + management_seed: &str, + profile: &DeploymentProfile, +) -> Result<(), StorageProbeError> { + profile + .validate() + .map_err(|_| StorageProbeError::Invalid("deployment profile is invalid"))?; + let disks = profile + .disks + .iter() + .map(|disk| parse_disk_id(&disk.disk_id)) + .collect::, _>>()?; + wait_for_registration(management_seed, profile).await?; + let config = DiskioClientConfig { + management_seeds: vec![management_seed.to_owned()], + default_timeout: Duration::from_secs(2), + ..DiskioClientConfig::default() + }; + let client = DiskioClient::connect(config).await?; + for disk in disks { + client + .fsync(disk, OperationOptions::within(Duration::from_secs(2))) + .await?; + } + Ok(()) +} + +async fn wait_for_registration( + management_seed: &str, + profile: &DeploymentProfile, +) -> Result<(), StorageProbeError> { + let node = profile + .nodes + .first() + .ok_or(StorageProbeError::Invalid("node is absent"))?; + let service = profile + .services + .iter() + .find(|service| service.id == "diskio") + .ok_or(StorageProbeError::Invalid("DiskIO service is absent"))?; + let group_id = profile + .disks + .first() + .ok_or(StorageProbeError::Invalid("disk is absent"))? + .disk_group_id; + let registry = ServiceRegistryClient::new(CrowdbKvClient::new(ClientConfig::new(vec![ + management_seed.to_owned() + ]))); + registry.kv().refresh_topology().await?; + let deadline = Instant::now() + READY_DEADLINE; + let started_ms = u64::try_from( + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_millis(), + ) + .unwrap_or(u64::MAX); + loop { + if let Some(instance) = registry.read_instance("diskio", node.node_id).await? { + if instance.last_heartbeat_ms <= started_ms { + if Instant::now() >= deadline { + return Err(StorageProbeError::Deadline); + } + sleep(PROBE_INTERVAL).await; + continue; + } + let owner = instance.extra.and_then(|extra| extra.diskdb); + if instance.rpc_endpoint == service.probe.target + && owner.as_ref().is_some_and(|owner| { + owner.rack_id == Some(node.rack_id) + && owner.node_id == Some(node.node_id) + && owner.owned_dg_ids == [group_id] + }) + { + return Ok(()); + } + return Err(StorageProbeError::Invalid( + "DiskIO registration conflicts with profile", + )); + } + if Instant::now() >= deadline { + return Err(StorageProbeError::Deadline); + } + sleep(PROBE_INTERVAL).await; + } +} + +fn parse_disk_id(value: &str) -> Result { + if value.len() != 32 || !value.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(StorageProbeError::Invalid("disk ID is not 32 hex digits")); + } + let high = u64::from_str_radix(&value[..16], 16) + .map_err(|_| StorageProbeError::Invalid("disk ID is invalid"))?; + let low = u64::from_str_radix(&value[16..], 16) + .map_err(|_| StorageProbeError::Invalid("disk ID is invalid"))?; + Ok(DiskId::new(high, low)) +} diff --git a/container/crowdb-monitor/src/credentials.rs b/container/crowdb-monitor/src/credentials.rs new file mode 100644 index 000000000..8a57ff678 --- /dev/null +++ b/container/crowdb-monitor/src/credentials.rs @@ -0,0 +1,333 @@ +use std::fs::{self, File, OpenOptions}; +use std::io::Write; +use std::os::unix::fs::{DirBuilderExt, OpenOptionsExt, PermissionsExt}; +use std::path::{Path, PathBuf}; + +use rand::rngs::OsRng; +use rand::RngCore; +use thiserror::Error; +use uuid::Uuid; + +const SERVER_FILE: &str = "server.env"; +const CLIENT_FILE: &str = "client.env"; +const MAX_ENV_BYTES: u64 = 4096; + +#[derive(Debug, Error)] +pub enum CredentialError { + #[error("credential storage failed: {0}")] + Io(#[from] std::io::Error), + #[error("credential state is invalid: {0}")] + Invalid(&'static str), +} + +pub struct ServerCredentials { + directory: PathBuf, + s3_master_key: String, + iceberg_read_token: String, + iceberg_write_token: String, + iceberg_manage_token: String, + iceberg_clear_token: String, +} + +impl ServerCredentials { + /// # Errors + /// Rejects missing or incompatible credentials without creating new secrets. + pub fn load_existing(data_root: &Path) -> Result { + let directory = data_root.join("secrets"); + let metadata = fs::symlink_metadata(&directory)?; + if !metadata.file_type().is_dir() || metadata.permissions().mode() & 0o777 != 0o700 { + return Err(CredentialError::Invalid("secrets directory must have mode 0700")); + } + let body = read_private(&directory.join(SERVER_FILE))?; + Self::parse(directory, &body) + } + + /// # Errors + /// Rejects missing or incompatible secret state without replacing it. + pub fn load_or_create(data_root: &Path) -> Result { + let directory = data_root.join("secrets"); + if !data_root.is_dir() { + return Err(CredentialError::Invalid("data root is not a directory")); + } + match fs::symlink_metadata(&directory) { + Ok(metadata) if !metadata.file_type().is_dir() => { + return Err(CredentialError::Invalid("secrets directory is not a directory")); + } + Ok(metadata) if metadata.permissions().mode() & 0o777 != 0o700 => { + return Err(CredentialError::Invalid("secrets directory must have mode 0700")); + } + Ok(_) => {} + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + fs::DirBuilder::new().mode(0o700).create(&directory)?; + File::open(data_root)?.sync_all()?; + } + Err(error) => return Err(error.into()), + } + let path = directory.join(SERVER_FILE); + if fs::symlink_metadata(&path).is_ok() { + let body = read_private(&path)?; + return Self::parse(directory, &body); + } + if fs::read_dir(&directory)?.next().is_some() { + return Err(CredentialError::Invalid( + "secrets directory has no server credentials", + )); + } + let result = Self { + directory, + s3_master_key: random_hex(), + iceberg_read_token: random_hex(), + iceberg_write_token: random_hex(), + iceberg_manage_token: random_hex(), + iceberg_clear_token: random_hex(), + }; + atomic_private_write(&path, result.server_env().as_bytes())?; + Ok(result) + } + + #[must_use] + pub fn server_env(&self) -> String { + format!( + "CROWDB_S3_MASTER_KEY={}\nCROWDB_ICEBERG_READ_TOKEN={}\nCROWDB_ICEBERG_WRITE_TOKEN={}\nCROWDB_ICEBERG_MANAGE_TOKEN={}\nCROWDB_ICEBERG_CLEAR_TOKEN={}\n", + self.s3_master_key, + self.iceberg_read_token, + self.iceberg_write_token, + self.iceberg_manage_token, + self.iceberg_clear_token, + ) + } + + #[must_use] + pub fn s3_master_key(&self) -> &str { + &self.s3_master_key + } + + #[must_use] + pub fn iceberg_manage_token(&self) -> &str { + &self.iceberg_manage_token + } + + /// # Errors + /// Requires an existing client file to match the authoritative user and endpoints. + pub fn verify_client(&self, client: &ClientCredentials) -> Result<(), CredentialError> { + client.validate()?; + let existing = read_private(&self.directory.join(CLIENT_FILE))?; + if existing != client.env(&self.iceberg_write_token) { + return Err(CredentialError::Invalid("existing client credentials conflict")); + } + Ok(()) + } + + /// # Errors + /// Rejects conflicting, incomplete, or invalid client credentials. + pub fn persist_client(&self, client: &ClientCredentials) -> Result<(), CredentialError> { + client.validate()?; + let path = self.directory.join(CLIENT_FILE); + if fs::symlink_metadata(&path).is_ok() { + let existing = read_private(&path)?; + if existing == client.env(&self.iceberg_write_token) { + return Ok(()); + } + return Err(CredentialError::Invalid("existing client credentials conflict")); + } + atomic_private_write(&path, client.env(&self.iceberg_write_token).as_bytes()) + } + + fn parse(directory: PathBuf, body: &str) -> Result { + let mut values = body.lines(); + let s3_master_key = take_hex(&mut values, "CROWDB_S3_MASTER_KEY")?; + let iceberg_read_token = take_hex(&mut values, "CROWDB_ICEBERG_READ_TOKEN")?; + let iceberg_write_token = take_hex(&mut values, "CROWDB_ICEBERG_WRITE_TOKEN")?; + let iceberg_manage_token = take_hex(&mut values, "CROWDB_ICEBERG_MANAGE_TOKEN")?; + let iceberg_clear_token = take_hex(&mut values, "CROWDB_ICEBERG_CLEAR_TOKEN")?; + if values.next().is_some() { + return Err(CredentialError::Invalid( + "server credentials contain extra values", + )); + } + let tokens = [ + &iceberg_read_token, + &iceberg_write_token, + &iceberg_manage_token, + &iceberg_clear_token, + ]; + if tokens + .iter() + .enumerate() + .any(|(index, token)| tokens[..index].contains(token)) + { + return Err(CredentialError::Invalid("Iceberg tokens must be distinct")); + } + Ok(Self { + directory, + s3_master_key, + iceberg_read_token, + iceberg_write_token, + iceberg_manage_token, + iceberg_clear_token, + }) + } +} + +pub struct ClientCredentials { + pub s3_endpoint: String, + pub iceberg_endpoint: String, + pub region: String, + pub access_key_id: String, + pub secret_access_key: String, +} + +impl ClientCredentials { + fn validate(&self) -> Result<(), CredentialError> { + if !has_http_authority(&self.s3_endpoint) + || !has_http_authority(&self.iceberg_endpoint) + || ![ + &self.s3_endpoint, + &self.iceberg_endpoint, + &self.region, + &self.access_key_id, + &self.secret_access_key, + ] + .iter() + .all(|value| { + !value.is_empty() + && value + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || b"-._~:/".contains(&byte)) + }) + { + return Err(CredentialError::Invalid("client credential value is invalid")); + } + Ok(()) + } + + fn env(&self, writer_token: &str) -> String { + format!( + "AWS_ENDPOINT_URL={}\nAWS_DEFAULT_REGION={}\nAWS_ACCESS_KEY_ID={}\nAWS_SECRET_ACCESS_KEY={}\nICEBERG_URI={}\nICEBERG_TOKEN={}\n", + self.s3_endpoint, + self.region, + self.access_key_id, + self.secret_access_key, + self.iceberg_endpoint, + writer_token, + ) + } +} + +fn has_http_authority(value: &str) -> bool { + value + .strip_prefix("http://") + .or_else(|| value.strip_prefix("https://")) + .is_some_and(|authority| !authority.is_empty() && !authority.starts_with('/')) +} + +/// # Errors +/// Rejects missing, symlinked, malformed, or publicly readable client files. +pub fn show_client_credentials(data_root: &Path) -> Result { + let directory = data_root.join("secrets"); + let metadata = fs::symlink_metadata(&directory)?; + if !metadata.file_type().is_dir() || metadata.permissions().mode() & 0o777 != 0o700 { + return Err(CredentialError::Invalid("secrets directory must have mode 0700")); + } + let body = read_private(&directory.join(CLIENT_FILE))?; + validate_client_env(&body)?; + Ok(body) +} + +fn validate_client_env(body: &str) -> Result<(), CredentialError> { + let mut values = body.lines(); + let endpoint = take_value(&mut values, "AWS_ENDPOINT_URL")?; + let region = take_value(&mut values, "AWS_DEFAULT_REGION")?; + let access_key = take_value(&mut values, "AWS_ACCESS_KEY_ID")?; + let secret_key = take_value(&mut values, "AWS_SECRET_ACCESS_KEY")?; + let iceberg_uri = take_value(&mut values, "ICEBERG_URI")?; + let writer_token = take_value(&mut values, "ICEBERG_TOKEN")?; + if values.next().is_some() { + return Err(CredentialError::Invalid( + "client credentials contain extra values", + )); + } + ClientCredentials { + s3_endpoint: endpoint.into(), + iceberg_endpoint: iceberg_uri.into(), + region: region.into(), + access_key_id: access_key.into(), + secret_access_key: secret_key.into(), + } + .validate()?; + if writer_token.len() != 64 || !writer_token.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(CredentialError::Invalid("Iceberg writer token is invalid")); + } + Ok(()) +} + +fn take_value<'a>(lines: &mut impl Iterator, key: &str) -> Result<&'a str, CredentialError> { + lines + .next() + .and_then(|line| line.strip_prefix(key)) + .and_then(|line| line.strip_prefix('=')) + .ok_or(CredentialError::Invalid("credential file is incomplete")) +} + +fn take_hex<'a>(lines: &mut impl Iterator, key: &str) -> Result { + let value = lines + .next() + .and_then(|line| line.strip_prefix(key)) + .and_then(|line| line.strip_prefix('=')) + .ok_or(CredentialError::Invalid("server credentials are incomplete"))?; + if value.len() != 64 || !value.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(CredentialError::Invalid("server credential value is invalid")); + } + Ok(value.to_owned()) +} + +fn random_hex() -> String { + let mut bytes = [0_u8; 32]; + OsRng.fill_bytes(&mut bytes); + let mut output = String::with_capacity(64); + for byte in bytes { + use std::fmt::Write as _; + write!(output, "{byte:02x}").expect("writing to String cannot fail"); + } + output +} + +fn read_private(path: &Path) -> Result { + let metadata = fs::symlink_metadata(path)?; + if !metadata.file_type().is_file() + || metadata.permissions().mode() & 0o777 != 0o600 + || metadata.len() > MAX_ENV_BYTES + { + return Err(CredentialError::Invalid( + "credential file must be a bounded mode-0600 regular file", + )); + } + Ok(fs::read_to_string(path)?) +} + +fn atomic_private_write(path: &Path, body: &[u8]) -> Result<(), CredentialError> { + if body.len() as u64 > MAX_ENV_BYTES { + return Err(CredentialError::Invalid("credential file exceeds size bound")); + } + let directory = path + .parent() + .ok_or(CredentialError::Invalid("credential path has no parent"))?; + let temporary = directory.join(format!(".credential-{}.tmp", Uuid::new_v4())); + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&temporary)?; + let result = (|| { + file.write_all(body)?; + file.sync_all()?; + fs::hard_link(&temporary, path)?; + File::open(directory)?.sync_all()?; + fs::remove_file(&temporary)?; + File::open(directory)?.sync_all() + })(); + if result.is_err() { + let _ = fs::remove_file(&temporary); + } + result.map_err(CredentialError::Io) +} diff --git a/container/crowdb-monitor/src/layout.rs b/container/crowdb-monitor/src/layout.rs new file mode 100644 index 000000000..e274a36fa --- /dev/null +++ b/container/crowdb-monitor/src/layout.rs @@ -0,0 +1,15 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::path::{Component, Path}; + +pub(crate) fn is_clean_absolute(path: &Path) -> bool { + path.is_absolute() + && path + .components() + .all(|component| !matches!(component, Component::CurDir | Component::ParentDir)) +} + +pub(crate) fn is_strict_descendant(root: &Path, path: &Path) -> bool { + is_clean_absolute(root) && is_clean_absolute(path) && path != root && path.starts_with(root) +} diff --git a/container/crowdb-monitor/src/lib.rs b/container/crowdb-monitor/src/lib.rs new file mode 100644 index 000000000..cdb72eba6 --- /dev/null +++ b/container/crowdb-monitor/src/lib.rs @@ -0,0 +1,38 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +mod bootstrap; +mod credentials; +mod layout; +mod liveness; +mod manifest; +mod monitor_log; +mod preview; +mod probe; +mod process; +mod profile; +mod render; +mod status; +mod supervisor; + +pub use bootstrap::{ + disk_step_names, ensure_disk_files, hardware_step_names, iceberg_step_names, kv_step_names, + logical_step_names, s3_step_names, verify_chunk_services, verify_diskio_disks, ChunkBootstrapError, + DiskBootstrapError, HardwareBootstrap, HardwareBootstrapError, IcebergBootstrap, IcebergBootstrapError, + KvBootstrap, KvBootstrapError, LogicalBootstrap, LogicalBootstrapError, S3Bootstrap, S3BootstrapError, + StorageProbeError, +}; +pub use credentials::{show_client_credentials, ClientCredentials, CredentialError, ServerCredentials}; +pub use liveness::{probe_liveness, LivenessError, LivenessServer}; +pub use manifest::{BootstrapManifest, BootstrapSession, ManifestError, ManifestState}; +pub use monitor_log::{MonitorEvent, MonitorEventKind, MonitorLog, MonitorLogError}; +pub use preview::{run_preview, PreviewError}; +pub use probe::{ProbeError, ProbeExecutor}; +pub use process::{ProcessError, ProcessManager}; +pub use profile::{ + DeploymentProfile, DiskProfile, GroupProfile, GroupRole, LogProfile, NodeProfile, PathProfile, ProbeKind, + ProbeProfile, ProfileError, PublicEndpoint, RestartProfile, ServiceProfile, +}; +pub use render::{render_configs, RenderError, RenderedConfig}; +pub use status::{MonitorPhase, MonitorStatus, ServiceStatus, StatusError, StatusStore}; +pub use supervisor::{Supervisor, SupervisorError}; diff --git a/container/crowdb-monitor/src/liveness.rs b/container/crowdb-monitor/src/liveness.rs new file mode 100644 index 000000000..7ea9b4f62 --- /dev/null +++ b/container/crowdb-monitor/src/liveness.rs @@ -0,0 +1,93 @@ +use std::fs; +use std::io; +use std::os::unix::fs::{DirBuilderExt, FileTypeExt, PermissionsExt}; +use std::path::{Path, PathBuf}; +use std::time::Duration; + +use thiserror::Error; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::{UnixListener, UnixStream}; +use tokio::task::JoinHandle; +use tokio::time::timeout; + +const SOCKET: &str = "monitor.sock"; +const DEADLINE: Duration = Duration::from_secs(2); + +#[derive(Debug, Error)] +pub enum LivenessError { + #[error("monitor liveness I/O failed: {0}")] + Io(#[from] io::Error), + #[error("monitor liveness timed out")] + Timeout, + #[error("monitor liveness state is invalid")] + Invalid, +} + +pub struct LivenessServer { + task: JoinHandle<()>, + path: PathBuf, +} + +impl LivenessServer { + /// # Errors + /// Refuses a competing monitor or an unsafe health path. + pub fn start(run_root: &Path) -> Result { + let directory = run_root.join("health"); + match fs::symlink_metadata(&directory) { + Ok(metadata) + if metadata.file_type().is_dir() && metadata.permissions().mode() & 0o777 == 0o700 => {} + Ok(_) => return Err(LivenessError::Invalid), + Err(error) if error.kind() == io::ErrorKind::NotFound => { + fs::DirBuilder::new().mode(0o700).create(&directory)?; + } + Err(error) => return Err(error.into()), + } + let path = directory.join(SOCKET); + if fs::symlink_metadata(&path).is_ok() { + return Err(LivenessError::Invalid); + } + let listener = UnixListener::bind(&path)?; + let task = tokio::spawn(async move { + while let Ok((mut connection, _)) = listener.accept().await { + tokio::spawn(async move { + let mut ping = [0_u8; 4]; + if timeout(DEADLINE, connection.read_exact(&mut ping)) + .await + .is_ok_and(|result| result.is_ok()) + && &ping == b"ping" + { + let _ = timeout(DEADLINE, connection.write_all(b"pong")).await; + } + }); + } + }); + Ok(Self { task, path }) + } +} + +impl Drop for LivenessServer { + fn drop(&mut self) { + self.task.abort(); + if fs::symlink_metadata(&self.path).is_ok_and(|metadata| metadata.file_type().is_socket()) { + let _ = fs::remove_file(&self.path); + } + } +} + +/// # Errors +/// Requires a responsive local monitor event loop, not a stale PID snapshot. +pub async fn probe_liveness(run_root: &Path) -> Result<(), LivenessError> { + let path = run_root.join("health").join(SOCKET); + timeout(DEADLINE, async { + let mut stream = UnixStream::connect(path).await?; + stream.write_all(b"ping").await?; + let mut pong = [0_u8; 4]; + stream.read_exact(&mut pong).await?; + if &pong != b"pong" { + return Err(LivenessError::Invalid); + } + Ok(()) + }) + .await + .map_err(|_| LivenessError::Timeout)? +} diff --git a/container/crowdb-monitor/src/main.rs b/container/crowdb-monitor/src/main.rs new file mode 100644 index 000000000..4d1093592 --- /dev/null +++ b/container/crowdb-monitor/src/main.rs @@ -0,0 +1,81 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::path::PathBuf; +use std::time::Duration; + +use clap::{Parser, Subcommand}; +use crowdb_monitor::{probe_liveness, run_preview, show_client_credentials, DeploymentProfile, StatusStore}; + +#[derive(Debug, Parser)] +#[command(name = "crowdb-monitor")] +struct Cli { + #[command(subcommand)] + command: Command, +} + +#[derive(Debug, Subcommand)] +enum Command { + Run { + #[arg(long, default_value = "/opt/crowdb/etc/profile.toml")] + profile: PathBuf, + }, + Validate { + profile: PathBuf, + }, + Liveness { + #[arg(long, default_value = "/opt/crowdb/run")] + run_root: PathBuf, + }, + Readiness { + #[arg(long, default_value = "/opt/crowdb/run")] + run_root: PathBuf, + }, + Credentials { + #[command(subcommand)] + command: CredentialsCommand, + }, +} + +#[derive(Debug, Subcommand)] +enum CredentialsCommand { + Show { + #[arg(long, value_enum)] + format: CredentialFormat, + #[arg(long, default_value = "/opt/crowdb/data")] + data_root: PathBuf, + }, +} + +#[derive(Clone, Copy, Debug, clap::ValueEnum)] +enum CredentialFormat { + Env, +} + +#[tokio::main] +async fn main() -> Result<(), Box> { + let cli = Cli::parse(); + match cli.command { + Command::Run { profile } => run_preview(&profile).await?, + Command::Validate { profile } => { + let profile = DeploymentProfile::load(profile)?; + println!("{}", profile.name); + } + Command::Liveness { run_root } => { + probe_liveness(&run_root).await?; + } + Command::Readiness { run_root } => { + StatusStore::open(&run_root)?.readiness(Duration::from_secs(10))?; + } + Command::Credentials { + command: + CredentialsCommand::Show { + format: CredentialFormat::Env, + data_root, + }, + } => { + print!("{}", show_client_credentials(&data_root)?); + } + } + Ok(()) +} diff --git a/container/crowdb-monitor/src/manifest.rs b/container/crowdb-monitor/src/manifest.rs new file mode 100644 index 000000000..ef5c95eb6 --- /dev/null +++ b/container/crowdb-monitor/src/manifest.rs @@ -0,0 +1,349 @@ +use std::collections::BTreeSet; +use std::fs::{self, File, OpenOptions}; +use std::io::Write; +use std::os::unix::fs::{OpenOptionsExt, PermissionsExt}; +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; +use thiserror::Error; +use uuid::Uuid; + +use crate::layout::is_clean_absolute; + +const MANIFEST_VERSION: u32 = 1; +const MAX_STEPS: usize = 64; +const MAX_MANIFEST_BYTES: u64 = 64 * 1024; + +#[derive(Debug, Error)] +pub enum ManifestError { + #[error("bootstrap storage failed: {0}")] + Io(#[from] std::io::Error), + #[error("bootstrap manifest cannot be decoded: {0}")] + Decode(#[from] serde_json::Error), + #[error("bootstrap manifest is invalid: {0}")] + Invalid(String), +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ManifestState { + Initializing, + Ready, +} + +#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct ManifestStep { + name: String, + complete: bool, + operation_id: Option, + catalog_id: Option, +} + +#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct BootstrapManifest { + version: u32, + state: ManifestState, + profile_digest: String, + config_digest: String, + deployment_id: Uuid, + steps: Vec, +} + +impl BootstrapManifest { + #[must_use] + pub fn state(&self) -> ManifestState { + self.state + } + + #[must_use] + pub fn deployment_id(&self) -> Uuid { + self.deployment_id + } + + #[must_use] + pub fn next_step(&self) -> Option<&str> { + self.steps + .iter() + .find(|step| !step.complete) + .map(|step| step.name.as_str()) + } + + #[must_use] + pub fn step_complete(&self, name: &str) -> Option { + self.steps + .iter() + .find(|step| step.name == name) + .map(|step| step.complete) + } + + #[must_use] + pub fn step_operation(&self, name: &str) -> Option { + self.steps.iter().find(|step| step.name == name)?.operation_id + } + + #[must_use] + pub fn step_catalog(&self, name: &str) -> Option { + self.steps.iter().find(|step| step.name == name)?.catalog_id + } + + /// # Errors + /// Rejects a step outside the persisted bootstrap plan. + pub fn operation_id(&self, step: &str) -> Result<[u8; 16], ManifestError> { + if !self.steps.iter().any(|entry| entry.name == step) { + return invalid("operation step is not part of the bootstrap plan"); + } + let mut digest = Sha256::new(); + digest.update(b"crowdb-monitor-bootstrap-operation-v1\0"); + digest.update(self.deployment_id.as_bytes()); + digest.update(step.as_bytes()); + let result = digest.finalize(); + let mut identity = [0; 16]; + identity.copy_from_slice(&result[..16]); + Ok(identity) + } + + fn validate( + &self, + profile_digest: &str, + config_digest: &str, + steps: &[&str], + ) -> Result<(), ManifestError> { + if self.version != MANIFEST_VERSION + || self.deployment_id.is_nil() + || self.profile_digest != profile_digest + || self.config_digest != config_digest + || self.steps.len() != steps.len() + { + return invalid("version, identity, profile, configuration, or step plan changed"); + } + let mut pending = false; + for (actual, expected) in self.steps.iter().zip(steps) { + if actual.name != *expected + || (pending && actual.complete) + || actual.operation_id.is_some_and(|id| id.get_version_num() != 7) + || actual.catalog_id.is_some_and(|id| id.is_nil()) + || (actual.catalog_id.is_some() && !actual.complete) + { + return invalid("bootstrap step order or completion is invalid"); + } + pending |= !actual.complete; + } + if self.state == ManifestState::Ready && pending { + return invalid("ready manifest has incomplete steps"); + } + Ok(()) + } +} + +pub struct BootstrapSession { + directory: PathBuf, + manifest: BootstrapManifest, +} + +impl BootstrapSession { + /// # Errors + /// Rejects missing, non-empty uninitialized, symlinked, or incompatible data roots. + pub fn open( + data_root: &Path, + profile_bytes: &[u8], + config_bytes: &[u8], + steps: &[&str], + ) -> Result { + validate_plan(steps)?; + if !is_clean_absolute(data_root) || !fs::symlink_metadata(data_root)?.file_type().is_dir() { + return invalid("data root must be an existing absolute directory"); + } + let profile_digest = digest_hex(profile_bytes); + let config_digest = digest_hex(config_bytes); + let directory = data_root.join("bootstrap"); + let path = directory.join("manifest.json"); + if let Ok(metadata) = fs::symlink_metadata(&directory) { + if !metadata.file_type().is_dir() { + return invalid("bootstrap path must be a directory, not a link"); + } + } + if path.exists() { + let metadata = fs::symlink_metadata(&path)?; + if !metadata.file_type().is_file() || metadata.permissions().mode() & 0o777 != 0o600 { + return invalid("manifest must be a regular mode-0600 file"); + } + if metadata.len() > MAX_MANIFEST_BYTES { + return invalid("manifest exceeds the size bound"); + } + let manifest: BootstrapManifest = serde_json::from_slice(&fs::read(&path)?)?; + manifest.validate(&profile_digest, &config_digest, steps)?; + return Ok(Self { directory, manifest }); + } + if fs::read_dir(data_root)?.next().is_some() { + return invalid("non-empty data root has no bootstrap manifest"); + } + fs::create_dir(&directory)?; + File::open(data_root)?.sync_all()?; + let manifest = BootstrapManifest { + version: MANIFEST_VERSION, + state: ManifestState::Initializing, + profile_digest, + config_digest, + deployment_id: Uuid::new_v4(), + steps: steps + .iter() + .map(|name| ManifestStep { + name: (*name).to_owned(), + complete: false, + operation_id: None, + catalog_id: None, + }) + .collect(), + }; + persist(&directory, &manifest)?; + Ok(Self { directory, manifest }) + } + + #[must_use] + pub fn manifest(&self) -> &BootstrapManifest { + &self.manifest + } + + /// # Errors + /// Persists a `UUIDv7` before an external mutation is attempted. + pub fn reserve_operation(&mut self, step: &str) -> Result { + if self.manifest.state == ManifestState::Ready || self.manifest.next_step() != Some(step) { + return invalid("operation step is not current"); + } + if let Some(id) = self.manifest.step_operation(step) { + return Ok(id); + } + let mut updated = self.manifest.clone(); + let entry = updated + .steps + .iter_mut() + .find(|entry| entry.name == step) + .ok_or_else(|| ManifestError::Invalid("operation step is missing".into()))?; + let id = Uuid::now_v7(); + entry.operation_id = Some(id); + persist(&self.directory, &updated)?; + self.manifest = updated; + Ok(id) + } + + /// # Errors + /// Records the observed catalog identity in the same durable step advance. + pub fn complete_catalog_step(&mut self, step: &str, catalog: Uuid) -> Result<(), ManifestError> { + if catalog.is_nil() + || self.manifest.state == ManifestState::Ready + || self.manifest.next_step() != Some(step) + { + return invalid("catalog step or identity is invalid"); + } + let mut updated = self.manifest.clone(); + let entry = updated + .steps + .iter_mut() + .find(|entry| entry.name == step) + .ok_or_else(|| ManifestError::Invalid("catalog step is missing".into()))?; + if entry.operation_id.is_none() { + return invalid("catalog step has no reserved operation"); + } + entry.catalog_id = Some(catalog); + entry.complete = true; + persist(&self.directory, &updated)?; + self.manifest = updated; + Ok(()) + } + + /// # Errors + /// Rejects out-of-order, unknown, or post-ready steps and failed durable writes. + pub fn complete_step(&mut self, step: &str) -> Result<(), ManifestError> { + if self.manifest.state == ManifestState::Ready { + return invalid("ready bootstrap cannot execute creation steps"); + } + let Some(next) = self.manifest.next_step() else { + return invalid("all bootstrap steps are already complete"); + }; + if next != step { + return invalid("bootstrap step is out of order"); + } + let mut updated = self.manifest.clone(); + let Some(entry) = updated.steps.iter_mut().find(|entry| entry.name == step) else { + return invalid("bootstrap step is missing from the manifest"); + }; + entry.complete = true; + persist(&self.directory, &updated)?; + self.manifest = updated; + Ok(()) + } + + /// # Errors + /// Rejects incomplete bootstrap work and failed durable writes. + pub fn mark_ready(&mut self) -> Result<(), ManifestError> { + if self.manifest.state == ManifestState::Ready { + return Ok(()); + } + if self.manifest.next_step().is_some() { + return invalid("bootstrap cannot become ready with incomplete steps"); + } + let mut updated = self.manifest.clone(); + updated.state = ManifestState::Ready; + persist(&self.directory, &updated)?; + self.manifest = updated; + Ok(()) + } +} + +fn validate_plan(steps: &[&str]) -> Result<(), ManifestError> { + if steps.is_empty() || steps.len() > MAX_STEPS { + return invalid("bootstrap step count is outside supported bounds"); + } + let mut names = BTreeSet::new(); + for name in steps { + if name.is_empty() + || name.len() > 64 + || !name + .bytes() + .all(|byte| byte.is_ascii_lowercase() || byte.is_ascii_digit() || byte == b'-') + || !names.insert(*name) + { + return invalid("bootstrap step name is invalid or duplicated"); + } + } + Ok(()) +} + +fn digest_hex(bytes: &[u8]) -> String { + let mut text = String::with_capacity(64); + for byte in Sha256::digest(bytes) { + use std::fmt::Write as _; + write!(text, "{byte:02x}").expect("writing to String cannot fail"); + } + text +} + +fn persist(directory: &Path, manifest: &BootstrapManifest) -> Result<(), ManifestError> { + let bytes = serde_json::to_vec(manifest)?; + if bytes.len() as u64 > MAX_MANIFEST_BYTES { + return invalid("manifest exceeds the size bound"); + } + let temporary = directory.join(format!(".manifest-{}.tmp", Uuid::new_v4())); + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&temporary)?; + let result = (|| { + file.write_all(&bytes)?; + file.sync_all()?; + fs::rename(&temporary, directory.join("manifest.json"))?; + File::open(directory)?.sync_all() + })(); + if result.is_err() { + let _ = fs::remove_file(&temporary); + } + result.map_err(ManifestError::Io) +} + +fn invalid(message: impl Into) -> Result { + Err(ManifestError::Invalid(message.into())) +} diff --git a/container/crowdb-monitor/src/monitor_log.rs b/container/crowdb-monitor/src/monitor_log.rs new file mode 100644 index 000000000..70b00dd5f --- /dev/null +++ b/container/crowdb-monitor/src/monitor_log.rs @@ -0,0 +1,128 @@ +use std::io; +use std::path::Path; +use std::time::{SystemTime, UNIX_EPOCH}; + +use serde::Serialize; +use thiserror::Error; + +use crate::process::log::RotatingLog; +use crate::LogProfile; + +#[derive(Debug, Error)] +pub enum MonitorLogError { + #[error("monitor log I/O failed: {0}")] + Io(#[from] io::Error), + #[error("monitor log serialization failed: {0}")] + Encode(#[from] serde_json::Error), + #[error("monitor clock is invalid")] + Clock, +} + +#[derive(Clone, Copy, Debug, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum MonitorEventKind { + Starting, + Ready, + Draining, + Stopped, + ChildStarted, + ChildStopped, + ProbeFailed, + ChildExited, + ChildStartFailed, + Restarting, + RestartBudgetReset, + ListenerFenceFailed, + RestartExhausted, + BootstrapStepStarted, + BootstrapStepCompleted, + BootstrapFailed, +} + +impl MonitorEventKind { + fn is_warning(self) -> bool { + matches!( + self, + Self::ProbeFailed + | Self::ChildExited + | Self::ChildStartFailed + | Self::Restarting + | Self::ListenerFenceFailed + | Self::RestartExhausted + | Self::BootstrapFailed + ) + } +} + +#[derive(Debug, Serialize)] +pub struct MonitorEvent<'a> { + pub kind: MonitorEventKind, + pub service: Option<&'a str>, + pub pid: Option, + pub attempt: Option, +} + +#[derive(Serialize)] +struct StampedEvent<'a> { + timestamp_ms: u64, + level: &'static str, + #[serde(flatten)] + event: &'a MonitorEvent<'a>, +} + +pub struct MonitorLog { + output: RotatingLog, + mirror_warnings_to_stderr: bool, +} + +impl MonitorLog { + /// # Errors + /// Rejects a symlinked or non-directory monitor log path. + pub async fn open(log_root: &Path, policy: LogProfile) -> Result { + let directory = log_root.join("monitor"); + match tokio::fs::symlink_metadata(&directory).await { + Ok(metadata) if !metadata.file_type().is_dir() => { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "monitor log path is not a directory", + ) + .into()); + } + Ok(_) => {} + Err(error) if error.kind() == io::ErrorKind::NotFound => { + tokio::fs::create_dir(&directory).await?; + } + Err(error) => return Err(error.into()), + } + let mirror_warnings_to_stderr = policy.mirror_warnings_to_stderr; + let output = RotatingLog::open(&directory, "monitor.log", policy).await?; + Ok(Self { + output, + mirror_warnings_to_stderr, + }) + } + + /// # Errors + /// Returns clock, encoding, or durable log write failures. + pub async fn record(&mut self, event: &MonitorEvent<'_>) -> Result<(), MonitorLogError> { + let timestamp_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_err(|_| MonitorLogError::Clock)? + .as_millis() + .try_into() + .map_err(|_| MonitorLogError::Clock)?; + let warning = event.kind.is_warning(); + let mut body = serde_json::to_vec(&StampedEvent { + timestamp_ms, + level: if warning { "warn" } else { "info" }, + event, + })?; + body.push(b'\n'); + self.output.write_record(&body).await?; + self.output.sync().await?; + if warning && self.mirror_warnings_to_stderr { + eprint!("{}", String::from_utf8_lossy(&body)); + } + Ok(()) + } +} diff --git a/container/crowdb-monitor/src/preview.rs b/container/crowdb-monitor/src/preview.rs new file mode 100644 index 000000000..ff72738d0 --- /dev/null +++ b/container/crowdb-monitor/src/preview.rs @@ -0,0 +1,527 @@ +use std::collections::{BTreeMap, BTreeSet}; +use std::fs; +use std::future::Future; +use std::path::Path; +use std::time::Duration; + +use thiserror::Error; +use tokio::time::{sleep, Instant}; + +use crate::{ + disk_step_names, ensure_disk_files, hardware_step_names, iceberg_step_names, kv_step_names, + logical_step_names, render_configs, s3_step_names, verify_chunk_services, verify_diskio_disks, + BootstrapSession, ChunkBootstrapError, CredentialError, DeploymentProfile, DiskBootstrapError, + HardwareBootstrap, HardwareBootstrapError, IcebergBootstrap, IcebergBootstrapError, KvBootstrap, + KvBootstrapError, LivenessError, LivenessServer, LogicalBootstrap, LogicalBootstrapError, ManifestError, + ManifestState, MonitorEvent, MonitorEventKind, MonitorLogError, ProfileError, RenderError, S3Bootstrap, + S3BootstrapError, ServerCredentials, StorageProbeError, Supervisor, SupervisorError, +}; + +const PROFILE_NAME: &str = "single-node-container"; +const MAX_TEMPLATE_BYTES: u64 = 1024 * 1024; + +#[derive(Debug, Error)] +pub enum PreviewError { + #[error("preview profile failed: {0}")] + Profile(#[from] ProfileError), + #[error("preview filesystem failed: {0}")] + Io(#[from] std::io::Error), + #[error("preview configuration failed: {0}")] + Render(#[from] RenderError), + #[error("preview manifest failed: {0}")] + Manifest(#[from] ManifestError), + #[error("preview credentials failed: {0}")] + Credentials(#[from] CredentialError), + #[error("preview liveness service failed: {0}")] + Liveness(#[from] LivenessError), + #[error("preview lifecycle log failed: {0}")] + MonitorLog(#[from] MonitorLogError), + #[error("preview supervision failed: {0}")] + Supervisor(#[from] SupervisorError), + #[error("preview KV bootstrap failed: {0}")] + Kv(#[from] KvBootstrapError), + #[error("preview disk bootstrap failed: {0}")] + Disk(#[from] DiskBootstrapError), + #[error("preview hardware bootstrap failed: {0}")] + Hardware(#[from] HardwareBootstrapError), + #[error("preview logical bootstrap failed: {0}")] + Logical(#[from] LogicalBootstrapError), + #[error("preview disk readiness failed: {0}")] + Storage(#[from] StorageProbeError), + #[error("preview chunk readiness failed: {0}")] + Chunk(#[from] ChunkBootstrapError), + #[error("preview S3 bootstrap failed: {0}")] + S3(#[from] S3BootstrapError), + #[error("preview Iceberg bootstrap failed: {0}")] + Iceberg(#[from] IcebergBootstrapError), + #[error("preview Web authority probe failed: {0}")] + WebAuthority(&'static str), + #[error("preview Web authority rejected bootstrap: {0}")] + WebAuthorityUnavailable(String), + #[error("preview state is invalid: {0}")] + Invalid(&'static str), +} + +/// # Errors +/// Fails closed on incompatible durable state or any unready child. +pub async fn run_preview(profile_path: &Path) -> Result<(), PreviewError> { + let profile_bytes = fs::read(profile_path)?; + let profile = DeploymentProfile::parse( + std::str::from_utf8(&profile_bytes).map_err(|_| PreviewError::Invalid("profile is not UTF-8"))?, + )?; + if profile.name != PROFILE_NAME { + return Err(PreviewError::Invalid( + "run supports only the named preview profile", + )); + } + let config_bytes = config_digest_input(&profile)?; + let steps = step_names(&profile)?; + let step_refs = steps.iter().map(String::as_str).collect::>(); + let mut session = BootstrapSession::open( + &profile.paths.data_root, + &profile_bytes, + &config_bytes, + &step_refs, + )?; + let credentials = if session.manifest().state() == ManifestState::Ready + || session.manifest().step_complete("s3-user") == Some(true) + { + ServerCredentials::load_existing(&profile.paths.data_root)? + } else { + ServerCredentials::load_or_create(&profile.paths.data_root)? + }; + ensure_directory(&profile.paths.run_root)?; + let _liveness = LivenessServer::start(&profile.paths.run_root)?; + ensure_directory(&profile.paths.log_root)?; + let monitor_log_root = profile.paths.log_root.join("monitor"); + ensure_directory(&monitor_log_root)?; + crowdb_rpc_ffi::logging::init_logging( + &monitor_log_root.to_string_lossy(), + "warn", + usize::try_from(profile.logs.max_file_bytes.div_ceil(1024 * 1024)) + .map_err(|_| PreviewError::Invalid("RPC log limit is invalid"))?, + usize::from(profile.logs.max_files), + "rpc", + ); + crowdb_rpc_ffi::logging::add_log_stderr("error"); + let kv_root = kv_root(&profile)?; + if session.manifest().state() == ManifestState::Ready { + require_directory(&profile.paths.data_root.join("kv"))?; + require_directory(&kv_root)?; + } else { + ensure_directory(&profile.paths.data_root.join("kv"))?; + ensure_directory(&kv_root)?; + } + render_configs(&profile, &profile.paths.template_root, &profile.paths.run_root)?; + let management_seed = management_seed(&profile)?; + let mut supervisor = Supervisor::new( + profile.clone(), + session.manifest().deployment_id(), + &profile.paths.log_root, + &profile.paths.run_root, + ) + .await?; + supervisor.require_recovery_validation(); + let startup = async { + bootstrap_services( + &mut supervisor, + &mut session, + &profile, + &credentials, + &management_seed, + ) + .await?; + session.mark_ready()?; + supervisor.mark_ready().await?; + Ok::<(), PreviewError>(()) + } + .await; + if let Err(error) = startup { + supervisor.shutdown().await?; + return Err(error); + } + eprintln!("CROWDB Single-Node Container preview ready; retrieve credentials with crowdb-monitor credentials show --format env"); + let runtime = run_ready_services( + &mut supervisor, + &profile, + &profile_bytes, + &config_bytes, + &step_refs, + &credentials, + &management_seed, + ) + .await; + supervisor.shutdown().await?; + runtime +} + +async fn run_ready_services( + supervisor: &mut Supervisor, + profile: &DeploymentProfile, + profile_bytes: &[u8], + config_bytes: &[u8], + steps: &[&str], + credentials: &ServerCredentials, + management_seed: &str, +) -> Result<(), PreviewError> { + let mut terminate = tokio::signal::unix::signal(tokio::signal::unix::SignalKind::terminate())?; + loop { + tokio::select! { + _ = terminate.recv() => break, + _ = tokio::signal::ctrl_c() => break, + result = supervisor.poll_once() => result?, + } + if supervisor.recovery_pending() { + let recovery_epoch = supervisor.recovery_epoch(); + tokio::select! { + _ = terminate.recv() => break, + _ = tokio::signal::ctrl_c() => break, + result = validate_recovery(supervisor, profile, profile_bytes, config_bytes, steps, credentials, management_seed) => result?, + } + supervisor.poll_once().await?; + if supervisor.recovery_pending() + && supervisor.recovery_epoch() == recovery_epoch + && supervisor + .status() + .services + .values() + .all(|service| service.healthy) + { + supervisor.finish_recovery().await?; + } + } + tokio::select! { + _ = terminate.recv() => break, + _ = tokio::signal::ctrl_c() => break, + () = sleep(Duration::from_secs(1)) => {}, + } + } + Ok(()) +} + +async fn validate_recovery( + supervisor: &mut Supervisor, + profile: &DeploymentProfile, + profile_bytes: &[u8], + config_bytes: &[u8], + steps: &[&str], + credentials: &ServerCredentials, + management_seed: &str, +) -> Result<(), PreviewError> { + let mut session = BootstrapSession::open(&profile.paths.data_root, profile_bytes, config_bytes, steps)?; + if session.manifest().state() != ManifestState::Ready + || session.manifest().deployment_id() != supervisor.status().deployment_id + { + return Err(PreviewError::Invalid("recovered deployment identity differs")); + } + let persisted_credentials = ServerCredentials::load_existing(&profile.paths.data_root)?; + if persisted_credentials.server_env() != credentials.server_env() { + return Err(PreviewError::Invalid("recovered server credentials differ")); + } + require_directory(&kv_root(profile)?)?; + KvBootstrap::new(management_seed)? + .reconcile(&mut session, profile, supervisor.monitor_log_mut()) + .await?; + ensure_disk_files(&mut session, profile, supervisor.monitor_log_mut()).await?; + HardwareBootstrap::new(management_seed.to_owned()) + .reconcile(&mut session, profile, supervisor.monitor_log_mut()) + .await?; + LogicalBootstrap::new(management_seed.to_owned()) + .reconcile(&mut session, profile, supervisor.monitor_log_mut()) + .await?; + verify_diskio_disks(management_seed, profile).await?; + verify_chunk_services(management_seed, profile).await?; + S3Bootstrap::reconcile(&mut session, profile, credentials, supervisor.monitor_log_mut()).await?; + let deadline = Instant::now() + Duration::from_secs(30); + loop { + match IcebergBootstrap::reconcile(&mut session, profile, credentials, supervisor.monitor_log_mut()) + .await + { + Ok(()) => break, + Err(IcebergBootstrapError::Command(_)) if Instant::now() < deadline => { + supervisor.refresh_status()?; + sleep(Duration::from_millis(200)).await; + } + Err(error) => return Err(error.into()), + } + } + verify_web_authority(supervisor, profile).await +} + +async fn bootstrap_services( + supervisor: &mut Supervisor, + session: &mut BootstrapSession, + profile: &DeploymentProfile, + credentials: &ServerCredentials, + management_seed: &str, +) -> Result<(), PreviewError> { + supervisor.start_service("kv", BTreeMap::new()).await?; + KvBootstrap::new(management_seed)? + .reconcile(session, profile, supervisor.monitor_log_mut()) + .await?; + ensure_disk_files(session, profile, supervisor.monitor_log_mut()).await?; + HardwareBootstrap::new(management_seed.to_owned()) + .reconcile(session, profile, supervisor.monitor_log_mut()) + .await?; + LogicalBootstrap::new(management_seed.to_owned()) + .reconcile(session, profile, supervisor.monitor_log_mut()) + .await?; + supervisor.start_service("diskdb", BTreeMap::new()).await?; + supervisor.start_service("diskio", BTreeMap::new()).await?; + verify_bootstrap_probe(supervisor, "diskio-authority", async { + verify_diskio_disks(management_seed, profile) + .await + .map_err(Into::into) + }) + .await?; + supervisor.start_service("chunkdb", BTreeMap::new()).await?; + supervisor.start_service("chunk-kv", BTreeMap::new()).await?; + verify_bootstrap_probe(supervisor, "chunk-authority", async { + verify_chunk_services(management_seed, profile) + .await + .map_err(Into::into) + }) + .await?; + S3Bootstrap::reconcile(session, profile, credentials, supervisor.monitor_log_mut()).await?; + IcebergBootstrap::reconcile(session, profile, credentials, supervisor.monitor_log_mut()).await?; + supervisor + .start_service( + "s3", + BTreeMap::from([("CROWDB_S3_MASTER_KEY".into(), credentials.s3_master_key().into())]), + ) + .await?; + let iceberg_environment = credentials + .server_env() + .lines() + .filter_map(|line| line.split_once('=')) + .filter(|(name, _)| name.starts_with("CROWDB_ICEBERG_")) + .map(|(name, value)| (name.to_owned(), value.to_owned())) + .collect(); + supervisor.start_service("iceberg", iceberg_environment).await?; + supervisor + .start_service( + "web", + BTreeMap::from([( + "CROWDB_ICEBERG_MANAGE_TOKEN".into(), + credentials.iceberg_manage_token().into(), + )]), + ) + .await?; + record_bootstrap_probe( + supervisor, + "web-authority", + MonitorEventKind::BootstrapStepStarted, + ) + .await?; + let web_result = verify_web_authority(supervisor, profile).await; + record_bootstrap_probe( + supervisor, + "web-authority", + if web_result.is_ok() { + MonitorEventKind::BootstrapStepCompleted + } else { + MonitorEventKind::BootstrapFailed + }, + ) + .await?; + web_result?; + Ok(()) +} + +async fn verify_bootstrap_probe( + supervisor: &mut Supervisor, + name: &'static str, + probe: impl Future>, +) -> Result<(), PreviewError> { + record_bootstrap_probe(supervisor, name, MonitorEventKind::BootstrapStepStarted).await?; + let result = probe.await; + let kind = if result.is_ok() { + MonitorEventKind::BootstrapStepCompleted + } else { + MonitorEventKind::BootstrapFailed + }; + record_bootstrap_probe(supervisor, name, kind).await?; + result +} + +async fn record_bootstrap_probe( + supervisor: &mut Supervisor, + name: &'static str, + kind: MonitorEventKind, +) -> Result<(), PreviewError> { + supervisor + .monitor_log_mut() + .record(&MonitorEvent { + kind, + service: Some(name), + pid: None, + attempt: None, + }) + .await?; + Ok(()) +} + +async fn verify_web_authority( + supervisor: &mut Supervisor, + profile: &DeploymentProfile, +) -> Result<(), PreviewError> { + let web = profile + .services + .iter() + .find(|service| service.id == "web") + .ok_or(PreviewError::Invalid("Web service is absent"))?; + let origin = web + .probe + .target + .strip_suffix("/healthz") + .ok_or(PreviewError::Invalid("Web health endpoint is incompatible"))?; + let client = reqwest::Client::builder() + .no_proxy() + .timeout(Duration::from_secs(2)) + .redirect(reqwest::redirect::Policy::none()) + .build() + .map_err(|_| PreviewError::WebAuthority("cannot construct authority probe"))?; + let deadline = Instant::now() + Duration::from_secs(30); + let mut last_status_refresh = Instant::now(); + loop { + if last_status_refresh.elapsed() >= Duration::from_secs(5) { + supervisor.refresh_status()?; + last_status_refresh = Instant::now(); + } + let response = client + .get(format!("{origin}/api/authority")) + .send() + .await + .map_err(|_| PreviewError::WebAuthority("authority endpoint is unavailable"))?; + let status = response.status(); + let body: serde_json::Value = response + .json() + .await + .map_err(|_| PreviewError::WebAuthority("authority response is invalid"))?; + if status.is_success() { + if body.get("source").and_then(serde_json::Value::as_str) == Some("group0") + && body.get("available").and_then(serde_json::Value::as_bool) == Some(true) + { + return Ok(()); + } + return Err(PreviewError::WebAuthority("Web is not serving Group 0 authority")); + } + let reason = body + .get("reason") + .and_then(serde_json::Value::as_str) + .unwrap_or("unspecified"); + if reason != "group0_unavailable" || Instant::now() >= deadline { + return Err(PreviewError::WebAuthorityUnavailable(format!( + "HTTP {status}: {reason}" + ))); + } + sleep(Duration::from_millis(200)).await; + } +} + +fn step_names(profile: &DeploymentProfile) -> Result, PreviewError> { + let steps = kv_step_names(profile)? + .into_iter() + .chain(disk_step_names(profile)) + .chain(hardware_step_names()) + .chain(logical_step_names().map(str::to_owned)) + .chain(s3_step_names().map(str::to_owned)) + .chain(iceberg_step_names().map(str::to_owned)) + .collect(); + Ok(steps) +} + +fn management_seed(profile: &DeploymentProfile) -> Result { + let service = profile + .services + .iter() + .find(|service| service.id == "s3") + .ok_or(PreviewError::Invalid("S3 service is absent"))?; + let seeds = service + .env + .get("CROWDB_MANAGEMENT_SEEDS") + .ok_or(PreviewError::Invalid("management seeds are absent"))?; + let parts = seeds.split(',').collect::>(); + let [seed] = parts.as_slice() else { + return Err(PreviewError::Invalid("preview requires one management seed")); + }; + if seed.is_empty() { + return Err(PreviewError::Invalid("management seed is empty")); + } + Ok((*seed).to_owned()) +} + +fn kv_root(profile: &DeploymentProfile) -> Result { + let service = profile + .services + .iter() + .find(|service| service.id == "kv") + .ok_or(PreviewError::Invalid("KV service is absent"))?; + let mut roots = service.args.windows(2).filter(|pair| pair[0] == "--root"); + let path = roots + .next() + .ok_or(PreviewError::Invalid("KV root argument is absent"))?[1] + .as_str(); + if roots.next().is_some() { + return Err(PreviewError::Invalid("KV root argument is duplicated")); + } + let path = Path::new(path); + if !path.starts_with(profile.paths.data_root.join("kv")) + || path == profile.paths.data_root.join("kv") + || !crate::layout::is_clean_absolute(path) + { + return Err(PreviewError::Invalid( + "KV root is outside the durable KV directory", + )); + } + Ok(path.to_owned()) +} + +fn config_digest_input(profile: &DeploymentProfile) -> Result, PreviewError> { + let mut files = BTreeSet::new(); + for service in &profile.services { + if let Some(path) = &service.config_template { + let name = path + .file_name() + .ok_or(PreviewError::Invalid("template has no name"))?; + if !files.insert(name.to_os_string()) { + return Err(PreviewError::Invalid("template name is duplicated")); + } + } + } + let mut input = Vec::new(); + for name in files { + let path = profile.paths.template_root.join(&name); + let metadata = fs::symlink_metadata(&path)?; + if !metadata.file_type().is_file() || metadata.len() > MAX_TEMPLATE_BYTES { + return Err(PreviewError::Invalid("template is not a bounded regular file")); + } + let body = fs::read(&path)?; + for field in [name.as_encoded_bytes(), body.as_slice()] { + input.extend_from_slice(&(field.len() as u64).to_be_bytes()); + input.extend_from_slice(field); + } + } + Ok(input) +} + +fn ensure_directory(path: &Path) -> Result<(), PreviewError> { + match fs::symlink_metadata(path) { + Ok(metadata) if !metadata.file_type().is_dir() => { + Err(PreviewError::Invalid("runtime path is not a directory")) + } + Ok(_) => Ok(()), + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + fs::create_dir(path)?; + Ok(()) + } + Err(error) => Err(error.into()), + } +} + +fn require_directory(path: &Path) -> Result<(), PreviewError> { + if !fs::symlink_metadata(path)?.file_type().is_dir() { + return Err(PreviewError::Invalid("required durable directory is missing")); + } + Ok(()) +} diff --git a/container/crowdb-monitor/src/probe.rs b/container/crowdb-monitor/src/probe.rs new file mode 100644 index 000000000..36677114f --- /dev/null +++ b/container/crowdb-monitor/src/probe.rs @@ -0,0 +1,165 @@ +use std::collections::BTreeMap; +use std::net::SocketAddr; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::Duration; + +use crowdb_protocol::fb::{ConnectionPingRequest, ConnectionPingRequestArgs, FBMsgType}; +use crowdb_rpc_ffi::{Buffer, RpcClient, RpcServer}; +use flatbuffers::FlatBufferBuilder; +use thiserror::Error; +use tokio::net::TcpStream; +use tokio::time::timeout; + +use crate::{ProbeKind, ServiceProfile}; + +#[derive(Debug, Error)] +pub enum ProbeError { + #[error("probe target is invalid")] + InvalidTarget, + #[error("probe timed out")] + Timeout, + #[error("probe endpoint is unavailable")] + Unavailable, + #[error("probe credential is absent")] + MissingCredential, +} + +pub struct ProbeExecutor { + client: reqwest::Client, + rpc: Option, +} + +struct RpcProbe { + client: RpcClient, + server: RpcServer, + request_id: AtomicU64, +} + +impl RpcProbe { + fn new() -> Result { + let server = RpcServer::new(None); + server + .listen("127.0.0.1", 0) + .map_err(|_| ProbeError::Unavailable)?; + server.start(); + let client = RpcClient::new(); + client.set_completion_pool_size(128); + client.start_reaper(2_000_000_000, 100_000_000); + Ok(Self { + client, + server, + request_id: AtomicU64::new(1), + }) + } + + async fn ping(&self, address: SocketAddr, duration: Duration) -> Result<(), ProbeError> { + let connection = self + .server + .connect(&address.ip().to_string(), i32::from(address.port())) + .map_err(|_| ProbeError::Unavailable)?; + self.client.attach(&connection); + let request_id = self.request_id.fetch_add(1, Ordering::Relaxed); + let mut builder = FlatBufferBuilder::new(); + let request = ConnectionPingRequest::create( + &mut builder, + &ConnectionPingRequestArgs { + id: request_id, + rpc_create_nano: 0, + }, + ); + builder.finish(request, None); + let control = Buffer::from_bytes(builder.finished_data()); + let response = self + .client + .call( + &self.server, + &connection, + request_id, + control, + None, + FBMsgType::EConnectionPingRequest.0 as u16, + ) + .map_err(|_| ProbeError::Unavailable)?; + let response = timeout(duration, response) + .await + .map_err(|_| ProbeError::Timeout)? + .map_err(|_| ProbeError::Unavailable)?; + if response.request_id == request_id { + Ok(()) + } else { + Err(ProbeError::Unavailable) + } + } +} + +impl ProbeExecutor { + /// # Errors + /// Rejects invalid HTTP client configuration. + pub fn new(enable_rpc: bool) -> Result { + let client = reqwest::Client::builder() + .no_proxy() + .redirect(reqwest::redirect::Policy::none()) + .build() + .map_err(|_| ProbeError::InvalidTarget)?; + let rpc = enable_rpc.then(RpcProbe::new).transpose()?; + Ok(Self { client, rpc }) + } + + /// # Errors + /// Rejects malformed, timed-out, non-success, and unreachable endpoints. + pub async fn probe_service( + &self, + service: &ServiceProfile, + environment: &BTreeMap, + ) -> Result<(), ProbeError> { + let duration = Duration::from_millis(service.probe.timeout_ms); + match service.probe.kind { + ProbeKind::Tcp => { + let address: SocketAddr = service + .probe + .target + .parse() + .map_err(|_| ProbeError::InvalidTarget)?; + timeout(duration, TcpStream::connect(address)) + .await + .map_err(|_| ProbeError::Timeout)? + .map_err(|_| ProbeError::Unavailable)?; + Ok(()) + } + ProbeKind::RpcPing => { + let address = service + .probe + .target + .parse() + .map_err(|_| ProbeError::InvalidTarget)?; + self.rpc + .as_ref() + .ok_or(ProbeError::Unavailable)? + .ping(address, duration) + .await + } + ProbeKind::Http => { + let mut request = self.client.get(&service.probe.target).timeout(duration); + if let Some(name) = &service.probe.bearer_env { + let token = environment + .get(name) + .filter(|token| !token.is_empty()) + .ok_or(ProbeError::MissingCredential)?; + request = request.bearer_auth(token); + } + let response = request.send().await.map_err(|error| { + if error.is_timeout() { + ProbeError::Timeout + } else { + ProbeError::Unavailable + } + })?; + if response.status().is_success() { + Ok(()) + } else { + Err(ProbeError::Unavailable) + } + } + } + } +} diff --git a/container/crowdb-monitor/src/process.rs b/container/crowdb-monitor/src/process.rs new file mode 100644 index 000000000..e8b92c1ad --- /dev/null +++ b/container/crowdb-monitor/src/process.rs @@ -0,0 +1,248 @@ +pub(crate) mod log; +mod retention; + +use std::collections::BTreeMap; +use std::io; +use std::path::PathBuf; +use std::process::Stdio; +use std::time::Duration; + +use rustix::process::{kill_process, Pid, Signal}; +use thiserror::Error; +use tokio::process::{Child, Command}; +use tokio::task::JoinHandle; +use tokio::time::{sleep, timeout}; + +use crate::{LogProfile, MonitorEvent, MonitorEventKind, MonitorLog, MonitorLogError, ServiceProfile}; + +#[derive(Debug, Error)] +pub enum ProcessError { + #[error("child process I/O failed: {0}")] + Io(#[from] io::Error), + #[error("process log task failed: {0}")] + Log(#[from] tokio::task::JoinError), + #[error("monitor lifecycle log failed: {0}")] + MonitorLog(#[from] MonitorLogError), + #[error("process state is invalid: {0}")] + Invalid(&'static str), +} + +struct ManagedProcess { + child: Child, + logger: JoinHandle>, +} + +pub struct ProcessManager { + processes: BTreeMap, + log_root: PathBuf, + log_policy: LogProfile, + events: MonitorLog, +} + +impl ProcessManager { + /// # Errors + /// Rejects an unavailable monitor lifecycle log. + pub async fn new(log_root: PathBuf, log_policy: LogProfile) -> Result { + let mut events = MonitorLog::open(&log_root, log_policy.clone()).await?; + events + .record(&MonitorEvent { + kind: MonitorEventKind::Starting, + service: None, + pid: Some(std::process::id()), + attempt: None, + }) + .await?; + Ok(Self { + processes: BTreeMap::new(), + log_root, + log_policy, + events, + }) + } + + /// # Errors + /// Rejects overlapping child ownership or failed spawn/log setup. + pub async fn start( + &mut self, + service: &ServiceProfile, + environment: &BTreeMap, + ) -> Result { + if self.processes.contains_key(&service.id) { + return Err(ProcessError::Invalid("service already has an owned process")); + } + let log_directory = self.log_root.join(&service.id); + std::fs::create_dir_all(&log_directory)?; + retention::prune(&log_directory, None, self.log_policy.max_files.saturating_sub(1)).await?; + let mut command = Command::new(&service.program); + command + .args(&service.args) + .envs(&service.env) + .envs(environment) + .stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .kill_on_drop(true); + let mut child = command.spawn()?; + let pid = child + .id() + .ok_or(ProcessError::Invalid("spawned child has no PID"))?; + let stdout = child + .stdout + .take() + .ok_or(ProcessError::Invalid("child stdout is unavailable"))?; + let stderr = child + .stderr + .take() + .ok_or(ProcessError::Invalid("child stderr is unavailable"))?; + let policy = self.log_policy.clone(); + let logger = + tokio::spawn(async move { Box::pin(log::pump(stdout, stderr, &log_directory, policy)).await }); + self.processes + .insert(service.id.clone(), ManagedProcess { child, logger }); + if let Err(error) = self + .events + .record(&MonitorEvent { + kind: MonitorEventKind::ChildStarted, + service: Some(&service.id), + pid: Some(pid), + attempt: None, + }) + .await + { + let _ = self.stop(&service.id, Duration::from_secs(2)).await; + return Err(error.into()); + } + Ok(pid) + } + + #[must_use] + pub fn pid(&self, id: &str) -> Option { + self.processes.get(id).and_then(|process| process.child.id()) + } + + #[must_use] + pub fn owns(&self, id: &str) -> bool { + self.processes.contains_key(id) + } + + pub(crate) async fn maintain_logs(&self) -> Result<(), ProcessError> { + for (id, process) in &self.processes { + retention::prune( + &self.log_root.join(id), + process.child.id(), + self.log_policy.max_files, + ) + .await?; + } + Ok(()) + } + + /// # Errors + /// Returns process observation failures. A completed process remains owned until stopped. + pub fn alive(&mut self, id: &str) -> Result { + let process = self + .processes + .get_mut(id) + .ok_or(ProcessError::Invalid("service is not owned"))?; + if process.logger.is_finished() { + return Ok(false); + } + Ok(process.child.try_wait()?.is_none()) + } + + /// # Errors + /// Sends TERM, waits for the owned PID, escalates to KILL after the deadline, and joins logs. + pub async fn stop(&mut self, id: &str, grace: Duration) -> Result<(), ProcessError> { + let mut process = self + .processes + .remove(id) + .ok_or(ProcessError::Invalid("service is not owned"))?; + if process.child.try_wait()?.is_none() { + if let Some(pid) = process + .child + .id() + .and_then(|value| i32::try_from(value).ok()) + .and_then(Pid::from_raw) + { + if let Err(error) = kill_process(pid, Signal::TERM) { + if error != rustix::io::Errno::SRCH { + return Err(io::Error::from_raw_os_error(error.raw_os_error()).into()); + } + } + } + if timeout(grace, process.child.wait()).await.is_err() { + process.child.start_kill()?; + process.child.wait().await?; + } + } + let log_result = timeout(Duration::from_secs(5), &mut process.logger).await; + if let Ok(joined) = log_result { + joined?.map_err(ProcessError::Io)?; + } else { + process.logger.abort(); + return Err(ProcessError::Invalid("child log pipes did not close after exit")); + } + self.events + .record(&MonitorEvent { + kind: MonitorEventKind::ChildStopped, + service: Some(id), + pid: None, + attempt: None, + }) + .await?; + retention::prune(&self.log_root.join(id), None, self.log_policy.max_files).await?; + Ok(()) + } + + /// # Errors + /// Stops all owned children in reverse start order. + pub async fn stop_all(&mut self, order: &[String], grace: Duration) -> Result<(), ProcessError> { + let mut first_error = None; + for id in order.iter().rev() { + if self.processes.contains_key(id) { + if let Err(error) = self.stop(id, grace).await { + first_error.get_or_insert(error); + } + } + } + if let Some(error) = first_error { + return Err(error); + } + Ok(()) + } + + /// # Errors + /// Rejects a still-serving endpoint after its previous process was reaped. + pub async fn wait_listener_closed(&self, address: &str, deadline: Duration) -> Result<(), ProcessError> { + let address = address + .parse::() + .map_err(|_| ProcessError::Invalid("listener address is invalid"))?; + let until = tokio::time::Instant::now() + deadline; + loop { + if timeout( + Duration::from_millis(200), + tokio::net::TcpStream::connect(address), + ) + .await + .is_ok_and(|result| result.is_err()) + { + return Ok(()); + } + if tokio::time::Instant::now() >= until { + return Err(ProcessError::Invalid("listener remains owned after process exit")); + } + sleep(Duration::from_millis(50)).await; + } + } + + /// # Errors + /// Returns failed durable monitor event writes. + pub async fn record_event(&mut self, event: &MonitorEvent<'_>) -> Result<(), ProcessError> { + self.events.record(event).await?; + Ok(()) + } + + pub fn monitor_log_mut(&mut self) -> &mut MonitorLog { + &mut self.events + } +} diff --git a/container/crowdb-monitor/src/process/log.rs b/container/crowdb-monitor/src/process/log.rs new file mode 100644 index 000000000..79e072525 --- /dev/null +++ b/container/crowdb-monitor/src/process/log.rs @@ -0,0 +1,141 @@ +use std::io; +use std::path::{Path, PathBuf}; + +use tokio::fs::{self, File, OpenOptions}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::process::{ChildStderr, ChildStdout}; + +use crate::LogProfile; + +pub(super) async fn pump( + mut stdout: ChildStdout, + mut stderr: ChildStderr, + directory: &Path, + policy: LogProfile, +) -> io::Result<()> { + let mut output = RotatingLog::open(directory, "service.log", policy).await?; + let mut mirrored_stderr = tokio::io::stderr(); + let mut stdout_open = true; + let mut stderr_open = true; + let mut stdout_buffer = [0_u8; 8192]; + let mut stderr_buffer = [0_u8; 8192]; + while stdout_open || stderr_open { + tokio::select! { + read = stdout.read(&mut stdout_buffer), if stdout_open => { + let size = read?; + stdout_open = size != 0; + if size != 0 { + output.write(&stdout_buffer[..size]).await?; + } + } + read = stderr.read(&mut stderr_buffer), if stderr_open => { + let size = read?; + stderr_open = size != 0; + if size != 0 { + output.write(&stderr_buffer[..size]).await?; + if output.policy.mirror_warnings_to_stderr { + mirrored_stderr.write_all(&stderr_buffer[..size]).await?; + } + } + } + } + } + output.sync().await +} + +pub(crate) struct RotatingLog { + directory: PathBuf, + name: String, + policy: LogProfile, + file: File, + size: u64, +} + +impl RotatingLog { + pub(crate) async fn open(directory: &Path, name: &str, policy: LogProfile) -> io::Result { + fs::create_dir_all(directory).await?; + let path = directory.join(name); + let size = fs::metadata(&path).await.map_or(0, |metadata| metadata.len()); + let file = OpenOptions::new().create(true).append(true).open(&path).await?; + Ok(Self { + directory: directory.to_owned(), + name: name.to_owned(), + policy, + file, + size, + }) + } + + pub(crate) async fn write(&mut self, mut bytes: &[u8]) -> io::Result<()> { + while !bytes.is_empty() { + if self.size >= self.policy.max_file_bytes { + self.rotate().await?; + } + let room = usize::try_from(self.policy.max_file_bytes - self.size).unwrap_or(usize::MAX); + let count = bytes.len().min(room); + self.file.write_all(&bytes[..count]).await?; + self.size += count as u64; + bytes = &bytes[count..]; + } + Ok(()) + } + + pub(crate) async fn sync(&self) -> io::Result<()> { + self.file.sync_all().await + } + + pub(crate) async fn write_record(&mut self, bytes: &[u8]) -> io::Result<()> { + if bytes.len() as u64 > self.policy.max_file_bytes { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "log record exceeds file limit", + )); + } + if self.size > self.policy.max_file_bytes - bytes.len() as u64 { + self.rotate().await?; + } + self.write(bytes).await + } + + async fn rotate(&mut self) -> io::Result<()> { + self.file.sync_all().await?; + let current = self.directory.join(&self.name); + if self.policy.max_files == 1 { + remove_if_exists(¤t).await?; + } else { + let oldest = self + .directory + .join(format!("{}.{}", self.name, self.policy.max_files - 1)); + remove_if_exists(&oldest).await?; + for index in (1..self.policy.max_files - 1).rev() { + let source = self.directory.join(format!("{}.{index}", self.name)); + let destination = self.directory.join(format!("{}.{}", self.name, index + 1)); + rename_if_exists(&source, &destination).await?; + } + rename_if_exists(¤t, &self.directory.join(format!("{}.1", self.name))).await?; + } + self.file = OpenOptions::new() + .create(true) + .append(true) + .open(¤t) + .await?; + self.size = 0; + Ok(()) + } +} + +async fn remove_if_exists(path: &Path) -> io::Result<()> { + match fs::remove_file(path).await { + Ok(()) => Ok(()), + Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(()), + Err(error) => Err(error), + } +} + +async fn rename_if_exists(source: &Path, destination: &Path) -> io::Result<()> { + match fs::rename(source, destination).await { + Ok(()) => Ok(()), + Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(()), + Err(error) => Err(error), + } +} diff --git a/container/crowdb-monitor/src/process/retention.rs b/container/crowdb-monitor/src/process/retention.rs new file mode 100644 index 000000000..4e042d86e --- /dev/null +++ b/container/crowdb-monitor/src/process/retention.rs @@ -0,0 +1,80 @@ +use std::collections::BTreeMap; +use std::io; +use std::path::{Path, PathBuf}; +use std::time::SystemTime; + +struct LogFile { + path: PathBuf, + modified: SystemTime, + active: bool, +} + +pub(super) async fn prune(directory: &Path, current_pid: Option, max_files: u16) -> io::Result<()> { + let mut entries = tokio::fs::read_dir(directory).await?; + let mut groups: BTreeMap> = BTreeMap::new(); + while let Some(entry) = entries.next_entry().await? { + let name = entry.file_name(); + let Some((prefix, pid, compressed)) = name.to_str().and_then(classify) else { + continue; + }; + let metadata = match tokio::fs::symlink_metadata(entry.path()).await { + Ok(metadata) => metadata, + Err(error) if error.kind() == io::ErrorKind::NotFound => continue, + Err(error) => return Err(error), + }; + if !metadata.is_file() { + continue; + } + groups.entry(prefix.to_owned()).or_default().push(LogFile { + path: entry.path(), + modified: metadata.modified()?, + active: current_pid == Some(pid) && !compressed, + }); + } + for files in groups.values_mut() { + files.sort_by(|left, right| { + right + .active + .cmp(&left.active) + .then_with(|| right.modified.cmp(&left.modified)) + .then_with(|| right.path.cmp(&left.path)) + }); + for file in files.iter().skip(usize::from(max_files)) { + if !file.active { + match tokio::fs::remove_file(&file.path).await { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::NotFound => {} + Err(error) => return Err(error), + } + } + } + } + Ok(()) +} + +fn classify(name: &str) -> Option<(&str, u32, bool)> { + let (stem, compressed) = match name.strip_suffix(".log.gz") { + Some(stem) => (stem, true), + None => (name.strip_suffix(".log")?, false), + }; + let (prefix, pid) = stem.rsplit_once('-')?; + let pid = pid.parse().ok()?; + let prefix = match prefix.rsplit_once('-') { + Some((dated, time)) + if time.len() == 10 + && time.as_bytes()[6] == b'.' + && time + .bytes() + .enumerate() + .all(|(index, byte)| index == 6 || byte.is_ascii_digit()) => + { + let (prefix, date) = dated.rsplit_once('-')?; + if date.len() != 8 || !date.bytes().all(|byte| byte.is_ascii_digit()) { + return None; + } + prefix + } + _ => prefix, + }; + Some((prefix, pid, compressed)) +} diff --git a/container/crowdb-monitor/src/profile.rs b/container/crowdb-monitor/src/profile.rs new file mode 100644 index 000000000..b39a921a1 --- /dev/null +++ b/container/crowdb-monitor/src/profile.rs @@ -0,0 +1,188 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +mod validation; + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; +use thiserror::Error; + +pub const PROFILE_VERSION: u32 = 1; + +#[derive(Debug, Error)] +pub enum ProfileError { + #[error("failed to read deployment profile: {0}")] + Io(#[from] std::io::Error), + #[error("failed to decode deployment profile: {0}")] + Decode(#[from] toml::de::Error), + #[error("invalid deployment profile: {0}")] + Invalid(String), +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct DeploymentProfile { + pub version: u32, + pub name: String, + pub display_name: String, + pub placement_mode: String, + pub s3_tenant: String, + pub iceberg_catalog: String, + pub paths: PathProfile, + pub logs: LogProfile, + #[serde(default)] + pub nodes: Vec, + #[serde(default)] + pub groups: Vec, + #[serde(default)] + pub disks: Vec, + #[serde(default)] + pub public_endpoints: Vec, + #[serde(default)] + pub services: Vec, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PathProfile { + pub install_root: PathBuf, + pub bin_root: PathBuf, + pub template_root: PathBuf, + pub data_root: PathBuf, + pub run_root: PathBuf, + pub log_root: PathBuf, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct LogProfile { + pub max_file_bytes: u64, + pub max_files: u16, + pub mirror_warnings_to_stderr: bool, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct NodeProfile { + pub node_id: u64, + pub rack_id: u64, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum GroupRole { + System, + Data, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct GroupProfile { + pub store_id: u64, + pub group_id: u64, + pub replica_id: u64, + pub node_id: u64, + pub rpc_endpoint: String, + pub role: GroupRole, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct DiskProfile { + pub disk_id: String, + pub disk_group_id: u64, + pub node_id: u64, + pub path: PathBuf, + pub capacity_bytes: u64, + pub zone_size_bytes: u64, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PublicEndpoint { + pub id: String, + pub bind: String, + pub port: u16, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum ProbeKind { + Http, + Tcp, + RpcPing, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ProbeProfile { + pub kind: ProbeKind, + pub target: String, + #[serde(default)] + pub bearer_env: Option, + pub timeout_ms: u64, + pub failure_threshold: u32, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct RestartProfile { + pub max_attempts: u32, + pub backoff_base_ms: u64, + pub backoff_max_ms: u64, + #[serde(default = "default_stable_after_ms")] + pub stable_after_ms: u64, +} + +const fn default_stable_after_ms() -> u64 { + 60_000 +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ServiceProfile { + pub id: String, + pub program: PathBuf, + #[serde(default)] + pub args: Vec, + #[serde(default)] + pub env: BTreeMap, + #[serde(default)] + pub dependencies: Vec, + #[serde(default)] + pub fence_listeners: Vec, + pub config_template: Option, + pub probe: ProbeProfile, + pub restart: RestartProfile, +} + +impl DeploymentProfile { + /// # Errors + /// Returns an error when the profile cannot be read, parsed, or validated. + pub fn load(path: impl AsRef) -> Result { + let body = std::fs::read_to_string(path)?; + Self::parse(&body) + } + + /// # Errors + /// Returns an error when the profile cannot be parsed or validated. + pub fn parse(body: &str) -> Result { + let profile: Self = toml::from_str(body)?; + profile.validate()?; + Ok(profile) + } + + /// # Errors + /// Returns an error when the profile violates deployment constraints. + pub fn validate(&self) -> Result<(), ProfileError> { + validation::validate(self) + } + + /// # Errors + /// Returns an error when the profile or its service dependencies are invalid. + pub fn services_in_start_order(&self) -> Result, ProfileError> { + validation::services_in_start_order(self) + } +} diff --git a/container/crowdb-monitor/src/profile/validation.rs b/container/crowdb-monitor/src/profile/validation.rs new file mode 100644 index 000000000..ced408cad --- /dev/null +++ b/container/crowdb-monitor/src/profile/validation.rs @@ -0,0 +1,309 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::{BTreeMap, BTreeSet}; +use std::net::{IpAddr, SocketAddr}; + +use super::{DeploymentProfile, GroupRole, ProbeKind, ProfileError, ServiceProfile, PROFILE_VERSION}; +use crate::layout::{is_clean_absolute, is_strict_descendant}; + +pub(super) fn validate(profile: &DeploymentProfile) -> Result<(), ProfileError> { + if profile.version != PROFILE_VERSION { + return invalid(format!( + "unsupported version {}; expected {PROFILE_VERSION}", + profile.version + )); + } + require_slug("profile name", &profile.name)?; + require_text("display name", &profile.display_name)?; + require_text("placement mode", &profile.placement_mode)?; + require_text("S3 tenant", &profile.s3_tenant)?; + require_text("Iceberg catalog", &profile.iceberg_catalog)?; + validate_paths(profile)?; + validate_logs(profile)?; + validate_topology(profile)?; + validate_endpoints(profile)?; + validate_services(profile)?; + topological_order(profile)?; + Ok(()) +} + +fn validate_logs(profile: &DeploymentProfile) -> Result<(), ProfileError> { + if !(1024 * 1024..=1024 * 1024 * 1024).contains(&profile.logs.max_file_bytes) + || !(1..=16).contains(&profile.logs.max_files) + { + return invalid("log rotation limits are outside supported bounds"); + } + Ok(()) +} + +pub(super) fn services_in_start_order( + profile: &DeploymentProfile, +) -> Result, ProfileError> { + validate(profile)?; + topological_order(profile) +} + +fn validate_paths(profile: &DeploymentProfile) -> Result<(), ProfileError> { + let paths = &profile.paths; + for (name, path) in [ + ("install_root", paths.install_root.as_path()), + ("bin_root", paths.bin_root.as_path()), + ("template_root", paths.template_root.as_path()), + ("data_root", paths.data_root.as_path()), + ("run_root", paths.run_root.as_path()), + ("log_root", paths.log_root.as_path()), + ] { + if !is_clean_absolute(path) { + return invalid(format!("{name} must be a clean absolute path")); + } + } + for (name, path) in [ + ("bin_root", paths.bin_root.as_path()), + ("template_root", paths.template_root.as_path()), + ("data_root", paths.data_root.as_path()), + ("run_root", paths.run_root.as_path()), + ] { + if !is_strict_descendant(&paths.install_root, path) { + return invalid(format!("{name} must be below install_root")); + } + } + if !is_strict_descendant(&paths.data_root, &paths.log_root) { + return invalid("log_root must be below data_root"); + } + if paths.data_root.starts_with(&paths.run_root) || paths.run_root.starts_with(&paths.data_root) { + return invalid("data_root and run_root must not overlap"); + } + Ok(()) +} + +fn validate_topology(profile: &DeploymentProfile) -> Result<(), ProfileError> { + if profile.nodes.is_empty() || profile.groups.is_empty() || profile.disks.is_empty() { + return invalid("nodes, groups, and disks must be non-empty"); + } + let mut node_ids = BTreeSet::new(); + for node in &profile.nodes { + if node.node_id == 0 || node.rack_id == 0 || !node_ids.insert(node.node_id) { + return invalid("node and rack IDs must be nonzero and node IDs unique"); + } + } + let mut groups = BTreeSet::new(); + let mut system_groups = 0_u32; + for group in &profile.groups { + if group.replica_id == 0 || !groups.insert((group.store_id, group.group_id)) { + return invalid("group identities must be unique and replica IDs nonzero"); + } + if !node_ids.contains(&group.node_id) || group.rpc_endpoint.parse::().is_err() { + return invalid("group replica node or RPC endpoint is invalid"); + } + if group.role == GroupRole::System { + system_groups += 1; + } + } + if system_groups != 1 { + return invalid("exactly one system group is required"); + } + let disk_root = profile.paths.data_root.join("disks"); + let mut disk_ids = BTreeSet::new(); + let mut disk_paths = BTreeSet::new(); + for disk in &profile.disks { + if !node_ids.contains(&disk.node_id) { + return invalid(format!("disk {} references an unknown node", disk.disk_id)); + } + if disk.disk_id.is_empty() || !disk_ids.insert(&disk.disk_id) || !disk_paths.insert(&disk.path) { + return invalid("disk identities and paths must be non-empty and unique"); + } + if disk.disk_group_id == 0 || disk.capacity_bytes == 0 || disk.zone_size_bytes == 0 { + return invalid(format!("disk {} has invalid capacity or group", disk.disk_id)); + } + if disk.capacity_bytes % disk.zone_size_bytes != 0 { + return invalid(format!("disk {} capacity must contain whole zones", disk.disk_id)); + } + if !is_strict_descendant(&disk_root, &disk.path) { + return invalid(format!( + "disk {} path must be below data_root/disks", + disk.disk_id + )); + } + } + Ok(()) +} + +fn validate_endpoints(profile: &DeploymentProfile) -> Result<(), ProfileError> { + if profile.public_endpoints.is_empty() { + return invalid("at least one public endpoint is required"); + } + let mut ids = BTreeSet::new(); + let mut listeners = BTreeSet::new(); + for endpoint in &profile.public_endpoints { + require_slug("public endpoint ID", &endpoint.id)?; + let address = endpoint.bind.parse::().map_err(|_| { + ProfileError::Invalid(format!("endpoint {} has invalid bind address", endpoint.id)) + })?; + if endpoint.port == 0 || !ids.insert(&endpoint.id) || !listeners.insert((address, endpoint.port)) { + return invalid("public endpoint IDs and listeners must be unique and nonzero"); + } + } + Ok(()) +} + +fn validate_services(profile: &DeploymentProfile) -> Result<(), ProfileError> { + if profile.services.is_empty() { + return invalid("at least one service is required"); + } + let ids = profile + .services + .iter() + .map(|service| service.id.as_str()) + .collect::>(); + if ids.len() != profile.services.len() { + return invalid("service IDs must be unique"); + } + for service in &profile.services { + require_slug("service ID", &service.id)?; + if !is_strict_descendant(&profile.paths.bin_root, &service.program) { + return invalid(format!("service {} program must be below bin_root", service.id)); + } + if let Some(template) = &service.config_template { + if !is_strict_descendant(&profile.paths.template_root, template) { + return invalid(format!( + "service {} template must be below template_root", + service.id + )); + } + } + let mut dependencies = BTreeSet::new(); + for dependency in &service.dependencies { + if dependency == &service.id + || !ids.contains(dependency.as_str()) + || !dependencies.insert(dependency) + { + return invalid(format!("service {} has an invalid dependency", service.id)); + } + } + for name in service.env.keys() { + let upper = name.to_ascii_uppercase(); + if ["SECRET", "TOKEN", "PASSWORD", "MASTER_KEY", "ACCESS_KEY"] + .iter() + .any(|marker| upper.contains(marker)) + { + return invalid(format!( + "service {} embeds a secret-like environment key", + service.id + )); + } + } + let mut listeners = BTreeSet::new(); + for listener in &service.fence_listeners { + let address = listener.parse::().map_err(|_| { + ProfileError::Invalid(format!("service {} has an invalid fence listener", service.id)) + })?; + if !address.ip().is_loopback() || !listeners.insert(address) { + return invalid(format!( + "service {} has a non-loopback or duplicate fence listener", + service.id + )); + } + } + validate_probe(service)?; + let restart = &service.restart; + if restart.max_attempts == 0 + || restart.backoff_base_ms == 0 + || restart.backoff_base_ms > restart.backoff_max_ms + || restart.stable_after_ms == 0 + { + return invalid(format!("service {} has invalid restart bounds", service.id)); + } + } + Ok(()) +} + +fn validate_probe(service: &ServiceProfile) -> Result<(), ProfileError> { + let probe = &service.probe; + if probe.timeout_ms == 0 || probe.failure_threshold == 0 { + return invalid(format!("service {} has invalid probe bounds", service.id)); + } + if let Some(name) = &probe.bearer_env { + if probe.kind != ProbeKind::Http + || !name.starts_with("CROWDB_") + || !name + .bytes() + .all(|byte| byte.is_ascii_uppercase() || byte.is_ascii_digit() || byte == b'_') + { + return invalid(format!( + "service {} has an invalid probe credential reference", + service.id + )); + } + } + match probe.kind { + ProbeKind::Http if !(probe.target.starts_with("http://") || probe.target.starts_with("https://")) => { + invalid(format!("service {} has invalid HTTP probe", service.id)) + } + ProbeKind::Tcp | ProbeKind::RpcPing if probe.target.parse::().is_err() => { + invalid(format!("service {} has invalid socket probe", service.id)) + } + ProbeKind::Http | ProbeKind::Tcp | ProbeKind::RpcPing => Ok(()), + } +} + +fn topological_order(profile: &DeploymentProfile) -> Result, ProfileError> { + let by_id = profile + .services + .iter() + .map(|service| (service.id.as_str(), service)) + .collect::>(); + let mut remaining = profile + .services + .iter() + .map(|service| { + ( + service.id.as_str(), + service + .dependencies + .iter() + .map(String::as_str) + .collect::>(), + ) + }) + .collect::>(); + let mut order = Vec::with_capacity(profile.services.len()); + while !remaining.is_empty() { + let Some(id) = profile + .services + .iter() + .map(|service| service.id.as_str()) + .find(|id| remaining.get(id).is_some_and(BTreeSet::is_empty)) + else { + return invalid("service dependency graph contains a cycle"); + }; + remaining.remove(id); + for dependencies in remaining.values_mut() { + dependencies.remove(id); + } + order.push(by_id[id]); + } + Ok(order) +} + +fn require_slug(field: &str, value: &str) -> Result<(), ProfileError> { + if value.is_empty() + || !value + .bytes() + .all(|byte| byte.is_ascii_lowercase() || byte.is_ascii_digit() || byte == b'-') + { + return invalid(format!("{field} must be a lowercase slug")); + } + Ok(()) +} + +fn require_text(field: &str, value: &str) -> Result<(), ProfileError> { + if value.trim().is_empty() || value.len() > 256 { + return invalid(format!("{field} must be non-empty and bounded")); + } + Ok(()) +} + +fn invalid(message: impl Into) -> Result { + Err(ProfileError::Invalid(message.into())) +} diff --git a/container/crowdb-monitor/src/render.rs b/container/crowdb-monitor/src/render.rs new file mode 100644 index 000000000..5ccc05036 --- /dev/null +++ b/container/crowdb-monitor/src/render.rs @@ -0,0 +1,182 @@ +use std::collections::{BTreeMap, BTreeSet}; +use std::fs::{self, File, OpenOptions}; +use std::io::Write; +use std::os::unix::fs::OpenOptionsExt; +use std::path::{Path, PathBuf}; + +use thiserror::Error; +use uuid::Uuid; + +use crate::DeploymentProfile; + +const MAX_TEMPLATE_BYTES: u64 = 1024 * 1024; + +#[derive(Debug, Error)] +pub enum RenderError { + #[error("configuration rendering failed: {0}")] + Io(#[from] std::io::Error), + #[error("configuration template is invalid: {0}")] + Invalid(String), +} + +#[derive(Debug, Eq, PartialEq)] +pub struct RenderedConfig { + pub service_id: String, + pub path: PathBuf, +} + +/// # Errors +/// Rejects missing, oversized, symlinked, ambiguous, or unsafe templates. +pub fn render_configs( + profile: &DeploymentProfile, + template_root: &Path, + run_root: &Path, +) -> Result, RenderError> { + let variables = variables(profile)?; + require_directory(template_root)?; + require_directory(run_root)?; + let destination = run_root.join("config"); + match fs::symlink_metadata(&destination) { + Ok(metadata) if !metadata.file_type().is_dir() => { + return invalid("run/config must be a directory, not a link"); + } + Ok(_) => {} + Err(error) if error.kind() == std::io::ErrorKind::NotFound => fs::create_dir(&destination)?, + Err(error) => return Err(error.into()), + } + let mut outputs = Vec::new(); + let mut file_names = BTreeSet::new(); + for service in &profile.services { + let Some(template) = &service.config_template else { + continue; + }; + let name = template + .file_name() + .ok_or_else(|| RenderError::Invalid("template path has no file name".into()))?; + if !file_names.insert(name.to_os_string()) { + return invalid("multiple services render to the same file name"); + } + let source = template_root.join(name); + let metadata = fs::symlink_metadata(&source)?; + if !metadata.file_type().is_file() || metadata.len() > MAX_TEMPLATE_BYTES { + return invalid(format!( + "template for {} is not a bounded regular file", + service.id + )); + } + let body = fs::read_to_string(&source)?; + let rendered = substitute(&body, &variables)?; + toml::from_str::(&rendered).map_err(|error| { + RenderError::Invalid(format!("template for {} is not TOML: {error}", service.id)) + })?; + let path = destination.join(name); + atomic_write(&path, rendered.as_bytes())?; + outputs.push(RenderedConfig { + service_id: service.id.clone(), + path, + }); + } + Ok(outputs) +} + +fn variables(profile: &DeploymentProfile) -> Result, RenderError> { + let mut values = BTreeMap::new(); + for (name, path) in [ + ("install_root", &profile.paths.install_root), + ("bin_root", &profile.paths.bin_root), + ("template_root", &profile.paths.template_root), + ("data_root", &profile.paths.data_root), + ("run_root", &profile.paths.run_root), + ("log_root", &profile.paths.log_root), + ] { + values.insert(name.into(), safe_value(&path.to_string_lossy())?); + } + values.insert("s3_tenant".into(), safe_value(&profile.s3_tenant)?); + values.insert("iceberg_catalog".into(), safe_value(&profile.iceberg_catalog)?); + for (index, node) in profile.nodes.iter().enumerate() { + values.insert(format!("node.{index}.id"), node.node_id.to_string()); + values.insert(format!("node.{index}.rack_id"), node.rack_id.to_string()); + } + for (index, group) in profile.groups.iter().enumerate() { + values.insert(format!("group.{index}.id"), group.group_id.to_string()); + values.insert(format!("group.{index}.store_id"), group.store_id.to_string()); + } + for (index, disk) in profile.disks.iter().enumerate() { + values.insert( + format!("disk.{index}.path"), + safe_value(&disk.path.to_string_lossy())?, + ); + values.insert( + format!("disk.{index}.capacity_bytes"), + disk.capacity_bytes.to_string(), + ); + } + Ok(values) +} + +fn safe_value(value: &str) -> Result { + if value.is_empty() + || !value + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || b"-._/".contains(&byte)) + { + return invalid("profile value is unsafe for template substitution"); + } + Ok(value.into()) +} + +fn substitute(template: &str, variables: &BTreeMap) -> Result { + let mut output = String::with_capacity(template.len()); + let mut rest = template; + while let Some(start) = rest.find("{{") { + output.push_str(&rest[..start]); + let after_open = &rest[start + 2..]; + let end = after_open + .find("}}") + .ok_or_else(|| RenderError::Invalid("unclosed template variable".into()))?; + let name = &after_open[..end]; + let value = variables + .get(name) + .ok_or_else(|| RenderError::Invalid(format!("unknown template variable {name}")))?; + output.push_str(value); + rest = &after_open[end + 2..]; + } + if rest.contains("}}") { + return invalid("stray template terminator"); + } + output.push_str(rest); + Ok(output) +} + +fn require_directory(path: &Path) -> Result<(), RenderError> { + if !fs::symlink_metadata(path)?.file_type().is_dir() { + return invalid("render root must be a directory, not a link"); + } + Ok(()) +} + +fn atomic_write(path: &Path, body: &[u8]) -> Result<(), RenderError> { + let directory = path + .parent() + .ok_or_else(|| RenderError::Invalid("render path has no parent".into()))?; + let temporary = directory.join(format!(".config-{}.tmp", Uuid::new_v4())); + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&temporary)?; + let result = (|| { + file.write_all(body)?; + file.sync_all()?; + fs::rename(&temporary, path)?; + File::open(directory)?.sync_all() + })(); + if result.is_err() { + let _ = fs::remove_file(&temporary); + } + result.map_err(RenderError::Io) +} + +fn invalid(message: impl Into) -> Result { + Err(RenderError::Invalid(message.into())) +} diff --git a/container/crowdb-monitor/src/status.rs b/container/crowdb-monitor/src/status.rs new file mode 100644 index 000000000..240308e2a --- /dev/null +++ b/container/crowdb-monitor/src/status.rs @@ -0,0 +1,202 @@ +use std::collections::BTreeMap; +use std::fs::{self, File, OpenOptions}; +use std::io::Write; +use std::os::unix::fs::OpenOptionsExt; +use std::path::{Path, PathBuf}; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use serde::{Deserialize, Serialize}; +use thiserror::Error; +use uuid::Uuid; + +const VERSION: u32 = 1; +const MAX_BYTES: u64 = 64 * 1024; + +#[derive(Debug, Error)] +pub enum StatusError { + #[error("monitor status storage failed: {0}")] + Io(#[from] std::io::Error), + #[error("monitor status cannot be decoded: {0}")] + Decode(#[from] serde_json::Error), + #[error("monitor status is invalid")] + Invalid, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum MonitorPhase { + Initializing, + Ready, + Restarting, + Draining, + Failed, +} + +#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ServiceStatus { + pub pid: Option, + pub generation: u64, + pub healthy: bool, + pub restart_attempts: u32, +} + +#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct MonitorStatus { + pub version: u32, + pub deployment_id: Uuid, + pub monitor_pid: u32, + pub revision: u64, + pub updated_at_ms: u64, + pub phase: MonitorPhase, + pub services: BTreeMap, +} + +impl MonitorStatus { + #[must_use] + pub fn new(deployment_id: Uuid, phase: MonitorPhase) -> Self { + Self { + version: VERSION, + deployment_id, + monitor_pid: std::process::id(), + revision: 0, + updated_at_ms: 0, + phase, + services: BTreeMap::new(), + } + } +} + +pub struct StatusStore { + path: PathBuf, +} + +impl StatusStore { + /// # Errors + /// Rejects a missing, symlinked, or non-directory parent. + pub fn open_file(path: &Path) -> Result { + let parent = path.parent().ok_or(StatusError::Invalid)?; + if !fs::symlink_metadata(parent)?.file_type().is_dir() { + return Err(StatusError::Invalid); + } + Ok(Self { + path: path.to_owned(), + }) + } + + /// # Errors + /// Rejects missing, symlinked, or non-directory runtime roots. + pub fn new(run_root: &Path) -> Result { + if !fs::symlink_metadata(run_root)?.file_type().is_dir() { + return Err(StatusError::Invalid); + } + let directory = run_root.join("status"); + match fs::symlink_metadata(&directory) { + Ok(metadata) if !metadata.file_type().is_dir() => return Err(StatusError::Invalid), + Ok(_) => {} + Err(error) if error.kind() == std::io::ErrorKind::NotFound => fs::create_dir(&directory)?, + Err(error) => return Err(error.into()), + } + Ok(Self { + path: directory.join("monitor.json"), + }) + } + + /// # Errors + /// Rejects missing or symlinked status directories without creating them. + pub fn open(run_root: &Path) -> Result { + let directory = run_root.join("status"); + if !fs::symlink_metadata(&directory)?.file_type().is_dir() { + return Err(StatusError::Invalid); + } + Ok(Self { + path: directory.join("monitor.json"), + }) + } + + /// # Errors + /// Returns failed durable writes or an invalid deployment identity. + pub fn publish(&self, status: &mut MonitorStatus) -> Result<(), StatusError> { + if status.deployment_id.is_nil() || status.monitor_pid == 0 { + return Err(StatusError::Invalid); + } + let mut updated = status.clone(); + updated.revision = updated.revision.checked_add(1).ok_or(StatusError::Invalid)?; + updated.updated_at_ms = now_ms()?; + let bytes = serde_json::to_vec(&updated)?; + if bytes.len() as u64 > MAX_BYTES { + return Err(StatusError::Invalid); + } + atomic_write(&self.path, &bytes)?; + *status = updated; + Ok(()) + } + + /// # Errors + /// Rejects missing, corrupt, stale, or incompatible status snapshots. + pub fn read(&self, max_age: Duration) -> Result { + let metadata = fs::symlink_metadata(&self.path)?; + if !metadata.file_type().is_file() || metadata.len() > MAX_BYTES { + return Err(StatusError::Invalid); + } + let status: MonitorStatus = serde_json::from_slice(&fs::read(&self.path)?)?; + let age = now_ms()? + .checked_sub(status.updated_at_ms) + .ok_or(StatusError::Invalid)?; + if status.version != VERSION + || status.deployment_id.is_nil() + || status.monitor_pid == 0 + || status.revision == 0 + || age > max_age.as_millis().try_into().unwrap_or(u64::MAX) + { + return Err(StatusError::Invalid); + } + Ok(status) + } + + /// # Errors + /// Rejects stale or non-ready status or any unhealthy child. + pub fn readiness(&self, max_age: Duration) -> Result<(), StatusError> { + let status = self.read(max_age)?; + if status.phase != MonitorPhase::Ready + || status.services.is_empty() + || status + .services + .values() + .any(|service| service.pid.is_none() || !service.healthy) + { + return Err(StatusError::Invalid); + } + Ok(()) + } +} + +fn now_ms() -> Result { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_err(|_| StatusError::Invalid)? + .as_millis() + .try_into() + .map_err(|_| StatusError::Invalid) +} + +fn atomic_write(path: &Path, bytes: &[u8]) -> Result<(), StatusError> { + let directory = path.parent().ok_or(StatusError::Invalid)?; + let temporary = directory.join(format!(".monitor-{}.tmp", Uuid::new_v4())); + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&temporary)?; + let result = (|| { + file.write_all(bytes)?; + file.sync_all()?; + fs::rename(&temporary, path)?; + File::open(directory)?.sync_all() + })(); + if result.is_err() { + let _ = fs::remove_file(&temporary); + } + result.map_err(StatusError::Io) +} diff --git a/container/crowdb-monitor/src/supervisor.rs b/container/crowdb-monitor/src/supervisor.rs new file mode 100644 index 000000000..1810adc82 --- /dev/null +++ b/container/crowdb-monitor/src/supervisor.rs @@ -0,0 +1,571 @@ +use std::collections::{BTreeMap, BTreeSet}; +use std::path::Path; +use std::time::Duration; + +use thiserror::Error; +use tokio::time::{sleep, Instant}; +use uuid::Uuid; + +use crate::{ + DeploymentProfile, MonitorEvent, MonitorEventKind, MonitorPhase, MonitorStatus, ProbeError, + ProbeExecutor, ProcessError, ProcessManager, ProfileError, ServiceProfile, ServiceStatus, StatusError, + StatusStore, +}; + +const STARTUP_DEADLINE: Duration = Duration::from_secs(30); +const PROBE_RETRY_DELAY: Duration = Duration::from_millis(100); +const STOP_GRACE: Duration = Duration::from_secs(10); +const LISTENER_FENCE_DEADLINE: Duration = Duration::from_secs(2); + +#[derive(Debug, Error)] +pub enum SupervisorError { + #[error("deployment profile is invalid: {0}")] + Profile(#[from] ProfileError), + #[error("process operation failed: {0}")] + Process(#[from] ProcessError), + #[error("probe failed: {0}")] + Probe(#[from] ProbeError), + #[error("status update failed: {0}")] + Status(#[from] StatusError), + #[error("supervision state is invalid: {0}")] + Invalid(&'static str), +} + +pub struct Supervisor { + profile: DeploymentProfile, + order: Vec, + processes: ProcessManager, + probes: ProbeExecutor, + status_store: StatusStore, + status: MonitorStatus, + environment: BTreeMap>, + probe_failures: BTreeMap, + healthy_since: BTreeMap, + bootstrapped: bool, + requires_recovery_validation: bool, + recovery_pending: bool, + recovery_epoch: u64, +} + +impl Supervisor { + /// # Errors + /// Rejects invalid profiles or unavailable log/status directories. + pub async fn new( + profile: DeploymentProfile, + deployment_id: Uuid, + log_root: &Path, + run_root: &Path, + ) -> Result { + profile.validate()?; + let order = profile + .services_in_start_order()? + .iter() + .map(|service| service.id.clone()) + .collect(); + let processes = ProcessManager::new(log_root.to_owned(), profile.logs.clone()).await?; + let probes = ProbeExecutor::new( + profile + .services + .iter() + .any(|service| service.probe.kind == crate::ProbeKind::RpcPing), + )?; + let status_store = StatusStore::new(run_root)?; + let mut status = MonitorStatus::new(deployment_id, MonitorPhase::Initializing); + status_store.publish(&mut status)?; + Ok(Self { + profile, + order, + processes, + probes, + status_store, + status, + environment: BTreeMap::new(), + probe_failures: BTreeMap::new(), + healthy_since: BTreeMap::new(), + bootstrapped: false, + requires_recovery_validation: false, + recovery_pending: false, + recovery_epoch: 0, + }) + } + + pub fn require_recovery_validation(&mut self) { + self.requires_recovery_validation = true; + } + + #[must_use] + pub fn recovery_pending(&self) -> bool { + self.recovery_pending + } + + #[must_use] + pub fn recovery_epoch(&self) -> u64 { + self.recovery_epoch + } + + /// # Errors + /// Refuses readiness until all restarted services and durable authority have been checked. + pub async fn finish_recovery(&mut self) -> Result<(), SupervisorError> { + if !self.recovery_pending + || self.status.phase != MonitorPhase::Restarting + || self.status.services.len() != self.order.len() + || self.status.services.values().any(|service| !service.healthy) + { + return Err(SupervisorError::Invalid("recovery is not ready for validation")); + } + self.recovery_pending = false; + self.status.phase = MonitorPhase::Ready; + self.status_store.publish(&mut self.status)?; + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::Ready, + service: None, + pid: None, + attempt: None, + }) + .await?; + Ok(()) + } + + #[must_use] + pub fn status(&self) -> &MonitorStatus { + &self.status + } + + pub fn monitor_log_mut(&mut self) -> &mut crate::MonitorLog { + self.processes.monitor_log_mut() + } + + /// # Errors + /// Refreshes the status timestamp during a bounded bootstrap probe. + pub fn refresh_status(&mut self) -> Result<(), SupervisorError> { + self.status_store.publish(&mut self.status)?; + Ok(()) + } + + /// # Errors + /// Starts one service only after its dependencies are healthy and waits for its probe. + pub async fn start_service( + &mut self, + id: &str, + environment: BTreeMap, + ) -> Result<(), SupervisorError> { + if matches!(self.status.phase, MonitorPhase::Draining | MonitorPhase::Failed) { + return Err(SupervisorError::Invalid( + "supervisor is no longer admitting children", + )); + } + let service = self.service(id)?.clone(); + if self.status.services.contains_key(id) { + return Err(SupervisorError::Invalid("service is already started")); + } + if service.dependencies.iter().any(|dependency| { + !self + .status + .services + .get(dependency) + .is_some_and(|status| status.healthy) + }) { + return Err(SupervisorError::Invalid("service dependencies are not healthy")); + } + let pid = self.processes.start(&service, &environment).await?; + if let Err(error) = self.wait_for_probe(&service, &environment).await { + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::ProbeFailed, + service: Some(id), + pid: Some(pid), + attempt: None, + }) + .await?; + self.processes.stop(id, STOP_GRACE).await?; + return Err(error); + } + self.environment.insert(id.to_owned(), environment); + self.healthy_since.insert(id.to_owned(), Instant::now()); + self.status.services.insert( + id.to_owned(), + ServiceStatus { + pid: Some(pid), + generation: 1, + healthy: true, + restart_attempts: 0, + }, + ); + self.status_store.publish(&mut self.status)?; + Ok(()) + } + + /// # Errors + /// Refuses to publish readiness until every profile service is healthy. + pub async fn mark_ready(&mut self) -> Result<(), SupervisorError> { + if self.status.services.len() != self.order.len() + || self.status.services.values().any(|service| !service.healthy) + { + return Err(SupervisorError::Invalid("not all services are healthy")); + } + self.bootstrapped = true; + self.status.phase = MonitorPhase::Ready; + self.status_store.publish(&mut self.status)?; + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::Ready, + service: None, + pid: None, + attempt: None, + }) + .await?; + Ok(()) + } + + /// # Errors + /// Returns failed probes, process errors, or exhausted restart budgets. + pub async fn poll_once(&mut self) -> Result<(), SupervisorError> { + if matches!(self.status.phase, MonitorPhase::Draining | MonitorPhase::Failed) { + return Err(SupervisorError::Invalid("supervisor is not running")); + } + self.processes.maintain_logs().await?; + for id in self.order.clone() { + if !self.processes.owns(&id) { + continue; + } + let service = self.service(&id)?.clone(); + let alive = self.processes.alive(&id)?; + let healthy = if alive { + match self.environment.get(&id) { + Some(environment) => self.probes.probe_service(&service, environment).await.is_ok(), + None => false, + } + } else { + false + }; + if healthy { + self.probe_failures.insert(id.clone(), 0); + if self.healthy_since.get(&id).is_some_and(|since| { + since.elapsed() >= Duration::from_millis(service.restart.stable_after_ms) + }) { + if let Some(state) = self.status.services.get_mut(&id) { + if state.restart_attempts != 0 { + state.restart_attempts = 0; + self.status_store.publish(&mut self.status)?; + let pid = self.processes.pid(&id); + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::RestartBudgetReset, + service: Some(&id), + pid, + attempt: None, + }) + .await?; + } + } + } + if let Some(state) = self.status.services.get_mut(&id) { + if !state.healthy { + state.healthy = true; + if self.bootstrapped + && !self.recovery_pending + && self.status.services.values().all(|service| service.healthy) + { + self.status.phase = MonitorPhase::Ready; + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::Ready, + service: None, + pid: None, + attempt: None, + }) + .await?; + } + self.status_store.publish(&mut self.status)?; + } + } + continue; + } + let failures = self.probe_failures.entry(id.clone()).or_default(); + self.healthy_since.remove(&id); + *failures = failures.saturating_add(1); + if *failures == 1 || !alive { + self.processes + .record_event(&MonitorEvent { + kind: if alive { + MonitorEventKind::ProbeFailed + } else { + MonitorEventKind::ChildExited + }, + service: Some(&id), + pid: self.processes.pid(&id), + attempt: None, + }) + .await?; + } + self.status.phase = MonitorPhase::Restarting; + if let Some(state) = self.status.services.get_mut(&id) { + state.healthy = false; + } + self.status_store.publish(&mut self.status)?; + if !alive || *failures >= service.probe.failure_threshold { + self.recover(&id).await?; + } + break; + } + self.status_store.publish(&mut self.status)?; + Ok(()) + } + + /// # Errors + /// Returns fatal supervision failures or failed signal registration. + pub async fn run_until_signal(&mut self) -> Result<(), SupervisorError> { + let mut terminate = tokio::signal::unix::signal(tokio::signal::unix::SignalKind::terminate()) + .map_err(|_| SupervisorError::Invalid("cannot register SIGTERM handler"))?; + loop { + tokio::select! { + _ = terminate.recv() => break, + _ = tokio::signal::ctrl_c() => break, + result = self.poll_once() => result?, + } + tokio::select! { + _ = terminate.recv() => break, + _ = tokio::signal::ctrl_c() => break, + () = sleep(Duration::from_secs(1)) => {} + } + } + self.shutdown().await + } + + /// # Errors + /// Disables restart, drains every child in reverse dependency order, and logs closure. + pub async fn shutdown(&mut self) -> Result<(), SupervisorError> { + self.status.phase = MonitorPhase::Draining; + self.status_store.publish(&mut self.status)?; + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::Draining, + service: None, + pid: None, + attempt: None, + }) + .await?; + self.processes.stop_all(&self.order, STOP_GRACE).await?; + for state in self.status.services.values_mut() { + state.pid = None; + state.healthy = false; + } + self.status_store.publish(&mut self.status)?; + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::Stopped, + service: None, + pid: None, + attempt: None, + }) + .await?; + Ok(()) + } + + fn service(&self, id: &str) -> Result<&ServiceProfile, SupervisorError> { + self.profile + .services + .iter() + .find(|service| service.id == id) + .ok_or(SupervisorError::Invalid("unknown service")) + } + + async fn wait_for_probe( + &mut self, + service: &ServiceProfile, + environment: &BTreeMap, + ) -> Result<(), SupervisorError> { + let deadline = Instant::now() + STARTUP_DEADLINE; + let mut last_heartbeat = Instant::now(); + loop { + if !self.processes.alive(&service.id)? { + return Err(SupervisorError::Invalid("service exited before readiness")); + } + if self.probes.probe_service(service, environment).await.is_ok() { + return Ok(()); + } + if Instant::now() >= deadline { + return Err(SupervisorError::Invalid("service readiness deadline expired")); + } + if last_heartbeat.elapsed() >= Duration::from_secs(1) { + self.status_store.publish(&mut self.status)?; + last_heartbeat = Instant::now(); + } + sleep(PROBE_RETRY_DELAY).await; + } + } + + fn affected_services(&self, root: &str) -> Vec { + let mut affected = BTreeSet::from([root.to_owned()]); + loop { + let before = affected.len(); + for service in &self.profile.services { + if service + .dependencies + .iter() + .any(|dependency| affected.contains(dependency)) + { + affected.insert(service.id.clone()); + } + } + if affected.len() == before { + break; + } + } + self.order + .iter() + .filter(|id| affected.contains(*id) && self.status.services.contains_key(*id)) + .cloned() + .collect() + } + + async fn recover(&mut self, root: &str) -> Result<(), SupervisorError> { + let affected = self.affected_services(root); + let service = self.service(root)?.clone(); + let first_attempt = self + .status + .services + .get(root) + .map_or(1, |state| state.restart_attempts.saturating_add(1)); + for attempt in first_attempt..=service.restart.max_attempts { + self.stop_affected(&affected).await?; + if let Some(state) = self.status.services.get_mut(root) { + state.restart_attempts = attempt; + } + self.status_store.publish(&mut self.status)?; + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::Restarting, + service: Some(root), + pid: None, + attempt: Some(attempt), + }) + .await?; + let shift = attempt.saturating_sub(1).min(31); + let backoff = service + .restart + .backoff_base_ms + .saturating_mul(1_u64 << shift) + .min(service.restart.backoff_max_ms); + sleep(Duration::from_millis(backoff)).await; + if self.start_affected(&affected).await? { + self.probe_failures.insert(root.to_owned(), 0); + if self.bootstrapped { + self.recovery_epoch = self.recovery_epoch.saturating_add(1); + self.recovery_pending = true; + if !self.requires_recovery_validation { + self.finish_recovery().await?; + } + } + return Ok(()); + } + } + self.stop_affected(&affected).await?; + self.fail_exhausted(root, first_attempt.max(service.restart.max_attempts)) + .await?; + Err(SupervisorError::Invalid("service restart budget exhausted")) + } + + async fn stop_affected(&mut self, affected: &[String]) -> Result<(), SupervisorError> { + for id in affected.iter().rev() { + if self.processes.owns(id) { + self.processes.stop(id, STOP_GRACE).await?; + } + if let Some(state) = self.status.services.get_mut(id) { + state.pid = None; + state.healthy = false; + } + self.status_store.publish(&mut self.status)?; + let service = self.service(id)?.clone(); + for address in &service.fence_listeners { + if let Err(error) = self + .processes + .wait_listener_closed(address, LISTENER_FENCE_DEADLINE) + .await + { + self.status.phase = MonitorPhase::Failed; + self.status_store.publish(&mut self.status)?; + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::ListenerFenceFailed, + service: Some(id), + pid: None, + attempt: None, + }) + .await?; + return Err(error.into()); + } + } + } + self.status_store.publish(&mut self.status)?; + Ok(()) + } + + async fn start_affected(&mut self, affected: &[String]) -> Result { + for id in affected { + let environment = self + .environment + .get(id) + .cloned() + .ok_or(SupervisorError::Invalid("service environment is missing"))?; + let restart_service = self.service(id)?.clone(); + let Ok(pid) = self.processes.start(&restart_service, &environment).await else { + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::ChildStartFailed, + service: Some(id), + pid: None, + attempt: None, + }) + .await?; + return Ok(false); + }; + let readiness = self.wait_for_probe(&restart_service, &environment).await; + if let Err(SupervisorError::Status(error)) = readiness { + self.processes.stop(id, STOP_GRACE).await?; + return Err(SupervisorError::Status(error)); + } + if readiness.is_err() { + self.processes.stop(id, STOP_GRACE).await?; + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::ProbeFailed, + service: Some(id), + pid: Some(pid), + attempt: None, + }) + .await?; + return Ok(false); + } + if let Some(state) = self.status.services.get_mut(id) { + state.pid = Some(pid); + state.generation = state.generation.saturating_add(1); + state.healthy = true; + } + self.healthy_since.insert(id.clone(), Instant::now()); + self.status_store.publish(&mut self.status)?; + } + Ok(true) + } + + async fn fail_exhausted(&mut self, root: &str, attempt: u32) -> Result<(), SupervisorError> { + self.status.phase = MonitorPhase::Failed; + self.status_store.publish(&mut self.status)?; + self.processes + .record_event(&MonitorEvent { + kind: MonitorEventKind::RestartExhausted, + service: Some(root), + pid: None, + attempt: Some(attempt), + }) + .await?; + self.processes.stop_all(&self.order, STOP_GRACE).await?; + for state in self.status.services.values_mut() { + state.pid = None; + state.healthy = false; + } + self.status_store.publish(&mut self.status)?; + Ok(()) + } +} diff --git a/container/crowdb-monitor/tests/access_bootstrap_test.rs b/container/crowdb-monitor/tests/access_bootstrap_test.rs new file mode 100644 index 000000000..ce1dd5ef1 --- /dev/null +++ b/container/crowdb-monitor/tests/access_bootstrap_test.rs @@ -0,0 +1,112 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs; +use std::os::unix::fs::PermissionsExt; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{ + s3_step_names, show_client_credentials, BootstrapSession, DeploymentProfile, MonitorLog, S3Bootstrap, + ServerCredentials, +}; +use uuid::Uuid; + +struct TestRoot(PathBuf); + +impl TestRoot { + fn new() -> Self { + let path = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-access-{}", Uuid::new_v4())); + fs::create_dir_all(path.join("data")).unwrap(); + fs::create_dir_all(path.join("log")).unwrap(); + Self(path.canonicalize().unwrap()) + } +} + +impl Drop for TestRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn profile(root: &TestRoot) -> DeploymentProfile { + let mut profile = DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap(); + let program = root.0.join("credential-command"); + let script = format!( + "#!/bin/sh\nprintf '%s\\n' \"$1\" >> '{}'\nprintf 'rpc initialization log\\nAWS_ACCESS_KEY_ID=CROW123\\nAWS_SECRET_ACCESS_KEY=secret_123\\n'\n", + root.0.join("calls").display() + ); + fs::write(&program, script).unwrap(); + fs::set_permissions(&program, fs::Permissions::from_mode(0o700)).unwrap(); + profile + .services + .iter_mut() + .find(|service| service.id == "s3") + .unwrap() + .program = program; + profile +} + +#[tokio::test] +async fn s3_bootstrap_reuses_user_and_validates_ready_without_creation() { + let root = TestRoot::new(); + let profile = profile(&root); + let data_root = root.0.join("data"); + let mut session = BootstrapSession::open(&data_root, b"profile", b"config", &s3_step_names()).unwrap(); + let credentials = ServerCredentials::load_or_create(&data_root).unwrap(); + let mut events = MonitorLog::open(&root.0.join("log"), profile.logs.clone()) + .await + .unwrap(); + + S3Bootstrap::reconcile(&mut session, &profile, &credentials, &mut events) + .await + .unwrap(); + assert_eq!(session.manifest().step_complete("s3-user"), Some(true)); + let client = show_client_credentials(&data_root).unwrap(); + assert!(client.contains("AWS_ENDPOINT_URL=http://localhost:81\n")); + assert!(client.contains("ICEBERG_URI=http://localhost\n")); + session.mark_ready().unwrap(); + + let mut restarted = BootstrapSession::open(&data_root, b"profile", b"config", &s3_step_names()).unwrap(); + S3Bootstrap::reconcile(&mut restarted, &profile, &credentials, &mut events) + .await + .unwrap(); + assert_eq!(show_client_credentials(&data_root).unwrap(), client); + assert_eq!( + fs::read_to_string(root.0.join("calls")).unwrap(), + "ensure-user\nlookup-user\n" + ); +} + +#[tokio::test] +async fn ready_s3_bootstrap_rejects_client_file_conflict() { + let root = TestRoot::new(); + let profile = profile(&root); + let data_root = root.0.join("data"); + let mut session = BootstrapSession::open(&data_root, b"profile", b"config", &s3_step_names()).unwrap(); + let credentials = ServerCredentials::load_or_create(&data_root).unwrap(); + let mut events = MonitorLog::open(&root.0.join("log"), profile.logs.clone()) + .await + .unwrap(); + S3Bootstrap::reconcile(&mut session, &profile, &credentials, &mut events) + .await + .unwrap(); + session.mark_ready().unwrap(); + fs::write(data_root.join("secrets/client.env"), b"conflict\n").unwrap(); + + assert!( + S3Bootstrap::reconcile(&mut session, &profile, &credentials, &mut events) + .await + .is_err() + ); + assert_eq!( + fs::read_to_string(root.0.join("calls")).unwrap(), + "ensure-user\nlookup-user\n" + ); +} diff --git a/container/crowdb-monitor/tests/credentials_test.rs b/container/crowdb-monitor/tests/credentials_test.rs new file mode 100644 index 000000000..31560476e --- /dev/null +++ b/container/crowdb-monitor/tests/credentials_test.rs @@ -0,0 +1,152 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs; +use std::os::unix::fs::{symlink, PermissionsExt}; +use std::path::{Path, PathBuf}; +use std::process::Command; + +use crowdb_monitor::{show_client_credentials, ClientCredentials, ServerCredentials}; +use uuid::Uuid; + +struct TestDataRoot(PathBuf); + +impl TestDataRoot { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-credentials-{}", Uuid::new_v4())); + fs::create_dir_all(&root).unwrap(); + Self(root.canonicalize().unwrap()) + } + + fn path(&self) -> &Path { + &self.0 + } +} + +impl Drop for TestDataRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn client() -> ClientCredentials { + ClientCredentials { + s3_endpoint: "http://localhost:16000".into(), + iceberg_endpoint: "http://localhost:8181".into(), + region: "us-east-1".into(), + access_key_id: "CROW123".into(), + secret_access_key: "secret_123".into(), + } +} + +#[test] +fn server_secrets_are_distinct_private_and_stable() { + let root = TestDataRoot::new(); + let first = ServerCredentials::load_or_create(root.path()).unwrap(); + let body = first.server_env(); + assert_eq!(body.lines().count(), 5); + let tokens = body + .lines() + .skip(1) + .map(|line| line.split_once('=').unwrap().1) + .collect::>(); + for (index, token) in tokens.iter().enumerate() { + assert_eq!(token.len(), 64); + assert!(!tokens[..index].contains(token)); + } + let server_path = root.path().join("secrets/server.env"); + assert_eq!( + fs::metadata(&server_path).unwrap().permissions().mode() & 0o777, + 0o600 + ); + assert_eq!( + fs::metadata(root.path().join("secrets")) + .unwrap() + .permissions() + .mode() + & 0o777, + 0o700 + ); + let second = ServerCredentials::load_or_create(root.path()).unwrap(); + assert_eq!(second.server_env(), body); + assert!(show_client_credentials(root.path()).is_err()); +} + +#[test] +fn client_output_excludes_server_only_values_and_cannot_be_replaced() { + let root = TestDataRoot::new(); + let server = ServerCredentials::load_or_create(root.path()).unwrap(); + server.persist_client(&client()).unwrap(); + server.persist_client(&client()).unwrap(); + let output = show_client_credentials(root.path()).unwrap(); + assert!(output.contains("AWS_ACCESS_KEY_ID=CROW123")); + assert!(output.contains("ICEBERG_TOKEN=")); + assert!(!output.contains("CROWDB_S3_MASTER_KEY")); + assert!(!output.contains("CROWDB_ICEBERG_MANAGE_TOKEN")); + let mut conflicting = client(); + conflicting.access_key_id = "CROW456".into(); + assert!(server.persist_client(&conflicting).is_err()); + assert_eq!(show_client_credentials(root.path()).unwrap(), output); + assert_eq!( + fs::metadata(root.path().join("secrets/client.env")) + .unwrap() + .permissions() + .mode() + & 0o777, + 0o600 + ); +} + +#[test] +fn malformed_or_exposed_secret_files_fail_closed() { + let root = TestDataRoot::new(); + let server = ServerCredentials::load_or_create(root.path()).unwrap(); + server.persist_client(&client()).unwrap(); + let client_path = root.path().join("secrets/client.env"); + fs::set_permissions(&client_path, fs::Permissions::from_mode(0o644)).unwrap(); + assert!(show_client_credentials(root.path()).is_err()); + fs::set_permissions(&client_path, fs::Permissions::from_mode(0o600)).unwrap(); + fs::write( + &client_path, + b"AWS_ENDPOINT_URL=http://localhost:16000\nCROWDB_S3_MASTER_KEY=leak\n", + ) + .unwrap(); + assert!(show_client_credentials(root.path()).is_err()); + let server_path = root.path().join("secrets/server.env"); + fs::remove_file(&server_path).unwrap(); + symlink(&client_path, &server_path).unwrap(); + assert!(ServerCredentials::load_or_create(root.path()).is_err()); +} + +#[test] +fn invalid_client_values_are_not_persisted() { + let root = TestDataRoot::new(); + let server = ServerCredentials::load_or_create(root.path()).unwrap(); + let mut value = client(); + value.secret_access_key = "secret\nINJECTED=yes".into(); + assert!(server.persist_client(&value).is_err()); + assert!(!root.path().join("secrets/client.env").exists()); +} + +#[test] +fn explicit_cli_prints_only_client_file() { + let root = TestDataRoot::new(); + let server = ServerCredentials::load_or_create(root.path()).unwrap(); + server.persist_client(&client()).unwrap(); + let output = Command::new(env!("CARGO_BIN_EXE_crowdb-monitor")) + .args(["credentials", "show", "--format", "env", "--data-root"]) + .arg(root.path()) + .output() + .unwrap(); + assert!(output.status.success()); + assert!(output.stderr.is_empty()); + assert_eq!( + output.stdout, + show_client_credentials(root.path()).unwrap().as_bytes() + ); + assert!(!String::from_utf8_lossy(&output.stdout).contains("CROWDB_S3_MASTER_KEY")); +} diff --git a/container/crowdb-monitor/tests/disk_bootstrap_test.rs b/container/crowdb-monitor/tests/disk_bootstrap_test.rs new file mode 100644 index 000000000..001aeab9f --- /dev/null +++ b/container/crowdb-monitor/tests/disk_bootstrap_test.rs @@ -0,0 +1,124 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs::{self, File, OpenOptions}; +use std::io::{Read, Seek, SeekFrom, Write}; +use std::os::unix::fs::symlink; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{disk_step_names, ensure_disk_files, BootstrapSession, DeploymentProfile, MonitorLog}; +use uuid::Uuid; + +struct TestRoots(PathBuf); + +impl TestRoots { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-disks-{}", Uuid::new_v4())); + fs::create_dir_all(root.join("data")).unwrap(); + Self(root.canonicalize().unwrap()) + } + + fn profile(&self) -> DeploymentProfile { + let mut profile = DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap(); + profile.paths.install_root.clone_from(&self.0); + profile.paths.bin_root = self.0.join("bin"); + profile.paths.template_root = self.0.join("templates"); + profile.paths.data_root = self.0.join("data"); + profile.paths.run_root = self.0.join("run"); + profile.paths.log_root = self.0.join("data/log"); + for disk in &mut profile.disks { + disk.path = self.0.join("data/disks").join(disk.path.file_name().unwrap()); + } + for service in &mut profile.services { + service.program = self.0.join("bin").join(service.program.file_name().unwrap()); + service.config_template = service + .config_template + .as_ref() + .map(|path| self.0.join("templates").join(path.file_name().unwrap())); + } + profile.validate().unwrap(); + profile + } + + fn session(&self, profile: &DeploymentProfile) -> BootstrapSession { + let names = disk_step_names(profile); + let steps = names.iter().map(String::as_str).collect::>(); + BootstrapSession::open(&self.0.join("data"), b"profile", b"config", &steps).unwrap() + } + + async fn events(&self, profile: &DeploymentProfile) -> MonitorLog { + let root = self.0.join("data/log"); + fs::create_dir_all(&root).unwrap(); + MonitorLog::open(&root, profile.logs.clone()).await.unwrap() + } +} + +impl Drop for TestRoots { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +#[tokio::test] +async fn creates_sparse_disks_and_ready_restart_preserves_bytes() { + let roots = TestRoots::new(); + let profile = roots.profile(); + let mut session = roots.session(&profile); + let mut events = roots.events(&profile).await; + ensure_disk_files(&mut session, &profile, &mut events) + .await + .unwrap(); + assert!(session.manifest().next_step().is_none()); + for disk in &profile.disks { + let metadata = fs::metadata(&disk.path).unwrap(); + assert_eq!(metadata.len(), disk.capacity_bytes); + } + let first = &profile.disks[0].path; + let mut file = OpenOptions::new().write(true).open(first).unwrap(); + file.seek(SeekFrom::Start(1024)).unwrap(); + file.write_all(b"keep").unwrap(); + file.sync_all().unwrap(); + session.mark_ready().unwrap(); + drop(session); + let mut restart = roots.session(&profile); + ensure_disk_files(&mut restart, &profile, &mut events) + .await + .unwrap(); + let mut marker = [0; 4]; + let mut file = File::open(first).unwrap(); + file.seek(SeekFrom::Start(1024)).unwrap(); + file.read_exact(&mut marker).unwrap(); + assert_eq!(&marker, b"keep"); + let body = fs::read_to_string(roots.0.join("data/log/monitor/monitor.log")).unwrap(); + assert_eq!(body.matches("bootstrap_step_completed").count(), 4); +} + +#[tokio::test] +async fn rejects_changed_or_missing_completed_disk() { + let roots = TestRoots::new(); + let profile = roots.profile(); + let mut session = roots.session(&profile); + let mut events = roots.events(&profile).await; + ensure_disk_files(&mut session, &profile, &mut events) + .await + .unwrap(); + session.mark_ready().unwrap(); + let first = &profile.disks[0].path; + fs::remove_file(first).unwrap(); + let mut restart = roots.session(&profile); + assert!(ensure_disk_files(&mut restart, &profile, &mut events) + .await + .is_err()); + assert!(!first.exists()); + symlink("/dev/null", first).unwrap(); + assert!(ensure_disk_files(&mut restart, &profile, &mut events) + .await + .is_err()); +} diff --git a/container/crowdb-monitor/tests/hardware_bootstrap_test.rs b/container/crowdb-monitor/tests/hardware_bootstrap_test.rs new file mode 100644 index 000000000..42529ebb7 --- /dev/null +++ b/container/crowdb-monitor/tests/hardware_bootstrap_test.rs @@ -0,0 +1,140 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{ + hardware_step_names, BootstrapSession, DeploymentProfile, HardwareBootstrap, MonitorLog, +}; +use crowdb_protocol::common::{HwStatus, RackValue}; +use crowdb_test_harness::cluster::KvCluster; +use uuid::Uuid; + +struct TestRoot(PathBuf); + +impl TestRoot { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-hardware-{}", Uuid::new_v4())); + fs::create_dir_all(root.join("data")).unwrap(); + Self(root.canonicalize().unwrap()) + } + + fn profile(&self) -> DeploymentProfile { + let mut profile = DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap(); + profile.paths.install_root.clone_from(&self.0); + profile.paths.bin_root = self.0.join("bin"); + profile.paths.template_root = self.0.join("templates"); + profile.paths.data_root = self.0.join("data"); + profile.paths.run_root = self.0.join("run"); + profile.paths.log_root = self.0.join("data/log"); + for disk in &mut profile.disks { + disk.path = self.0.join("data/disks").join(disk.path.file_name().unwrap()); + } + for service in &mut profile.services { + service.program = self.0.join("bin").join(service.program.file_name().unwrap()); + service.config_template = service + .config_template + .as_ref() + .map(|path| self.0.join("templates").join(path.file_name().unwrap())); + } + profile.validate().unwrap(); + profile + } + + fn session(&self) -> BootstrapSession { + let names = hardware_step_names(); + let steps = names.iter().map(String::as_str).collect::>(); + BootstrapSession::open(&self.0.join("data"), b"profile", b"config", &steps).unwrap() + } + + async fn events(&self, profile: &DeploymentProfile) -> MonitorLog { + let root = self.0.join("data/log"); + fs::create_dir_all(&root).unwrap(); + MonitorLog::open(&root, profile.logs.clone()).await.unwrap() + } +} + +impl Drop for TestRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +#[tokio::test] +async fn writes_and_validates_hardware_with_real_group_zero() { + if crowdb_test_harness::cluster::crowdb_kv_server_bin().is_none() { + eprintln!("skipping real KV bootstrap: crowdb-kv-server binary is unavailable"); + return; + } + let cluster = KvCluster::start().await; + let root = TestRoot::new(); + let profile = root.profile(); + let mut session = root.session(); + let mut events = root.events(&profile).await; + let bootstrap = HardwareBootstrap::new(cluster.mgmt_endpoints[0].clone()); + bootstrap + .reconcile(&mut session, &profile, &mut events) + .await + .unwrap(); + assert_eq!(session.manifest().next_step(), None); + session.mark_ready().unwrap(); + drop(session); + let mut restart = root.session(); + bootstrap + .reconcile(&mut restart, &profile, &mut events) + .await + .unwrap(); + let hardware = cluster.make_hardware_client(); + let disks = hardware.list_all_disks().await.unwrap(); + assert_eq!(disks.len(), 4); + assert!(disks.iter().all(|disk| disk.value.zone_count == 1)); + assert!(disks + .iter() + .all(|disk| disk.capacity_bytes() == 16 * 1024 * 1024 * 1024)); + let owner = hardware.get_owner(1, 1, 101).await.unwrap().unwrap(); + assert_eq!(owner.instance_id, 1); + let bind = hardware.get_bind(1, 1, 101).await.unwrap().unwrap(); + assert_eq!((bind.store_id, bind.group_id), (0, 1)); + let body = fs::read_to_string(root.0.join("data/log/monitor/monitor.log")).unwrap(); + assert_eq!(body.matches("bootstrap_step_completed").count(), 1); +} + +#[tokio::test] +async fn conflicting_group_zero_record_rejects_without_creating_hardware() { + if crowdb_test_harness::cluster::crowdb_kv_server_bin().is_none() { + eprintln!("skipping real KV bootstrap: crowdb-kv-server binary is unavailable"); + return; + } + let cluster = KvCluster::start().await; + let hardware = cluster.make_hardware_client(); + hardware + .add_rack( + 1, + &RackValue { + status: HwStatus::Up as i32, + node_ids: vec![99], + }, + ) + .await + .unwrap(); + let root = TestRoot::new(); + let profile = root.profile(); + let mut session = root.session(); + let mut events = root.events(&profile).await; + let bootstrap = HardwareBootstrap::new(cluster.mgmt_endpoints[0].clone()); + assert!(bootstrap + .reconcile(&mut session, &profile, &mut events) + .await + .is_err()); + assert!(hardware.list_nodes().await.unwrap().is_empty()); + assert!(hardware.list_all_disks().await.unwrap().is_empty()); + assert_eq!(session.manifest().next_step(), Some("hardware-topology")); +} diff --git a/container/crowdb-monitor/tests/iceberg_bootstrap_test.rs b/container/crowdb-monitor/tests/iceberg_bootstrap_test.rs new file mode 100644 index 000000000..245bd4861 --- /dev/null +++ b/container/crowdb-monitor/tests/iceberg_bootstrap_test.rs @@ -0,0 +1,135 @@ +use std::fs; +use std::os::unix::fs::PermissionsExt; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{ + iceberg_step_names, BootstrapSession, DeploymentProfile, IcebergBootstrap, MonitorLog, ServerCredentials, +}; +use uuid::Uuid; + +struct TestRoot(PathBuf); + +impl TestRoot { + fn new() -> Self { + let path = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-iceberg-{}", Uuid::new_v4())); + fs::create_dir_all(path.join("data")).unwrap(); + fs::create_dir_all(path.join("log")).unwrap(); + Self(path.canonicalize().unwrap()) + } +} + +impl Drop for TestRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn profile(root: &TestRoot) -> DeploymentProfile { + let mut profile = DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap(); + let program = root.0.join("iceberg-management"); + let script = format!( + r#"#!/bin/sh +set -eu +root='{}' +printf '%s\n' "$1" >> "$root/calls" +if [ "$1" = inspect ]; then + if [ ! -f "$root/initialized" ]; then + printf '%s\n' '{{"initialized":false}}' + exit 0 + fi + catalog=$(cat "$root/initialized") + operation=$(cat "$root/operation") + capabilities=0x0000 + if [ -f "$root/activated" ]; then capabilities=0x3fff; fi + printf '{{"initialized":true,"catalog_id":"%s","display_name":"preview","activation_epoch":1,"state":"Ready","capability_bits":"%s","root_operation_id":"%s"}}\n' "$catalog" "$capabilities" "$operation" + exit 0 +fi +if [ "$1" = initialize ]; then + printf '%s' '11111111-1111-4111-8111-111111111111' > "$root/initialized" + printf '%s' "$2" | tr -d '-' > "$root/operation" + exit 1 +fi +if [ "$1" = activate ]; then + printf '%s' "$2" | tr -d '-' > "$root/operation" + touch "$root/activated" + exit 1 +fi +exit 2 +"#, + root.0.display() + ); + fs::write(&program, script).unwrap(); + fs::set_permissions(&program, fs::Permissions::from_mode(0o700)).unwrap(); + profile + .services + .iter_mut() + .find(|service| service.id == "iceberg") + .unwrap() + .program = program; + profile +} + +#[tokio::test] +async fn lost_management_responses_are_proved_then_ready_restart_is_read_only() { + let root = TestRoot::new(); + let profile = profile(&root); + let data_root = root.0.join("data"); + let mut session = + BootstrapSession::open(&data_root, b"profile", b"config", &iceberg_step_names()).unwrap(); + let credentials = ServerCredentials::load_or_create(&data_root).unwrap(); + let mut events = MonitorLog::open(&root.0.join("log"), profile.logs.clone()) + .await + .unwrap(); + + IcebergBootstrap::reconcile(&mut session, &profile, &credentials, &mut events) + .await + .unwrap(); + assert_eq!(session.manifest().step_complete("iceberg-initialize"), Some(true)); + assert_eq!(session.manifest().step_complete("iceberg-activate"), Some(true)); + session.mark_ready().unwrap(); + let before = fs::read_to_string(root.0.join("calls")).unwrap(); + assert!(before.contains("initialize\n")); + assert!(before.contains("activate\n")); + + let mut restarted = + BootstrapSession::open(&data_root, b"profile", b"config", &iceberg_step_names()).unwrap(); + IcebergBootstrap::reconcile(&mut restarted, &profile, &credentials, &mut events) + .await + .unwrap(); + let after = fs::read_to_string(root.0.join("calls")).unwrap(); + assert_eq!(after.matches("initialize\n").count(), 1); + assert_eq!(after.matches("activate\n").count(), 1); +} + +#[tokio::test] +async fn foreign_catalog_is_rejected_before_any_management_write() { + let root = TestRoot::new(); + let profile = profile(&root); + let data_root = root.0.join("data"); + let mut session = + BootstrapSession::open(&data_root, b"profile", b"config", &iceberg_step_names()).unwrap(); + let credentials = ServerCredentials::load_or_create(&data_root).unwrap(); + let mut events = MonitorLog::open(&root.0.join("log"), profile.logs.clone()) + .await + .unwrap(); + fs::write(root.0.join("initialized"), "11111111-1111-4111-8111-111111111111").unwrap(); + fs::write(root.0.join("operation"), "foreign").unwrap(); + + assert!( + IcebergBootstrap::reconcile(&mut session, &profile, &credentials, &mut events) + .await + .is_err() + ); + assert_eq!(fs::read_to_string(root.0.join("calls")).unwrap(), "inspect\n"); + assert_eq!( + session.manifest().step_complete("iceberg-initialize"), + Some(false) + ); +} diff --git a/container/crowdb-monitor/tests/kv_bootstrap_test.rs b/container/crowdb-monitor/tests/kv_bootstrap_test.rs new file mode 100644 index 000000000..900056a1b --- /dev/null +++ b/container/crowdb-monitor/tests/kv_bootstrap_test.rs @@ -0,0 +1,269 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{kv_step_names, BootstrapSession, DeploymentProfile, KvBootstrap, MonitorLog}; +use crowdb_protocol::mgmt::GroupSummary; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::{TcpListener, TcpStream}; +use tokio::sync::oneshot; +use uuid::Uuid; + +struct TestDataRoot(PathBuf); + +impl TestDataRoot { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-kv-bootstrap-{}", Uuid::new_v4())); + fs::create_dir_all(&root).unwrap(); + Self(root.canonicalize().unwrap()) + } + + fn path(&self) -> &Path { + &self.0 + } +} + +impl Drop for TestDataRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn profile() -> DeploymentProfile { + DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap() +} + +fn session(root: &TestDataRoot, profile: &DeploymentProfile) -> BootstrapSession { + let names = kv_step_names(profile).unwrap(); + let steps = names.iter().map(String::as_str).collect::>(); + BootstrapSession::open(root.path(), b"profile", b"config", &steps).unwrap() +} + +async fn monitor_log(root: &TestDataRoot, profile: &DeploymentProfile) -> MonitorLog { + let log_root = root.path().join("log"); + fs::create_dir_all(&log_root).unwrap(); + MonitorLog::open(&log_root, profile.logs.clone()).await.unwrap() +} + +#[derive(Default)] +struct MockState { + groups: Vec, + system_posts: u32, + data_posts: u32, + lose_system_response: bool, +} + +struct MockKvServer { + base_url: String, + stop: oneshot::Sender<()>, + task: tokio::task::JoinHandle, +} + +impl MockKvServer { + async fn start(state: MockState) -> Self { + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let base_url = format!("http://{}", listener.local_addr().unwrap()); + let (stop, mut stop_rx) = oneshot::channel(); + let task = tokio::spawn(async move { + let mut state = state; + loop { + tokio::select! { + accepted = listener.accept() => { + let (stream, _) = accepted.unwrap(); + handle(stream, &mut state).await; + } + _ = &mut stop_rx => break, + } + } + state + }); + Self { base_url, stop, task } + } + + async fn finish(self) -> MockState { + self.stop.send(()).unwrap(); + self.task.await.unwrap() + } +} + +async fn handle(mut stream: TcpStream, state: &mut MockState) { + let mut bytes = Vec::new(); + let mut buffer = [0_u8; 4096]; + loop { + let count = stream.read(&mut buffer).await.unwrap(); + if count == 0 { + return; + } + bytes.extend_from_slice(&buffer[..count]); + if let Some(end) = bytes.windows(4).position(|window| window == b"\r\n\r\n") { + let headers = std::str::from_utf8(&bytes[..end]).unwrap(); + let content_length = headers + .lines() + .find_map(|line| { + let (name, value) = line.split_once(':')?; + name.eq_ignore_ascii_case("content-length") + .then(|| value.trim().parse::().ok()) + .flatten() + }) + .unwrap_or(0); + if bytes.len() >= end + 4 + content_length { + let request = headers.lines().next().unwrap(); + let mut words = request.split_whitespace(); + let method = words.next().unwrap(); + let path = words.next().unwrap(); + respond(&mut stream, state, method, path).await; + return; + } + } + } +} + +async fn respond(stream: &mut TcpStream, state: &mut MockState, method: &str, path: &str) { + let (status, body) = match (method, path) { + ("GET", "/stores") => { + let stores = if state.groups.is_empty() { + Vec::new() + } else { + vec![ + serde_json::json!({"store_id":0,"listen_addr":"127.0.0.1:10100","group_count":state.groups.len()}), + ] + }; + (200, serde_json::json!({"stores":stores}).to_string()) + } + ("GET", "/stores/0/groups") if state.groups.is_empty() => (404, "{}".into()), + ("GET", "/stores/0/groups") => (200, serde_json::to_string(&state.groups).unwrap()), + ("GET", "/stores/0/groups/0/ready" | "/stores/0/groups/1/ready") => { + let group_id = path.split('/').nth(4).unwrap().parse::().unwrap(); + let group = state + .groups + .iter() + .find(|group| group.group_id == group_id) + .unwrap(); + (200, serde_json::json!({"ready":true,"leader_id":group.local_replica_id,"voting_replicas":1,"reachable_replicas":1}).to_string()) + } + ("POST", "/system/init") => { + state.system_posts += 1; + state.groups.push(GroupSummary { + group_id: 0, + local_replica_id: 1, + leader_id: 1, + remote_count: 0, + }); + if state.lose_system_response { + state.lose_system_response = false; + return; + } + (201, "{}".into()) + } + ("POST", "/stores/0/groups") => { + state.data_posts += 1; + state.groups.push(GroupSummary { + group_id: 1, + local_replica_id: 2, + leader_id: 2, + remote_count: 0, + }); + (201, "{}".into()) + } + _ => (404, "{}".into()), + }; + let response = format!("HTTP/1.1 {status} Test\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len()); + stream.write_all(response.as_bytes()).await.unwrap(); +} + +#[tokio::test] +async fn creates_once_then_ready_restart_only_validates() { + let root = TestDataRoot::new(); + let profile = profile(); + let server = MockKvServer::start(MockState::default()).await; + let bootstrap = KvBootstrap::new(&server.base_url).unwrap(); + let mut initial = session(&root, &profile); + let mut events = monitor_log(&root, &profile).await; + bootstrap + .reconcile(&mut initial, &profile, &mut events) + .await + .unwrap(); + assert_eq!(initial.manifest().next_step(), None); + initial.mark_ready().unwrap(); + drop(initial); + let mut ready = session(&root, &profile); + bootstrap + .reconcile(&mut ready, &profile, &mut events) + .await + .unwrap(); + let state = server.finish().await; + assert_eq!((state.system_posts, state.data_posts), (1, 1)); + let body = fs::read_to_string(root.path().join("log/monitor/monitor.log")).unwrap(); + let entries = body + .lines() + .map(|line| serde_json::from_str::(line).unwrap()) + .collect::>(); + assert_eq!(entries.len(), 4); + assert_eq!(entries[0]["kind"], "bootstrap_step_started"); + assert_eq!(entries[0]["service"], "kv-group-0-0"); + assert_eq!(entries[1]["kind"], "bootstrap_step_completed"); + assert_eq!(entries[2]["service"], "kv-group-0-1"); + assert_eq!(entries[3]["kind"], "bootstrap_step_completed"); +} + +#[tokio::test] +async fn lost_create_response_is_proven_without_replaying_post() { + let root = TestDataRoot::new(); + let profile = profile(); + let server = MockKvServer::start(MockState { + lose_system_response: true, + ..MockState::default() + }) + .await; + let bootstrap = KvBootstrap::new(&server.base_url).unwrap(); + let mut session = session(&root, &profile); + let mut events = monitor_log(&root, &profile).await; + bootstrap + .reconcile(&mut session, &profile, &mut events) + .await + .unwrap(); + let state = server.finish().await; + assert_eq!((state.system_posts, state.data_posts), (1, 1)); +} + +#[tokio::test] +async fn conflicting_existing_group_fails_without_mutation() { + let root = TestDataRoot::new(); + let profile = profile(); + let server = MockKvServer::start(MockState { + groups: vec![GroupSummary { + group_id: 0, + local_replica_id: 99, + leader_id: 99, + remote_count: 0, + }], + ..MockState::default() + }) + .await; + let bootstrap = KvBootstrap::new(&server.base_url).unwrap(); + let mut session = session(&root, &profile); + let mut events = monitor_log(&root, &profile).await; + assert!(bootstrap + .reconcile(&mut session, &profile, &mut events) + .await + .is_err()); + assert_eq!(session.manifest().next_step(), Some("kv-group-0-0")); + let state = server.finish().await; + assert_eq!((state.system_posts, state.data_posts), (0, 0)); + let body = fs::read_to_string(root.path().join("log/monitor/monitor.log")).unwrap(); + let entries = body + .lines() + .map(|line| serde_json::from_str::(line).unwrap()) + .collect::>(); + assert_eq!(entries.last().unwrap()["kind"], "bootstrap_failed"); + assert_eq!(entries.last().unwrap()["service"], "kv-group-0-0"); +} diff --git a/container/crowdb-monitor/tests/kv_process_bootstrap_test.rs b/container/crowdb-monitor/tests/kv_process_bootstrap_test.rs new file mode 100644 index 000000000..cae9ef034 --- /dev/null +++ b/container/crowdb-monitor/tests/kv_process_bootstrap_test.rs @@ -0,0 +1,130 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::BTreeMap; +use std::fs; +use std::os::unix::fs::symlink; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{kv_step_names, BootstrapSession, DeploymentProfile, KvBootstrap, Supervisor}; +use uuid::Uuid; + +struct TestRoot(PathBuf); + +impl TestRoot { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-real-kv-{}", Uuid::new_v4())); + for path in ["bin", "data", "run"] { + fs::create_dir_all(root.join(path)).unwrap(); + } + Self(root.canonicalize().unwrap()) + } + + fn profile(&self, binary: &Path, management_port: u16, rpc_port: u16) -> DeploymentProfile { + let mut profile = DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap(); + profile.paths.install_root.clone_from(&self.0); + profile.paths.bin_root = self.0.join("bin"); + profile.paths.template_root = self.0.join("templates"); + profile.paths.data_root = self.0.join("data"); + profile.paths.run_root = self.0.join("run"); + profile.paths.log_root = self.0.join("data/log"); + for disk in &mut profile.disks { + disk.path = self.0.join("data/disks").join(disk.path.file_name().unwrap()); + } + profile.services.retain(|service| service.id == "kv"); + let service = &mut profile.services[0]; + service.program = self.0.join("bin/crowdb-kv-server"); + symlink(binary, &service.program).unwrap(); + service.config_template = None; + service.fence_listeners = vec![ + format!("127.0.0.1:{management_port}"), + format!("127.0.0.1:{rpc_port}"), + ]; + service.args = vec![ + "--root".into(), + self.0.join("data/kv/node-1").to_string_lossy().into_owned(), + "--management-addr".into(), + "127.0.0.1".into(), + "--management-port".into(), + management_port.to_string(), + "--ports".into(), + rpc_port.to_string(), + ]; + service.probe.target = format!("http://127.0.0.1:{management_port}/health"); + profile.validate().unwrap(); + profile + } + + fn session(&self, profile: &DeploymentProfile) -> BootstrapSession { + let names = kv_step_names(profile).unwrap(); + let steps = names.iter().map(String::as_str).collect::>(); + BootstrapSession::open(&self.0.join("data"), b"profile", b"config", &steps).unwrap() + } +} + +impl Drop for TestRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +#[tokio::test] +async fn monitor_creates_real_kv_groups_then_validates_after_restart() { + let Some(binary) = crowdb_test_harness::cluster::crowdb_kv_server_bin() else { + eprintln!("skipping real KV bootstrap: crowdb-kv-server binary is unavailable"); + return; + }; + let root = TestRoot::new(); + let management = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let management_port = management.local_addr().unwrap().port(); + drop(management); + let rpc = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let rpc_port = rpc.local_addr().unwrap().port(); + drop(rpc); + let profile = root.profile(&binary, management_port, rpc_port); + let mut session = root.session(&profile); + fs::create_dir_all(root.0.join("data/kv/node-1")).unwrap(); + fs::create_dir_all(root.0.join("data/log")).unwrap(); + let endpoint = format!("http://127.0.0.1:{management_port}"); + let bootstrap = KvBootstrap::new(&endpoint).unwrap(); + let mut supervisor = Supervisor::new( + profile.clone(), + session.manifest().deployment_id(), + &root.0.join("data/log"), + &root.0.join("run"), + ) + .await + .unwrap(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + bootstrap + .reconcile(&mut session, &profile, supervisor.monitor_log_mut()) + .await + .unwrap(); + session.mark_ready().unwrap(); + supervisor.mark_ready().await.unwrap(); + supervisor.shutdown().await.unwrap(); + drop(supervisor); + let mut restart = root.session(&profile); + let mut supervisor = Supervisor::new( + profile.clone(), + restart.manifest().deployment_id(), + &root.0.join("data/log"), + &root.0.join("run"), + ) + .await + .unwrap(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + bootstrap + .reconcile(&mut restart, &profile, supervisor.monitor_log_mut()) + .await + .unwrap(); + supervisor.mark_ready().await.unwrap(); + supervisor.shutdown().await.unwrap(); +} diff --git a/container/crowdb-monitor/tests/liveness_test.rs b/container/crowdb-monitor/tests/liveness_test.rs new file mode 100644 index 000000000..0b434fd41 --- /dev/null +++ b/container/crowdb-monitor/tests/liveness_test.rs @@ -0,0 +1,48 @@ +use std::fs; +use std::path::PathBuf; +use std::process::Command; + +use crowdb_monitor::{probe_liveness, LivenessServer}; +use uuid::Uuid; + +struct TestRunRoot(PathBuf); + +impl TestRunRoot { + fn new() -> Self { + let root = std::env::temp_dir().join(format!("cm-live-{}", Uuid::new_v4().simple())); + fs::create_dir_all(&root).unwrap(); + Self(root.canonicalize().unwrap()) + } +} + +impl Drop for TestRunRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn liveness_requires_responsive_monitor_not_status_file() { + let root = TestRunRoot::new(); + assert!(probe_liveness(&root.0).await.is_err()); + let server = LivenessServer::start(&root.0).unwrap(); + assert!(LivenessServer::start(&root.0).is_err()); + assert!(probe_liveness(&root.0).await.is_ok()); + let path = root.0.clone(); + let success = tokio::task::spawn_blocking(move || { + Command::new(env!("CARGO_BIN_EXE_crowdb-monitor")) + .args(["liveness", "--run-root"]) + .arg(path) + .output() + .unwrap() + .status + .success() + }) + .await + .unwrap(); + assert!(success); + drop(server); + assert!(probe_liveness(&root.0).await.is_err()); +} diff --git a/container/crowdb-monitor/tests/manifest_test.rs b/container/crowdb-monitor/tests/manifest_test.rs new file mode 100644 index 000000000..04f950f52 --- /dev/null +++ b/container/crowdb-monitor/tests/manifest_test.rs @@ -0,0 +1,144 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs; +use std::os::unix::fs::{symlink, PermissionsExt}; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{BootstrapSession, ManifestState}; +use uuid::Uuid; + +const STEPS: &[&str] = &["kv", "storage", "catalog"]; + +struct TestDataRoot(PathBuf); + +impl TestDataRoot { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-manifest-{}", Uuid::new_v4())); + fs::create_dir_all(&root).unwrap(); + Self(root.canonicalize().unwrap()) + } + + fn path(&self) -> &Path { + &self.0 + } + + fn manifest_path(&self) -> PathBuf { + self.0.join("bootstrap/manifest.json") + } +} + +impl Drop for TestDataRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn open(root: &TestDataRoot) -> BootstrapSession { + BootstrapSession::open(root.path(), b"profile", b"configuration", STEPS).unwrap() +} + +#[test] +fn interrupted_steps_resume_with_stable_identity_and_order() { + let root = TestDataRoot::new(); + let mut session = open(&root); + let identity = session.manifest().deployment_id(); + let operation_id = session.manifest().operation_id("kv").unwrap(); + assert_eq!(session.manifest().state(), ManifestState::Initializing); + assert_eq!(session.manifest().next_step(), Some("kv")); + assert_eq!( + fs::metadata(root.manifest_path()).unwrap().permissions().mode() & 0o777, + 0o600 + ); + assert!(session.complete_step("catalog").is_err()); + assert!(session.mark_ready().is_err()); + session.complete_step("kv").unwrap(); + drop(session); + + let mut resumed = open(&root); + assert_eq!(resumed.manifest().deployment_id(), identity); + assert_eq!(resumed.manifest().operation_id("kv").unwrap(), operation_id); + assert_eq!(resumed.manifest().next_step(), Some("storage")); + assert!(resumed.complete_step("kv").is_err()); + resumed.complete_step("storage").unwrap(); + resumed.complete_step("catalog").unwrap(); + resumed.mark_ready().unwrap(); + drop(resumed); + + let ready = open(&root); + assert_eq!(ready.manifest().state(), ManifestState::Ready); + assert_eq!(ready.manifest().deployment_id(), identity); + assert_eq!(ready.manifest().next_step(), None); + assert!(ready.manifest().operation_id("unknown").is_err()); +} + +#[test] +fn changed_inputs_and_step_plan_fail_without_mutation() { + let root = TestDataRoot::new(); + open(&root); + let original = fs::read(root.manifest_path()).unwrap(); + assert!(BootstrapSession::open(root.path(), b"different", b"configuration", STEPS).is_err()); + assert!(BootstrapSession::open(root.path(), b"profile", b"different", STEPS).is_err()); + assert!(BootstrapSession::open(root.path(), b"profile", b"configuration", &["kv", "catalog"]).is_err()); + assert_eq!(fs::read(root.manifest_path()).unwrap(), original); +} + +#[test] +fn nonempty_and_corrupt_roots_fail_closed() { + let nonempty = TestDataRoot::new(); + fs::write(nonempty.path().join("orphan"), b"data").unwrap(); + assert!(BootstrapSession::open(nonempty.path(), b"profile", b"configuration", STEPS).is_err()); + assert!(!nonempty.path().join("bootstrap").exists()); + + let corrupt = TestDataRoot::new(); + open(&corrupt); + fs::write(corrupt.manifest_path(), b"not json").unwrap(); + assert!(BootstrapSession::open(corrupt.path(), b"profile", b"configuration", STEPS).is_err()); + assert_eq!(fs::read(corrupt.manifest_path()).unwrap(), b"not json"); +} + +#[test] +fn symlinked_or_wrong_mode_manifest_is_rejected() { + let root = TestDataRoot::new(); + open(&root); + fs::set_permissions(root.manifest_path(), fs::Permissions::from_mode(0o644)).unwrap(); + assert!(BootstrapSession::open(root.path(), b"profile", b"configuration", STEPS).is_err()); + fs::set_permissions(root.manifest_path(), fs::Permissions::from_mode(0o600)).unwrap(); + let original = root.path().join("original.json"); + fs::rename(root.manifest_path(), &original).unwrap(); + symlink(&original, root.manifest_path()).unwrap(); + assert!(BootstrapSession::open(root.path(), b"profile", b"configuration", STEPS).is_err()); +} + +#[test] +fn invalid_plan_is_rejected_before_root_mutation() { + let root = TestDataRoot::new(); + assert!(BootstrapSession::open(root.path(), b"profile", b"configuration", &[]).is_err()); + assert!(BootstrapSession::open(root.path(), b"profile", b"configuration", &["kv", "kv"]).is_err()); + assert!(BootstrapSession::open(root.path(), b"profile", b"configuration", &["bad/name"]).is_err()); + assert!(fs::read_dir(root.path()).unwrap().next().is_none()); +} + +#[test] +fn reserved_uuidv7_survives_interruption_before_catalog_commit() { + let root = TestDataRoot::new(); + let mut session = open(&root); + let operation = session.reserve_operation("kv").unwrap(); + assert_eq!(operation.get_version_num(), 7); + drop(session); + + let mut resumed = open(&root); + assert_eq!(resumed.reserve_operation("kv").unwrap(), operation); + assert!(resumed.complete_catalog_step("kv", Uuid::nil()).is_err()); + let catalog = Uuid::new_v4(); + resumed.complete_catalog_step("kv", catalog).unwrap(); + drop(resumed); + + let reopened = open(&root); + assert_eq!(reopened.manifest().step_operation("kv"), Some(operation)); + assert_eq!(reopened.manifest().step_catalog("kv"), Some(catalog)); +} diff --git a/container/crowdb-monitor/tests/monitor_log_test.rs b/container/crowdb-monitor/tests/monitor_log_test.rs new file mode 100644 index 000000000..bda5408ac --- /dev/null +++ b/container/crowdb-monitor/tests/monitor_log_test.rs @@ -0,0 +1,111 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{LogProfile, MonitorEvent, MonitorEventKind, MonitorLog}; +use uuid::Uuid; + +struct TestLogs(PathBuf); + +impl TestLogs { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-events-{}", Uuid::new_v4())); + fs::create_dir_all(&root).unwrap(); + Self(root.canonicalize().unwrap()) + } +} + +impl Drop for TestLogs { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +#[tokio::test] +async fn important_events_are_persisted_under_monitor_log() { + let logs = TestLogs::new(); + let policy = LogProfile { + max_file_bytes: 1024 * 1024, + max_files: 2, + mirror_warnings_to_stderr: false, + }; + let mut monitor = MonitorLog::open(&logs.0, policy).await.unwrap(); + monitor + .record(&MonitorEvent { + kind: MonitorEventKind::ChildStarted, + service: Some("kv"), + pid: Some(123), + attempt: None, + }) + .await + .unwrap(); + monitor + .record(&MonitorEvent { + kind: MonitorEventKind::Restarting, + service: Some("kv"), + pid: None, + attempt: Some(2), + }) + .await + .unwrap(); + let body = fs::read_to_string(logs.0.join("monitor/monitor.log")).unwrap(); + let entries = body + .lines() + .map(|line| serde_json::from_str::(line).unwrap()) + .collect::>(); + assert_eq!(entries.len(), 2); + assert_eq!(entries[0]["kind"], "child_started"); + assert_eq!(entries[1]["kind"], "restarting"); + assert_eq!(entries[1]["level"], "warn"); + assert_eq!(entries[1]["attempt"], 2); +} + +#[tokio::test] +async fn rotation_keeps_whole_events_within_file_and_count_bounds() { + let logs = TestLogs::new(); + let mut monitor = MonitorLog::open( + &logs.0, + LogProfile { + max_file_bytes: 512, + max_files: 3, + mirror_warnings_to_stderr: false, + }, + ) + .await + .unwrap(); + for attempt in 0..100 { + monitor + .record(&MonitorEvent { + kind: MonitorEventKind::Restarting, + service: Some("web"), + pid: None, + attempt: Some(attempt), + }) + .await + .unwrap(); + } + drop(monitor); + let files = fs::read_dir(logs.0.join("monitor")) + .unwrap() + .collect::, _>>() + .unwrap(); + assert_eq!(files.len(), 3); + let mut attempts = Vec::new(); + for file in files { + assert!(file.metadata().unwrap().len() <= 512); + for line in fs::read_to_string(file.path()).unwrap().lines() { + let event: serde_json::Value = serde_json::from_str(line).unwrap(); + attempts.push(event["attempt"].as_u64().unwrap()); + } + } + attempts.sort_unstable(); + assert_eq!(attempts.last(), Some(&99)); + assert!(attempts.len() < 100); + assert!(attempts.windows(2).all(|pair| pair[1] == pair[0] + 1)); +} diff --git a/container/crowdb-monitor/tests/preview_run_test.rs b/container/crowdb-monitor/tests/preview_run_test.rs new file mode 100644 index 000000000..d07530558 --- /dev/null +++ b/container/crowdb-monitor/tests/preview_run_test.rs @@ -0,0 +1,101 @@ +use std::fs; +use std::os::unix::fs::symlink; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{run_preview, DeploymentProfile, MonitorPhase, StatusStore}; +use uuid::Uuid; + +struct TestRoot(PathBuf); + +impl TestRoot { + fn new() -> Self { + let path = std::env::temp_dir().join(format!("cm-preview-{}", Uuid::new_v4().simple())); + for name in ["bin", "templates", "data", "run"] { + fs::create_dir_all(path.join(name)).unwrap(); + } + Self(path.canonicalize().unwrap()) + } + + fn profile_path(&self) -> PathBuf { + let mut profile = DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap(); + profile.paths.install_root.clone_from(&self.0); + profile.paths.bin_root = self.0.join("bin"); + profile.paths.template_root = self.0.join("templates"); + profile.paths.data_root = self.0.join("data"); + profile.paths.run_root = self.0.join("run"); + profile.paths.log_root = self.0.join("data/log"); + for disk in &mut profile.disks { + disk.path = self.0.join("data/disks").join(disk.path.file_name().unwrap()); + } + let source = Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/templates"); + for service in &mut profile.services { + let name = service.program.file_name().unwrap(); + service.program = self.0.join("bin").join(name); + if service.id == "kv" { + symlink("/bin/false", &service.program).unwrap(); + for argument in &mut service.args { + if argument == "/opt/crowdb/data/kv/node-1" { + *argument = self.0.join("data/kv/node-1").to_string_lossy().into_owned(); + } + } + } + if let Some(template) = &service.config_template { + let name = template.file_name().unwrap(); + fs::copy(source.join(name), self.0.join("templates").join(name)).unwrap(); + service.config_template = Some(self.0.join("templates").join(name)); + } + } + profile.validate().unwrap(); + let path = self.0.join("profile.toml"); + fs::write(&path, toml::to_string(&profile).unwrap()).unwrap(); + path + } +} + +impl Drop for TestRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +#[tokio::test] +async fn failed_child_never_marks_preview_ready_and_template_drift_fails_closed() { + let root = TestRoot::new(); + let profile = root.profile_path(); + assert!(run_preview(&profile).await.is_err()); + let status = StatusStore::open(&root.0.join("run")) + .unwrap() + .read(std::time::Duration::from_secs(10)) + .unwrap(); + assert_eq!(status.phase, MonitorPhase::Draining); + assert!(root.0.join("data/bootstrap/manifest.json").exists()); + assert!(root.0.join("data/secrets/server.env").exists()); + assert!(StatusStore::open(&root.0.join("run")) + .unwrap() + .readiness(std::time::Duration::from_secs(10)) + .is_err()); + + let template = root.0.join("templates/kv.toml"); + fs::write(&template, format!("{}\n", fs::read_to_string(&template).unwrap())).unwrap(); + let before = fs::read(root.0.join("data/bootstrap/manifest.json")).unwrap(); + assert!(run_preview(&profile).await.is_err()); + assert_eq!( + fs::read(root.0.join("data/bootstrap/manifest.json")).unwrap(), + before + ); +} + +#[tokio::test] +async fn nonempty_uninitialized_root_is_not_adopted() { + let root = TestRoot::new(); + let profile = root.profile_path(); + fs::write(root.0.join("data/foreign"), b"unrelated").unwrap(); + assert!(run_preview(&profile).await.is_err()); + assert!(!root.0.join("data/bootstrap").exists()); + assert!(!root.0.join("data/secrets").exists()); +} diff --git a/container/crowdb-monitor/tests/probe_test.rs b/container/crowdb-monitor/tests/probe_test.rs new file mode 100644 index 000000000..78ad7d09b --- /dev/null +++ b/container/crowdb-monitor/tests/probe_test.rs @@ -0,0 +1,127 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::BTreeMap; +use std::path::PathBuf; + +use crowdb_monitor::{ProbeExecutor, ProbeKind, ProbeProfile, RestartProfile, ServiceProfile}; +use crowdb_rpc_ffi::RpcServer; +use tokio::net::TcpListener; + +fn service(kind: ProbeKind, target: String) -> ServiceProfile { + ServiceProfile { + id: "test".into(), + program: PathBuf::from("/bin/true"), + args: Vec::new(), + env: BTreeMap::new(), + dependencies: Vec::new(), + fence_listeners: Vec::new(), + config_template: None, + probe: ProbeProfile { + kind, + target, + bearer_env: None, + timeout_ms: 1000, + failure_threshold: 1, + }, + restart: RestartProfile { + max_attempts: 1, + backoff_base_ms: 1, + backoff_max_ms: 1, + stable_after_ms: 60_000, + }, + } +} + +#[tokio::test] +async fn tcp_probe_requires_a_listener() { + let probes = ProbeExecutor::new(false).unwrap(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let target = listener.local_addr().unwrap().to_string(); + assert!(probes + .probe_service(&service(ProbeKind::Tcp, target.clone()), &BTreeMap::new()) + .await + .is_ok()); + drop(listener); + assert!(probes + .probe_service(&service(ProbeKind::Tcp, target), &BTreeMap::new()) + .await + .is_err()); +} + +#[tokio::test] +async fn http_probe_requires_success_status() { + let probes = ProbeExecutor::new(false).unwrap(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let target = format!("http://{}/ready", listener.local_addr().unwrap()); + let server = tokio::spawn(async move { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + for response in [ + "HTTP/1.1 503 Unavailable\r\nContent-Length: 0\r\n\r\n", + "HTTP/1.1 200 OK\r\nContent-Length: 0\r\n\r\n", + ] { + let (mut stream, _) = listener.accept().await.unwrap(); + let mut request = [0_u8; 4]; + stream.read_exact(&mut request).await.unwrap(); + stream.write_all(response.as_bytes()).await.unwrap(); + } + }); + assert!(probes + .probe_service(&service(ProbeKind::Http, target.clone()), &BTreeMap::new()) + .await + .is_err()); + assert!(probes + .probe_service(&service(ProbeKind::Http, target), &BTreeMap::new()) + .await + .is_ok()); + server.await.unwrap(); +} + +#[tokio::test] +async fn authenticated_probe_uses_runtime_token_without_storing_it_in_profile() { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + + let probes = ProbeExecutor::new(false).unwrap(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let mut service = service( + ProbeKind::Http, + format!("http://{}/v1/config", listener.local_addr().unwrap()), + ); + service.probe.bearer_env = Some("CROWDB_ICEBERG_READ_TOKEN".into()); + assert!(probes.probe_service(&service, &BTreeMap::new()).await.is_err()); + let server = tokio::spawn(async move { + let (mut stream, _) = listener.accept().await.unwrap(); + let mut request = vec![0_u8; 4096]; + let size = stream.read(&mut request).await.unwrap(); + let body = std::str::from_utf8(&request[..size]).unwrap(); + assert!(body + .to_ascii_lowercase() + .contains("authorization: bearer private-token\r\n")); + stream + .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 0\r\n\r\n") + .await + .unwrap(); + }); + let environment = BTreeMap::from([("CROWDB_ICEBERG_READ_TOKEN".into(), "private-token".into())]); + probes.probe_service(&service, &environment).await.unwrap(); + server.await.unwrap(); +} + +#[tokio::test] +async fn rpc_ping_requires_an_application_response() { + let probes = ProbeExecutor::new(true).unwrap(); + let server = RpcServer::new(None); + server.listen("127.0.0.1", 0).unwrap(); + server.start(); + let target = format!("127.0.0.1:{}", server.port()); + probes + .probe_service(&service(ProbeKind::RpcPing, target), &BTreeMap::new()) + .await + .unwrap(); + server.stop(); + + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let mut stalled = service(ProbeKind::RpcPing, listener.local_addr().unwrap().to_string()); + stalled.probe.timeout_ms = 100; + assert!(probes.probe_service(&stalled, &BTreeMap::new()).await.is_err()); +} diff --git a/container/crowdb-monitor/tests/process_test.rs b/container/crowdb-monitor/tests/process_test.rs new file mode 100644 index 000000000..68f6c9107 --- /dev/null +++ b/container/crowdb-monitor/tests/process_test.rs @@ -0,0 +1,174 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::BTreeMap; +use std::fs; +use std::path::{Path, PathBuf}; +use std::time::Duration; + +use crowdb_monitor::{LogProfile, ProbeKind, ProbeProfile, ProcessManager, RestartProfile, ServiceProfile}; +use uuid::Uuid; + +struct TestLogs(PathBuf); + +impl TestLogs { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-process-{}", Uuid::new_v4())); + fs::create_dir_all(&root).unwrap(); + Self(root.canonicalize().unwrap()) + } +} + +impl Drop for TestLogs { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn policy(max_files: u16) -> LogProfile { + LogProfile { + max_file_bytes: 1024 * 1024, + max_files, + mirror_warnings_to_stderr: false, + } +} + +fn service(script: &str) -> ServiceProfile { + ServiceProfile { + id: "fake".into(), + program: PathBuf::from("/bin/sh"), + args: vec!["-c".into(), script.into()], + env: BTreeMap::new(), + dependencies: Vec::new(), + fence_listeners: Vec::new(), + config_template: None, + probe: ProbeProfile { + kind: ProbeKind::Tcp, + target: "127.0.0.1:1".into(), + bearer_env: None, + timeout_ms: 100, + failure_threshold: 1, + }, + restart: RestartProfile { + max_attempts: 1, + backoff_base_ms: 1, + backoff_max_ms: 1, + stable_after_ms: 60_000, + }, + } +} + +#[tokio::test] +async fn owned_child_is_reaped_before_replacement() { + let logs = TestLogs::new(); + let mut manager = ProcessManager::new(logs.0.clone(), policy(2)).await.unwrap(); + let service = service("printf 'started\\n'; exec sleep 30"); + let first_pid = manager.start(&service, &BTreeMap::new()).await.unwrap(); + assert!(manager.start(&service, &BTreeMap::new()).await.is_err()); + assert_eq!(manager.pid("fake"), Some(first_pid)); + assert!(manager.alive("fake").unwrap()); + let log_path = logs.0.join("fake/service.log"); + tokio::time::timeout(Duration::from_secs(2), async { + loop { + if fs::read_to_string(&log_path).is_ok_and(|body| body.contains("started")) { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + manager.stop("fake", Duration::from_secs(2)).await.unwrap(); + assert_eq!(manager.pid("fake"), None); + let second_pid = manager.start(&service, &BTreeMap::new()).await.unwrap(); + assert_ne!(second_pid, first_pid); + manager.stop("fake", Duration::from_secs(2)).await.unwrap(); + assert!(fs::read_to_string(log_path).unwrap().contains("started")); + let events = fs::read_to_string(logs.0.join("monitor/monitor.log")).unwrap(); + assert_eq!(events.matches("child_started").count(), 2); + assert_eq!(events.matches("child_stopped").count(), 2); +} + +#[tokio::test] +async fn logs_rotate_with_total_file_and_byte_limits() { + let logs = TestLogs::new(); + let mut manager = ProcessManager::new(logs.0.clone(), policy(2)).await.unwrap(); + let service = + service("count=0; while [ \"$count\" -lt 2500 ]; do printf '%01024d\\n' 0; count=$((count+1)); done"); + manager.start(&service, &BTreeMap::new()).await.unwrap(); + for _ in 0..100 { + if !manager.alive("fake").unwrap() { + break; + } + tokio::time::sleep(Duration::from_millis(50)).await; + } + assert!(!manager.alive("fake").unwrap()); + manager.stop("fake", Duration::from_secs(2)).await.unwrap(); + let files = fs::read_dir(logs.0.join("fake")) + .unwrap() + .map(|entry| entry.unwrap()) + .collect::>(); + assert!(files.len() <= 2); + assert!(files + .iter() + .all(|entry| entry.metadata().unwrap().len() <= 1024 * 1024)); +} + +#[tokio::test] +async fn child_restarts_prune_old_pid_logs_without_crossing_log_channels() { + let logs = TestLogs::new(); + let directory = logs.0.join("fake"); + fs::create_dir_all(&directory).unwrap(); + for prefix in ["crowdb-fake", "crowdb-fake-rpc"] { + for pid in 1..=6 { + fs::write( + directory.join(format!("{prefix}-20260101-010101.000-{pid}.log")), + "old", + ) + .unwrap(); + } + } + fs::write(directory.join("operator-notes.txt"), "keep").unwrap(); + let mut manager = ProcessManager::new(logs.0.clone(), policy(2)).await.unwrap(); + let child = service("printf 'new' > \"$LOG_DIR/crowdb-fake-20260102-010101.000-$$.log\"; exec sleep 30"); + let pid = manager + .start( + &child, + &BTreeMap::from([("LOG_DIR".into(), directory.to_string_lossy().into_owned())]), + ) + .await + .unwrap(); + let active = directory.join(format!("crowdb-fake-20260102-010101.000-{pid}.log")); + tokio::time::timeout(Duration::from_secs(2), async { + while !active.exists() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + manager.stop("fake", Duration::from_secs(2)).await.unwrap(); + let names = fs::read_dir(&directory) + .unwrap() + .map(|entry| entry.unwrap().file_name().to_string_lossy().into_owned()) + .collect::>(); + assert!( + names + .iter() + .filter(|name| name.starts_with("crowdb-fake-2026")) + .count() + <= 2 + ); + assert_eq!( + names + .iter() + .filter(|name| name.starts_with("crowdb-fake-rpc-")) + .count(), + 1 + ); + assert!(directory.join("operator-notes.txt").exists()); + assert_eq!(fs::read_to_string(active).unwrap(), "new"); +} diff --git a/container/crowdb-monitor/tests/profile_test.rs b/container/crowdb-monitor/tests/profile_test.rs new file mode 100644 index 000000000..1416a6995 --- /dev/null +++ b/container/crowdb-monitor/tests/profile_test.rs @@ -0,0 +1,137 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_monitor::{DeploymentProfile, ProfileError}; + +fn profile() -> String { + r#" +version = 1 +name = "test-profile" +display_name = "Test Profile" +placement_mode = "test" +s3_tenant = "test" +iceberg_catalog = "test" + +[paths] +install_root = "/opt/crowdb" +bin_root = "/opt/crowdb/bin" +template_root = "/opt/crowdb/etc/templates" +data_root = "/opt/crowdb/data" +run_root = "/opt/crowdb/run" +log_root = "/opt/crowdb/data/log" + +[logs] +max_file_bytes = 31457280 +max_files = 5 +mirror_warnings_to_stderr = true + +[[nodes]] +node_id = 1 +rack_id = 1 + +[[groups]] +store_id = 0 +group_id = 0 +replica_id = 1 +node_id = 1 +rpc_endpoint = "127.0.0.1:10100" +role = "system" + +[[groups]] +store_id = 0 +group_id = 1 +replica_id = 2 +node_id = 1 +rpc_endpoint = "127.0.0.1:10100" +role = "data" + +[[disks]] +disk_id = "00000000000000000000000000000001" +disk_group_id = 101 +node_id = 1 +path = "/opt/crowdb/data/disks/disk-0001.img" +capacity_bytes = 17179869184 +zone_size_bytes = 17179869184 + +[[public_endpoints]] +id = "web" +bind = "0.0.0.0" +port = 14000 + +[[services]] +id = "kv" +program = "/opt/crowdb/bin/crowdb-kv-server" +args = [] +dependencies = [] +config_template = "/opt/crowdb/etc/templates/kv.toml" +[services.probe] +kind = "http" +target = "http://127.0.0.1:10000/health" +timeout_ms = 1000 +failure_threshold = 3 +[services.restart] +max_attempts = 5 +backoff_base_ms = 100 +backoff_max_ms = 1000 + +[[services]] +id = "web" +program = "/opt/crowdb/bin/crowdb-web" +args = [] +dependencies = ["kv"] +config_template = "/opt/crowdb/etc/templates/crowdb-web.toml" +[services.probe] +kind = "http" +target = "http://127.0.0.1:14000/healthz" +timeout_ms = 1000 +failure_threshold = 3 +[services.restart] +max_attempts = 5 +backoff_base_ms = 100 +backoff_max_ms = 1000 +"# + .into() +} + +#[test] +fn valid_profile_orders_dependencies() { + let profile = DeploymentProfile::parse(&profile()).unwrap(); + let order = profile + .services_in_start_order() + .unwrap() + .into_iter() + .map(|service| service.id.as_str()) + .collect::>(); + assert_eq!(order, ["kv", "web"]); +} + +#[test] +fn dependency_cycle_is_rejected() { + let body = profile().replace("dependencies = []", "dependencies = [\"web\"]"); + let error = DeploymentProfile::parse(&body).unwrap_err(); + assert!(matches!(error, ProfileError::Invalid(message) if message.contains("cycle"))); +} + +#[test] +fn secret_environment_is_rejected() { + let body = profile().replace("args = []", "args = []\nenv = { API_TOKEN = \"secret\" }"); + let error = DeploymentProfile::parse(&body).unwrap_err(); + assert!(matches!(error, ProfileError::Invalid(message) if message.contains("secret-like"))); +} + +#[test] +fn path_escape_is_rejected() { + let body = profile().replace( + "/opt/crowdb/data/disks/disk-0001.img", + "/opt/crowdb/data/../etc/disk-0001.img", + ); + let error = DeploymentProfile::parse(&body).unwrap_err(); + assert!(matches!(error, ProfileError::Invalid(message) if message.contains("data_root/disks"))); +} + +#[test] +fn unbounded_log_policy_is_rejected() { + let body = profile().replace("max_files = 5", "max_files = 0"); + let error = DeploymentProfile::parse(&body).unwrap_err(); + assert!(matches!(error, ProfileError::Invalid(message) if message.contains("log rotation"))); +} diff --git a/container/crowdb-monitor/tests/render_test.rs b/container/crowdb-monitor/tests/render_test.rs new file mode 100644 index 000000000..644a18679 --- /dev/null +++ b/container/crowdb-monitor/tests/render_test.rs @@ -0,0 +1,78 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{render_configs, DeploymentProfile}; +use uuid::Uuid; + +struct TestDirs(PathBuf); + +impl TestDirs { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-render-{}", Uuid::new_v4())); + fs::create_dir_all(root.join("templates")).unwrap(); + fs::create_dir(root.join("run")).unwrap(); + let source = Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/templates"); + for entry in fs::read_dir(source).unwrap() { + let entry = entry.unwrap(); + fs::copy(entry.path(), root.join("templates").join(entry.file_name())).unwrap(); + } + Self(root.canonicalize().unwrap()) + } + + fn templates(&self) -> PathBuf { + self.0.join("templates") + } + + fn run(&self) -> PathBuf { + self.0.join("run") + } +} + +impl Drop for TestDirs { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn profile() -> DeploymentProfile { + DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap() +} + +#[test] +fn renders_profile_paths_and_topology_without_secrets() { + let dirs = TestDirs::new(); + let outputs = render_configs(&profile(), &dirs.templates(), &dirs.run()).unwrap(); + assert_eq!(outputs.len(), 6); + let diskio = fs::read_to_string(dirs.run().join("config/diskio.toml")).unwrap(); + assert!(diskio.contains("path = \"/opt/crowdb/data/disks/disk-0004.img\"")); + assert!(diskio.contains("zone_capacity = 17179869184")); + let web = fs::read_to_string(dirs.run().join("config/crowdb-web.toml")).unwrap(); + assert!(web.contains("monitor_status = \"/opt/crowdb/run/status/monitor.json\"")); + assert!(!web.contains("{{")); + for output in outputs { + let body = fs::read_to_string(output.path).unwrap(); + assert!(!body.contains("MASTER_KEY")); + assert!(!body.contains("ICEBERG_WRITE_TOKEN")); + toml::from_str::(&body).unwrap(); + } +} + +#[test] +fn unknown_or_malformed_variables_fail_closed() { + let dirs = TestDirs::new(); + let path = dirs.templates().join("diskio.toml"); + fs::write(&path, b"value = \"{{unknown}}\"\n").unwrap(); + assert!(render_configs(&profile(), &dirs.templates(), &dirs.run()).is_err()); + fs::write(&path, b"value = \"{{disk.0.path\"\n").unwrap(); + assert!(render_configs(&profile(), &dirs.templates(), &dirs.run()).is_err()); +} diff --git a/container/crowdb-monitor/tests/single_node_profile_test.rs b/container/crowdb-monitor/tests/single_node_profile_test.rs new file mode 100644 index 000000000..302d29319 --- /dev/null +++ b/container/crowdb-monitor/tests/single_node_profile_test.rs @@ -0,0 +1,104 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; + +use crowdb_monitor::{DeploymentProfile, GroupRole}; + +fn profile_path() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml") +} + +#[test] +fn single_node_preview_has_exact_topology_and_endpoints() { + let profile = DeploymentProfile::load(profile_path()).unwrap(); + assert_eq!(profile.name, "single-node-container"); + assert_eq!(profile.display_name, "CROWDB Single-Node Container"); + assert_eq!(profile.logs.max_file_bytes, 30 * 1024 * 1024); + assert_eq!(profile.logs.max_files, 5); + assert!(profile.logs.mirror_warnings_to_stderr); + assert_eq!(profile.nodes.len(), 1); + assert_eq!(profile.groups.len(), 2); + assert_eq!(profile.groups[0].role, GroupRole::System); + assert_eq!(profile.groups[0].group_id, 0); + assert_eq!(profile.groups[1].role, GroupRole::Data); + assert_eq!(profile.groups[1].group_id, 1); + assert_eq!(profile.disks.len(), 4); + assert!(profile + .disks + .iter() + .all(|disk| disk.capacity_bytes == 16 * 1024 * 1024 * 1024 + && disk.zone_size_bytes == disk.capacity_bytes)); + let endpoints = profile + .public_endpoints + .iter() + .map(|endpoint| (endpoint.id.as_str(), endpoint.port)) + .collect::>(); + assert_eq!( + endpoints, + BTreeMap::from([("iceberg", 80), ("s3", 81), ("web", 8080)]) + ); + let iceberg = profile + .services + .iter() + .find(|service| service.id == "iceberg") + .unwrap(); + assert_eq!( + iceberg.env.get("CROWDB_ICEBERG_PUBLIC_URI"), + Some(&"http://localhost".to_owned()) + ); + assert_eq!( + iceberg.env.get("CROWDB_ICEBERG_LISTEN"), + Some(&"0.0.0.0:80".to_owned()) + ); + assert_eq!(iceberg.probe.target, "http://127.0.0.1:80/v1/config"); + let s3 = profile + .services + .iter() + .find(|service| service.id == "s3") + .unwrap(); + assert_eq!(s3.env.get("CROWDB_S3_LISTEN"), Some(&"0.0.0.0:81".to_owned())); + assert_eq!(s3.probe.target, "http://127.0.0.1:81/_crowdb/health/ready"); + let web = profile + .services + .iter() + .find(|service| service.id == "web") + .unwrap(); + assert_eq!(web.probe.target, "http://127.0.0.1:8080/healthz"); + let template = std::fs::read_to_string( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/templates/crowdb-web.toml"), + ) + .unwrap(); + let config: toml::Value = toml::from_str(&template).unwrap(); + assert_eq!(config["port"].as_integer(), Some(8080)); +} + +#[test] +fn single_node_preview_declares_complete_dependency_order() { + let profile = DeploymentProfile::load(profile_path()).unwrap(); + let order = profile + .services_in_start_order() + .unwrap() + .into_iter() + .map(|service| service.id.as_str()) + .collect::>(); + assert_eq!( + order, + ["kv", "diskdb", "diskio", "chunkdb", "chunk-kv", "s3", "iceberg", "web"] + ); + for service in &profile.services { + if let Some(template) = &service.config_template { + let source = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../single-node-container/templates") + .join(template.file_name().unwrap()); + assert!(source.is_file(), "missing template for {}", service.id); + } + } + let chunkdb = std::fs::read_to_string( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/templates/chunkdb.toml"), + ) + .unwrap(); + let config: toml::Value = toml::from_str(&chunkdb).unwrap(); + assert_eq!(config["placement"]["mode"].as_str(), Some("unsafe_colocated")); +} diff --git a/container/crowdb-monitor/tests/status_test.rs b/container/crowdb-monitor/tests/status_test.rs new file mode 100644 index 000000000..f7cb42a0c --- /dev/null +++ b/container/crowdb-monitor/tests/status_test.rs @@ -0,0 +1,86 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs; +use std::path::{Path, PathBuf}; +use std::process::Command; +use std::time::Duration; + +use crowdb_monitor::{MonitorPhase, MonitorStatus, ServiceStatus, StatusStore}; +use uuid::Uuid; + +struct TestRunRoot(PathBuf); + +impl TestRunRoot { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-status-{}", Uuid::new_v4())); + fs::create_dir_all(&root).unwrap(); + Self(root.canonicalize().unwrap()) + } +} + +impl Drop for TestRunRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn command(name: &str, root: &TestRunRoot) -> bool { + Command::new(env!("CARGO_BIN_EXE_crowdb-monitor")) + .args([name, "--run-root"]) + .arg(&root.0) + .output() + .unwrap() + .status + .success() +} + +#[test] +fn readiness_requires_fresh_ready_snapshot() { + let root = TestRunRoot::new(); + assert!(!command("liveness", &root)); + assert!(!root.0.join("status").exists()); + let store = StatusStore::new(&root.0).unwrap(); + let mut status = MonitorStatus::new(Uuid::new_v4(), MonitorPhase::Initializing); + store.publish(&mut status).unwrap(); + assert!(!command("liveness", &root)); + assert!(!command("readiness", &root)); + status.phase = MonitorPhase::Ready; + status.services.insert( + "kv".into(), + ServiceStatus { + pid: Some(123), + generation: 1, + healthy: true, + restart_attempts: 0, + }, + ); + store.publish(&mut status).unwrap(); + assert_eq!(status.revision, 2); + assert!(command("readiness", &root)); + let path = root.0.join("status/monitor.json"); + let mut stale: serde_json::Value = serde_json::from_slice(&fs::read(&path).unwrap()).unwrap(); + stale["updated_at_ms"] = 0.into(); + fs::write(&path, serde_json::to_vec(&stale).unwrap()).unwrap(); + assert!(store.read(Duration::from_secs(10)).is_err()); + status.phase = MonitorPhase::Restarting; + store.publish(&mut status).unwrap(); + assert!(!command("readiness", &root)); +} + +#[test] +fn corrupt_or_symlinked_status_fails_closed() { + let root = TestRunRoot::new(); + let store = StatusStore::new(&root.0).unwrap(); + let mut status = MonitorStatus::new(Uuid::new_v4(), MonitorPhase::Ready); + store.publish(&mut status).unwrap(); + fs::write(root.0.join("status/monitor.json"), b"not json").unwrap(); + assert!(!command("readiness", &root)); + fs::remove_file(root.0.join("status/monitor.json")).unwrap(); + std::os::unix::fs::symlink("/etc/passwd", root.0.join("status/monitor.json")).unwrap(); + assert!(!command("readiness", &root)); +} diff --git a/container/crowdb-monitor/tests/storage_bootstrap_test.rs b/container/crowdb-monitor/tests/storage_bootstrap_test.rs new file mode 100644 index 000000000..3b57a9b0d --- /dev/null +++ b/container/crowdb-monitor/tests/storage_bootstrap_test.rs @@ -0,0 +1,466 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::BTreeMap; +use std::fs; +use std::os::unix::fs::symlink; +use std::path::{Path, PathBuf}; + +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, CrowdbSysmdClient}; +use crowdb_monitor::{ + disk_step_names, ensure_disk_files, hardware_step_names, iceberg_step_names, kv_step_names, + logical_step_names, render_configs, verify_chunk_services, verify_diskio_disks, BootstrapSession, + DeploymentProfile, HardwareBootstrap, IcebergBootstrap, KvBootstrap, LogicalBootstrap, ServerCredentials, + Supervisor, +}; +use uuid::Uuid; + +struct TestRoot(PathBuf); + +impl TestRoot { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-storage-{}", Uuid::new_v4())); + for path in ["bin", "templates", "data", "run"] { + fs::create_dir_all(root.join(path)).unwrap(); + } + Self(root.canonicalize().unwrap()) + } + + fn profile(&self, ports: &Ports, binaries: &[(&str, &Path)]) -> DeploymentProfile { + let mut profile = DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap(); + profile.paths.install_root.clone_from(&self.0); + profile.paths.bin_root = self.0.join("bin"); + profile.paths.template_root = self.0.join("templates"); + profile.paths.data_root = self.0.join("data"); + profile.paths.run_root = self.0.join("run"); + profile.paths.log_root = self.0.join("data/log"); + for disk in &mut profile.disks { + disk.path = self.0.join("data/disks").join(disk.path.file_name().unwrap()); + } + for group in &mut profile.groups { + group.rpc_endpoint = format!("127.0.0.1:{}", ports.kv_rpc); + } + profile + .services + .retain(|service| binaries.iter().any(|(id, _)| *id == service.id)); + for service in &mut profile.services { + let name = service.program.file_name().unwrap(); + let binary = binaries.iter().find(|(id, _)| *id == service.id).unwrap().1; + service.program = self.0.join("bin").join(name); + symlink(binary, &service.program).unwrap(); + service.config_template = service + .config_template + .as_ref() + .map(|path| self.0.join("templates").join(path.file_name().unwrap())); + service.args = match service.id.as_str() { + "kv" => vec![ + "--root".into(), + self.0.join("data/kv/node-1").to_string_lossy().into_owned(), + "--config".into(), + self.0.join("run/config/kv.toml").to_string_lossy().into_owned(), + "--management-addr".into(), + "127.0.0.1".into(), + "--management-port".into(), + ports.kv_management.to_string(), + "--ports".into(), + ports.kv_rpc.to_string(), + "--binding-monitor-interval".into(), + "1".into(), + ], + "diskdb" | "diskio" | "chunkdb" | "chunk-kv" => vec![ + "--config".into(), + self.0 + .join(format!("run/config/{}.toml", service.id)) + .to_string_lossy() + .into_owned(), + "--log-dir".into(), + self.0 + .join(format!("data/log/{}", service.id)) + .to_string_lossy() + .into_owned(), + ], + "iceberg" => vec!["serve".into()], + _ => unreachable!(), + }; + if service.id == "iceberg" { + service.env.insert( + "CROWDB_MANAGEMENT_SEEDS".into(), + format!("http://127.0.0.1:{}", ports.kv_management), + ); + service.env.insert( + "CROWDB_ICEBERG_LISTEN".into(), + format!("127.0.0.1:{}", ports.iceberg), + ); + service.env.insert( + "CROWDB_ICEBERG_PUBLIC_URI".into(), + format!("http://127.0.0.1:{}", ports.iceberg), + ); + } + service.fence_listeners = match service.id.as_str() { + "kv" => vec![ports.kv_management, ports.kv_rpc], + "diskdb" => vec![ports.diskdb_listen, ports.diskdb_http, ports.diskdb_rpc], + "diskio" => vec![ports.diskio_rpc], + "chunkdb" => vec![ports.chunkdb_http, ports.chunkdb_rpc], + "chunk-kv" => vec![ports.chunk_kv_http, ports.chunk_kv_rpc], + "iceberg" => vec![ports.iceberg], + _ => unreachable!(), + } + .into_iter() + .map(|port| format!("127.0.0.1:{port}")) + .collect(); + service.probe.target = match service.id.as_str() { + "kv" => format!("http://127.0.0.1:{}/health", ports.kv_management), + "diskdb" => format!("http://127.0.0.1:{}/ready", ports.diskdb_http), + "diskio" => format!("127.0.0.1:{}", ports.diskio_rpc), + "chunkdb" => format!("http://127.0.0.1:{}/ready", ports.chunkdb_http), + "chunk-kv" => format!("http://127.0.0.1:{}/ready", ports.chunk_kv_http), + "iceberg" => format!("http://127.0.0.1:{}/v1/config", ports.iceberg), + _ => unreachable!(), + }; + } + profile.validate().unwrap(); + profile + } + + fn templates(&self, ports: &Ports) { + let source = Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/templates"); + for name in [ + "kv.toml", + "diskdb.toml", + "diskio.toml", + "chunkdb.toml", + "chunk-kv.toml", + ] { + let body = fs::read_to_string(source.join(name)).unwrap(); + let body = body + .replace("127.0.0.1:10000", &format!("127.0.0.1:{}", ports.kv_management)) + .replace("127.0.0.1:11000", &format!("127.0.0.1:{}", ports.diskdb_listen)) + .replace("127.0.0.1:11100", &format!("127.0.0.1:{}", ports.diskdb_http)) + .replace("127.0.0.1:11200", &format!("127.0.0.1:{}", ports.diskdb_rpc)) + .replace( + "listen_port = 13000", + &format!("listen_port = {}", ports.diskio_rpc), + ) + .replace("127.0.0.1:12100", &format!("127.0.0.1:{}", ports.chunkdb_http)) + .replace("127.0.0.1:12200", &format!("127.0.0.1:{}", ports.chunkdb_rpc)) + .replace("127.0.0.1:15100", &format!("127.0.0.1:{}", ports.chunk_kv_http)) + .replace("127.0.0.1:15200", &format!("127.0.0.1:{}", ports.chunk_kv_rpc)); + fs::write(self.0.join("templates").join(name), body).unwrap(); + } + } + + fn session(&self, profile: &DeploymentProfile) -> BootstrapSession { + let names = kv_step_names(profile) + .unwrap() + .into_iter() + .chain(disk_step_names(profile)) + .chain(hardware_step_names()) + .chain(logical_step_names().map(str::to_owned)) + .chain( + profile + .services + .iter() + .any(|service| service.id == "iceberg") + .then(iceberg_step_names) + .into_iter() + .flatten() + .map(str::to_owned), + ) + .collect::>(); + let steps = names.iter().map(String::as_str).collect::>(); + BootstrapSession::open(&self.0.join("data"), b"profile", b"config", &steps).unwrap() + } +} + +impl Drop for TestRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +struct Ports { + kv_management: u16, + kv_rpc: u16, + diskdb_listen: u16, + diskdb_http: u16, + diskdb_rpc: u16, + diskio_rpc: u16, + chunkdb_http: u16, + chunkdb_rpc: u16, + chunk_kv_http: u16, + chunk_kv_rpc: u16, + iceberg: u16, +} + +impl Ports { + async fn allocate() -> Self { + let mut listeners = Vec::new(); + for _ in 0..11 { + listeners.push(tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap()); + } + let ports = listeners + .iter() + .map(|listener| listener.local_addr().unwrap().port()) + .collect::>(); + Self { + kv_management: ports[0], + kv_rpc: ports[1], + diskdb_listen: ports[2], + diskdb_http: ports[3], + diskdb_rpc: ports[4], + diskio_rpc: ports[5], + chunkdb_http: ports[6], + chunkdb_rpc: ports[7], + chunk_kv_http: ports[8], + chunk_kv_rpc: ports[9], + iceberg: ports[10], + } + } +} + +#[tokio::test] +async fn preview_chunk_services_start_and_recover() { + let Some(kv_binary) = crowdb_test_harness::cluster::crowdb_kv_server_bin() else { + eprintln!("skipping storage process test: KV binary unavailable"); + return; + }; + let diskdb_binary = Path::new(env!("CARGO_MANIFEST_DIR")).join("../../target/debug/crowdb-diskdb"); + let diskio_binary = + Path::new(env!("CARGO_MANIFEST_DIR")).join("../../app/crowdb-diskio/build/crowdb-diskio"); + let chunkdb_binary = Path::new(env!("CARGO_MANIFEST_DIR")).join("../../target/debug/crowdb-chunkdb"); + let chunk_kv_binary = + Path::new(env!("CARGO_MANIFEST_DIR")).join("../../target/debug/crowdb-chunk-kv-server"); + if !diskdb_binary.exists() + || !diskio_binary.exists() + || !chunkdb_binary.exists() + || !chunk_kv_binary.exists() + { + eprintln!("skipping storage process test: storage or chunk binary unavailable"); + return; + } + let root = TestRoot::new(); + let ports = Ports::allocate().await; + root.templates(&ports); + let profile = root.profile( + &ports, + &[ + ("kv", kv_binary.as_path()), + ("diskdb", diskdb_binary.as_path()), + ("diskio", diskio_binary.as_path()), + ("chunkdb", chunkdb_binary.as_path()), + ("chunk-kv", chunk_kv_binary.as_path()), + ], + ); + let mut session = root.session(&profile); + fs::create_dir_all(root.0.join("data/kv/node-1")).unwrap(); + fs::create_dir_all(root.0.join("data/log")).unwrap(); + render_configs(&profile, &root.0.join("templates"), &root.0.join("run")).unwrap(); + let management_seed = format!("http://127.0.0.1:{}", ports.kv_management); + let mut supervisor = Supervisor::new( + profile.clone(), + session.manifest().deployment_id(), + &root.0.join("data/log"), + &root.0.join("run"), + ) + .await + .unwrap(); + start_preview_storage(&mut supervisor, &mut session, &profile, &management_seed).await; + let chunkdb_config = root.0.join("run/config/chunkdb.toml"); + let original = fs::read_to_string(&chunkdb_config).unwrap(); + fs::write( + &chunkdb_config, + original.replace("instance_id = \"1\"", "instance_id = \"2\""), + ) + .unwrap(); + assert!(verify_chunk_services(&management_seed, &profile) + .await + .unwrap_err() + .to_string() + .contains("ChunkDB registration conflicts")); + fs::write(&chunkdb_config, original).unwrap(); + let chunk_kv_config = root.0.join("run/config/chunk-kv.toml"); + let original = fs::read_to_string(&chunk_kv_config).unwrap(); + fs::write( + &chunk_kv_config, + original.replace("instance_id = 1", "instance_id = 2"), + ) + .unwrap(); + assert!(verify_chunk_services(&management_seed, &profile) + .await + .unwrap_err() + .to_string() + .contains("Chunk-KV registration conflicts")); + fs::write(&chunk_kv_config, original).unwrap(); + verify_chunk_services(&management_seed, &profile).await.unwrap(); + assert_logical_topology_and_conflict(&management_seed, &profile, &mut session, &mut supervisor).await; + session.mark_ready().unwrap(); + supervisor.mark_ready().await.unwrap(); + supervisor.shutdown().await.unwrap(); + + let mut restarted_session = root.session(&profile); + let mut restarted = Supervisor::new( + profile.clone(), + restarted_session.manifest().deployment_id(), + &root.0.join("data/log"), + &root.0.join("run"), + ) + .await + .unwrap(); + start_preview_storage(&mut restarted, &mut restarted_session, &profile, &management_seed).await; + restarted.mark_ready().await.unwrap(); + restarted.shutdown().await.unwrap(); +} + +async fn assert_logical_topology_and_conflict( + management_seed: &str, + profile: &DeploymentProfile, + session: &mut BootstrapSession, + supervisor: &mut Supervisor, +) { + let sysmd = CrowdbSysmdClient::new(CrowdbKvClient::new(ClientConfig::new(vec![ + management_seed.to_owned() + ]))); + sysmd.kv().refresh_topology().await.unwrap(); + assert_eq!(sysmd.list_stores().await.unwrap().len(), 1); + assert_eq!(sysmd.list_groups_in_store(0).await.unwrap().len(), 2); + assert_eq!(sysmd.list_replicas_in_group(0, 0).await.unwrap().len(), 1); + assert_eq!(sysmd.list_replicas_in_group(0, 1).await.unwrap().len(), 1); + sysmd.add_group(0, 99).await.unwrap(); + assert!(LogicalBootstrap::new(management_seed.to_owned()) + .reconcile(session, profile, supervisor.monitor_log_mut()) + .await + .is_err()); + assert_eq!(sysmd.list_groups_in_store(0).await.unwrap().len(), 3); + sysmd.remove_group(0, 99).await.unwrap(); +} + +async fn start_preview_storage( + supervisor: &mut Supervisor, + session: &mut BootstrapSession, + profile: &DeploymentProfile, + management_seed: &str, +) { + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + KvBootstrap::new(management_seed) + .unwrap() + .reconcile(session, profile, supervisor.monitor_log_mut()) + .await + .unwrap(); + ensure_disk_files(session, profile, supervisor.monitor_log_mut()) + .await + .unwrap(); + HardwareBootstrap::new(management_seed.to_owned()) + .reconcile(session, profile, supervisor.monitor_log_mut()) + .await + .unwrap(); + LogicalBootstrap::new(management_seed.to_owned()) + .reconcile(session, profile, supervisor.monitor_log_mut()) + .await + .unwrap(); + supervisor.start_service("diskdb", BTreeMap::new()).await.unwrap(); + supervisor.start_service("diskio", BTreeMap::new()).await.unwrap(); + verify_diskio_disks(management_seed, profile).await.unwrap(); + supervisor + .start_service("chunkdb", BTreeMap::new()) + .await + .unwrap(); + supervisor + .start_service("chunk-kv", BTreeMap::new()) + .await + .unwrap(); + verify_chunk_services(management_seed, profile).await.unwrap(); +} + +#[tokio::test] +async fn preview_real_iceberg_catalog_and_listener_survive_restart() { + let Some(kv_binary) = crowdb_test_harness::cluster::crowdb_kv_server_bin() else { + eprintln!("skipping real Iceberg bootstrap: KV binary unavailable"); + return; + }; + let binary_root = Path::new(env!("CARGO_MANIFEST_DIR")).join("../../target/debug"); + let diskio_binary = + Path::new(env!("CARGO_MANIFEST_DIR")).join("../../app/crowdb-diskio/build/crowdb-diskio"); + let binaries = [ + ("diskdb", binary_root.join("crowdb-diskdb")), + ("diskio", diskio_binary), + ("chunkdb", binary_root.join("crowdb-chunkdb")), + ("chunk-kv", binary_root.join("crowdb-chunk-kv-server")), + ("iceberg", binary_root.join("crowdb-iceberg")), + ]; + if binaries.iter().any(|(_, binary)| !binary.exists()) { + eprintln!("skipping real Iceberg bootstrap: storage or Iceberg binary unavailable"); + return; + } + let root = TestRoot::new(); + let ports = Ports::allocate().await; + root.templates(&ports); + let mut links = vec![("kv", kv_binary.as_path())]; + links.extend(binaries.iter().map(|(id, path)| (*id, path.as_path()))); + let profile = root.profile(&ports, &links); + let mut session = root.session(&profile); + let credentials = ServerCredentials::load_or_create(&root.0.join("data")).unwrap(); + fs::create_dir_all(root.0.join("data/kv/node-1")).unwrap(); + fs::create_dir_all(root.0.join("data/log")).unwrap(); + render_configs(&profile, &root.0.join("templates"), &root.0.join("run")).unwrap(); + let seed = format!("http://127.0.0.1:{}", ports.kv_management); + let mut supervisor = Supervisor::new( + profile.clone(), + session.manifest().deployment_id(), + &root.0.join("data/log"), + &root.0.join("run"), + ) + .await + .unwrap(); + start_preview_storage(&mut supervisor, &mut session, &profile, &seed).await; + IcebergBootstrap::reconcile(&mut session, &profile, &credentials, supervisor.monitor_log_mut()) + .await + .unwrap(); + let environment = iceberg_environment(&credentials); + supervisor + .start_service("iceberg", environment.clone()) + .await + .unwrap(); + session.mark_ready().unwrap(); + supervisor.mark_ready().await.unwrap(); + supervisor.shutdown().await.unwrap(); + drop(supervisor); + + let mut restarted_session = root.session(&profile); + let mut restarted = Supervisor::new( + profile.clone(), + restarted_session.manifest().deployment_id(), + &root.0.join("data/log"), + &root.0.join("run"), + ) + .await + .unwrap(); + start_preview_storage(&mut restarted, &mut restarted_session, &profile, &seed).await; + IcebergBootstrap::reconcile( + &mut restarted_session, + &profile, + &credentials, + restarted.monitor_log_mut(), + ) + .await + .unwrap(); + restarted.start_service("iceberg", environment).await.unwrap(); + restarted.mark_ready().await.unwrap(); + restarted.shutdown().await.unwrap(); +} + +fn iceberg_environment(credentials: &ServerCredentials) -> BTreeMap { + credentials + .server_env() + .lines() + .filter_map(|line| line.split_once('=')) + .filter(|(key, _)| key.starts_with("CROWDB_ICEBERG_")) + .map(|(key, value)| (key.to_owned(), value.to_owned())) + .collect() +} diff --git a/container/crowdb-monitor/tests/supervisor_test.rs b/container/crowdb-monitor/tests/supervisor_test.rs new file mode 100644 index 000000000..4e358b369 --- /dev/null +++ b/container/crowdb-monitor/tests/supervisor_test.rs @@ -0,0 +1,356 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::BTreeMap; +use std::fs; +use std::os::unix::fs::symlink; +use std::path::{Path, PathBuf}; +use std::time::Duration; + +use crowdb_monitor::{DeploymentProfile, MonitorPhase, ProbeKind, Supervisor}; +use tokio::net::TcpListener; +use uuid::Uuid; + +struct TestRoots(PathBuf); + +impl TestRoots { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-supervisor-{}", Uuid::new_v4())); + for directory in ["bin", "templates", "data/log", "data/disks", "run"] { + fs::create_dir_all(root.join(directory)).unwrap(); + } + symlink("/bin/sh", root.join("bin/sh")).unwrap(); + Self(root.canonicalize().unwrap()) + } + + fn profile(&self, script: String, port: u16, max_attempts: u32) -> DeploymentProfile { + let mut profile = DeploymentProfile::load( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/profile.toml"), + ) + .unwrap(); + profile.paths.install_root.clone_from(&self.0); + profile.paths.bin_root = self.0.join("bin"); + profile.paths.template_root = self.0.join("templates"); + profile.paths.data_root = self.0.join("data"); + profile.paths.run_root = self.0.join("run"); + profile.paths.log_root = self.0.join("data/log"); + for disk in &mut profile.disks { + disk.path = self.0.join("data/disks").join(disk.path.file_name().unwrap()); + } + profile.services.retain(|service| service.id == "kv"); + let service = &mut profile.services[0]; + service.program = self.0.join("bin/sh"); + service.args = vec!["-c".into(), script]; + service.config_template = None; + service.fence_listeners.clear(); + service.probe.kind = ProbeKind::Tcp; + service.probe.target = format!("127.0.0.1:{port}"); + service.probe.failure_threshold = 1; + service.restart.max_attempts = max_attempts; + service.restart.backoff_base_ms = 10; + service.restart.backoff_max_ms = 20; + profile.logs.mirror_warnings_to_stderr = false; + profile.validate().unwrap(); + profile + } +} + +impl Drop for TestRoots { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +#[tokio::test] +async fn exited_service_restarts_with_same_identity_and_event_log() { + let roots = TestRoots::new(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let marker = roots.0.join("first-exit"); + let script = format!( + "if [ ! -e '{}' ]; then : > '{}'; sleep 0.2; exit 0; fi; exec sleep 30", + marker.display(), + marker.display() + ); + let profile = roots.profile(script, listener.local_addr().unwrap().port(), 2); + let identity = Uuid::new_v4(); + let mut supervisor = Supervisor::new(profile, identity, &roots.0.join("data/log"), &roots.0.join("run")) + .await + .unwrap(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + supervisor.mark_ready().await.unwrap(); + tokio::time::sleep(Duration::from_millis(350)).await; + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().deployment_id, identity); + assert_eq!(supervisor.status().phase, MonitorPhase::Ready); + assert_eq!(supervisor.status().services["kv"].generation, 2); + supervisor.shutdown().await.unwrap(); + let body = fs::read_to_string(roots.0.join("data/log/monitor/monitor.log")).unwrap(); + for event in [ + "starting", + "child_exited", + "restarting", + "child_stopped", + "draining", + "stopped", + ] { + assert!(body.contains(&format!("\"kind\":\"{event}\"")), "missing {event}"); + } +} + +#[tokio::test] +async fn restarted_service_waits_for_authority_validation() { + let roots = TestRoots::new(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let marker = roots.0.join("validation-exit"); + let script = format!( + "if [ ! -e '{}' ]; then : > '{}'; sleep 0.2; exit 0; fi; exec sleep 30", + marker.display(), + marker.display() + ); + let profile = roots.profile(script, listener.local_addr().unwrap().port(), 2); + let mut supervisor = Supervisor::new( + profile, + Uuid::new_v4(), + &roots.0.join("data/log"), + &roots.0.join("run"), + ) + .await + .unwrap(); + supervisor.require_recovery_validation(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + supervisor.mark_ready().await.unwrap(); + tokio::time::sleep(Duration::from_millis(350)).await; + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().phase, MonitorPhase::Restarting); + assert!(supervisor.recovery_pending()); + assert_eq!(supervisor.recovery_epoch(), 1); + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().phase, MonitorPhase::Restarting); + supervisor.finish_recovery().await.unwrap(); + assert_eq!(supervisor.status().phase, MonitorPhase::Ready); + assert!(!supervisor.recovery_pending()); + supervisor.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn repeated_exits_exhaust_budget_and_leave_unready() { + let roots = TestRoots::new(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let profile = roots.profile( + "sleep 0.2; exit 1".into(), + listener.local_addr().unwrap().port(), + 1, + ); + let mut supervisor = Supervisor::new( + profile, + Uuid::new_v4(), + &roots.0.join("data/log"), + &roots.0.join("run"), + ) + .await + .unwrap(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + supervisor.mark_ready().await.unwrap(); + tokio::time::sleep(Duration::from_millis(350)).await; + supervisor.poll_once().await.unwrap(); + tokio::time::sleep(Duration::from_millis(350)).await; + assert!(supervisor.poll_once().await.is_err()); + assert_eq!(supervisor.status().phase, MonitorPhase::Failed); + assert!(supervisor.status().services["kv"].pid.is_none()); + let body = fs::read_to_string(roots.0.join("data/log/monitor/monitor.log")).unwrap(); + assert!(body.contains("restart_exhausted")); +} + +#[tokio::test] +async fn stable_health_resets_crash_loop_budget() { + let roots = TestRoots::new(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let mut profile = roots.profile( + "sleep 0.3; exit 1".into(), + listener.local_addr().unwrap().port(), + 1, + ); + profile.services[0].restart.stable_after_ms = 50; + let mut supervisor = Supervisor::new( + profile, + Uuid::new_v4(), + &roots.0.join("data/log"), + &roots.0.join("run"), + ) + .await + .unwrap(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + supervisor.mark_ready().await.unwrap(); + tokio::time::sleep(Duration::from_millis(350)).await; + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().services["kv"].restart_attempts, 1); + tokio::time::sleep(Duration::from_millis(80)).await; + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().services["kv"].restart_attempts, 0); + tokio::time::sleep(Duration::from_millis(300)).await; + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().services["kv"].generation, 3); + supervisor.shutdown().await.unwrap(); + let body = fs::read_to_string(roots.0.join("data/log/monitor/monitor.log")).unwrap(); + assert!(body.contains("restart_budget_reset")); +} + +#[tokio::test] +async fn live_listener_prevents_replacement_after_child_exit() { + let roots = TestRoots::new(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap().to_string(); + let mut profile = roots.profile( + "sleep 0.2; exit 1".into(), + listener.local_addr().unwrap().port(), + 1, + ); + profile.services[0].fence_listeners = vec![address]; + let mut supervisor = Supervisor::new( + profile, + Uuid::new_v4(), + &roots.0.join("data/log"), + &roots.0.join("run"), + ) + .await + .unwrap(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + supervisor.mark_ready().await.unwrap(); + tokio::time::sleep(Duration::from_millis(350)).await; + assert!(supervisor.poll_once().await.is_err()); + assert_eq!(supervisor.status().phase, MonitorPhase::Failed); + assert_eq!(supervisor.status().services["kv"].generation, 1); + assert_eq!(supervisor.status().services["kv"].pid, None); + let body = fs::read_to_string(roots.0.join("data/log/monitor/monitor.log")).unwrap(); + assert!(body.contains("listener_fence_failed")); +} + +#[tokio::test] +async fn transient_probe_failure_clears_readiness_without_restarting() { + let roots = TestRoots::new(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let mut profile = roots.profile("exec sleep 30".into(), address.port(), 2); + profile.services[0].probe.failure_threshold = 2; + let mut supervisor = Supervisor::new( + profile, + Uuid::new_v4(), + &roots.0.join("data/log"), + &roots.0.join("run"), + ) + .await + .unwrap(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + supervisor.mark_ready().await.unwrap(); + let original_pid = supervisor.status().services["kv"].pid; + drop(listener); + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().phase, MonitorPhase::Restarting); + assert!(!supervisor.status().services["kv"].healthy); + let listener = TcpListener::bind(address).await.unwrap(); + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().phase, MonitorPhase::Ready); + assert_eq!(supervisor.status().services["kv"].pid, original_pid); + assert_eq!(supervisor.status().services["kv"].generation, 1); + supervisor.shutdown().await.unwrap(); + drop(listener); +} + +#[tokio::test] +#[cfg(target_os = "linux")] +async fn child_exit_after_probe_failure_is_recorded() { + let roots = TestRoots::new(); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let mut profile = roots.profile("exec sleep 30".into(), address.port(), 2); + profile.services[0].probe.failure_threshold = 5; + let mut supervisor = Supervisor::new( + profile, + Uuid::new_v4(), + &roots.0.join("data/log"), + &roots.0.join("run"), + ) + .await + .unwrap(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + supervisor.mark_ready().await.unwrap(); + let pid = supervisor.status().services["kv"].pid.unwrap(); + drop(listener); + supervisor.poll_once().await.unwrap(); + let pid = rustix::process::Pid::from_raw(i32::try_from(pid).unwrap()).unwrap(); + rustix::process::kill_process(pid, rustix::process::Signal::KILL).unwrap(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(3); + // Wait for the killed process to enter zombie state, without reaping the + // supervisor's child or resetting its prior failed-probe observation. + loop { + let status = fs::read_to_string(format!("/proc/{}/status", pid.as_raw_pid())).unwrap(); + if status + .lines() + .any(|line| line.starts_with("State:") && line.contains('Z')) + { + break; + } + assert!(tokio::time::Instant::now() < deadline); + tokio::task::yield_now().await; + } + let _listener = TcpListener::bind(address).await.unwrap(); + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().services["kv"].generation, 2); + supervisor.shutdown().await.unwrap(); + let body = fs::read_to_string(roots.0.join("data/log/monitor/monitor.log")).unwrap(); + assert!(body.contains("\"kind\":\"probe_failed\"")); + assert!(body.contains("\"kind\":\"child_exited\"")); +} + +#[tokio::test] +async fn dependency_restart_stops_dependents_before_replacement() { + let roots = TestRoots::new(); + let root_listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let dependent_listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let marker = roots.0.join("first-exit"); + let script = format!( + "if [ ! -e '{}' ]; then : > '{}'; sleep 0.2; exit 0; fi; exec sleep 30", + marker.display(), + marker.display() + ); + let mut profile = roots.profile(script, root_listener.local_addr().unwrap().port(), 2); + let mut dependent = profile.services[0].clone(); + dependent.id = "web".into(); + dependent.dependencies = vec!["kv".into()]; + dependent.args = vec!["-c".into(), "exec sleep 30".into()]; + dependent.probe.target = dependent_listener.local_addr().unwrap().to_string(); + profile.services.push(dependent); + profile.validate().unwrap(); + let mut supervisor = Supervisor::new( + profile, + Uuid::new_v4(), + &roots.0.join("data/log"), + &roots.0.join("run"), + ) + .await + .unwrap(); + supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); + supervisor.start_service("web", BTreeMap::new()).await.unwrap(); + supervisor.mark_ready().await.unwrap(); + tokio::time::sleep(Duration::from_millis(350)).await; + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().phase, MonitorPhase::Ready); + assert_eq!(supervisor.status().services["kv"].generation, 2); + assert_eq!(supervisor.status().services["web"].generation, 2); + supervisor.shutdown().await.unwrap(); + let body = fs::read_to_string(roots.0.join("data/log/monitor/monitor.log")).unwrap(); + let events = body + .lines() + .map(|line| serde_json::from_str::(line).unwrap()) + .collect::>(); + let stops = events + .iter() + .filter(|event| event["kind"] == "child_stopped") + .map(|event| event["service"].as_str().unwrap()) + .collect::>(); + assert_eq!(&stops[..2], &["web", "kv"]); +} diff --git a/container/single-node-container/Dockerfile b/container/single-node-container/Dockerfile new file mode 100644 index 000000000..682b73374 --- /dev/null +++ b/container/single-node-container/Dockerfile @@ -0,0 +1,45 @@ +FROM ubuntu:24.04@sha256:496754492fb28b4d3049432f2ca787449331e23fb14f0dd3fffea86bf5a93eb4 +RUN apt-get update && apt-get install -y --no-install-recommends ca-certificates libcap2-bin && rm -rf /var/lib/apt/lists/* \ + && groupadd --system --gid 10001 crowdb \ + && useradd --system --uid 10001 --gid 10001 --home-dir /opt/crowdb --shell /usr/sbin/nologin crowdb +RUN --mount=type=bind,source=.,target=/staged,ro \ + mkdir -p /opt/crowdb \ + && cp -a /staged/bin /staged/lib /opt/crowdb/ \ + && setcap cap_net_bind_service=+ep /opt/crowdb/bin/crowdb-iceberg \ + && setcap cap_net_bind_service=+ep /opt/crowdb/bin/crowdb-access-server \ + && mkdir -p /opt/crowdb/data /opt/crowdb/run \ + && chown -R crowdb:crowdb /opt/crowdb/data /opt/crowdb/run +COPY ui/ /opt/crowdb/ui/ +COPY profile.toml /opt/crowdb/etc/profile.toml +COPY templates/ /opt/crowdb/etc/templates/ +COPY entrypoint.sh /opt/crowdb/bin/entrypoint +RUN chmod 0755 /opt/crowdb/bin/entrypoint \ + && for binary in /opt/crowdb/bin/crowdb-*; do \ + LD_LIBRARY_PATH=/opt/crowdb/lib ldd "$binary" > /tmp/dependencies.txt 2>&1 || { cat /tmp/dependencies.txt; exit 1; }; \ + if grep -q 'not found' /tmp/dependencies.txt; then cat /tmp/dependencies.txt; exit 1; fi; \ + done && rm /tmp/dependencies.txt + +ARG SOURCE_REVISION +ARG PREVIEW_VERSION +RUN --mount=type=bind,source=.,target=/staged,ro \ + test -n "$SOURCE_REVISION" && test -n "$PREVIEW_VERSION" \ + && test "$(cat /staged/SOURCE_REVISION)" = "$SOURCE_REVISION" \ + && test "$(cat /staged/VERSION)" = "$PREVIEW_VERSION" +LABEL org.opencontainers.image.title="CROWDB Iceberg" \ + org.opencontainers.image.description="Apache Iceberg REST catalog with native storage, in one container for development and testing." \ + org.opencontainers.image.url="https://crowdb.dev/" \ + org.opencontainers.image.source="https://github.com/buzzcrow/crowdb" \ + org.opencontainers.image.documentation="https://github.com/buzzcrow/crowdb/blob/v${PREVIEW_VERSION}/doc/user-manual/docker-single-node-user-guide.md" \ + org.opencontainers.image.authors="Gian " \ + org.opencontainers.image.licenses="Apache-2.0" \ + org.opencontainers.image.revision="$SOURCE_REVISION" \ + org.opencontainers.image.version="$PREVIEW_VERSION" +ENV PATH="/opt/crowdb/bin:${PATH}" \ + LD_LIBRARY_PATH="/opt/crowdb/lib" \ + CROWDB_RUNTIME_ROOT="/opt/crowdb/run" +USER crowdb:crowdb +VOLUME ["/opt/crowdb/data"] +EXPOSE 80 81 8080 +STOPSIGNAL SIGTERM +HEALTHCHECK --interval=10s --timeout=5s --start-period=120s --retries=3 CMD crowdb-monitor liveness && crowdb-monitor readiness +ENTRYPOINT ["/opt/crowdb/bin/entrypoint"] diff --git a/container/single-node-container/Dockerfile.dockerignore b/container/single-node-container/Dockerfile.dockerignore new file mode 100644 index 000000000..73af00081 --- /dev/null +++ b/container/single-node-container/Dockerfile.dockerignore @@ -0,0 +1,17 @@ +.git +.github +.agents +doc +container/single-node-container/tests +.pixi +.crowdb-runtime +target +**/target +**/build +**/build-* +**/node_modules +**/dist +**/.cache +**/.env +**/*.log +**/secrets diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md new file mode 100644 index 000000000..facd8f9d2 --- /dev/null +++ b/container/single-node-container/README.md @@ -0,0 +1,31 @@ + + + +# Single-node container development + +For running CROWDB, see the [Docker user guide](../../doc/user-manual/docker-single-node-user-guide.md). +This page describes building from source on a Linux amd64 development or CI host. + +```sh +pixi run build-single-node-container +pixi run test-single-node-container +``` + +- Compilation runs on the host with the locked repository dependencies. Cargo + and CMake reuse existing build outputs; npm uses its local download cache. +- The build stages release programs, their required shared libraries, UI and + deployment files under `target/container-runtime`. Existing runtime data and + credentials are never part of the Docker context. +- Docker only packages these files into the pinned Ubuntu runtime image. It + does not install Pixi or compilers, compile source, or use a custom base image. +- Packaging checks dynamic linkage inside Ubuntu and verifies the staged + revision/version against image metadata. A different host ABI must pass these + checks and the container tests before its artifacts can be used. +- The default local image is `crowdb-iceberg-single-node:dev`. Set + `CROWDB_CONTAINER_IMAGE` to build and test a separate candidate tag. + +`pixi run stage-single-node-container` produces the runtime directory without +building a Docker image. The release workflow archives the verified directory +and packages those same files in its publish job, without recompiling them. +Docker Hub publication is manual; actual publication verification is deferred +until administrator preparation is complete. diff --git a/container/single-node-container/build.sh b/container/single-node-container/build.sh new file mode 100644 index 000000000..663ad44cb --- /dev/null +++ b/container/single-node-container/build.sh @@ -0,0 +1,55 @@ +#!/bin/bash +# Copyright 2026-present Gian +# Licensed under the Apache License, Version 2.0. +set -euo pipefail + +mode=${1:-image} +[[ "$mode" == stage || "$mode" == image ]] || { echo 'Expected stage or image' >&2; exit 1; } +[[ $(uname -sm) == 'Linux x86_64' ]] || { echo 'Container artifacts require a Linux amd64 build host' >&2; exit 1; } +cd "$(git rev-parse --show-toplevel)" +for tool in patchelf strip ldd; do + command -v "$tool" >/dev/null || { echo "Missing packaging tool: $tool" >&2; exit 1; } +done +if [[ "$mode" == image ]]; then + docker info >/dev/null +fi + +# Build on the host, reusing the existing Cargo, CMake and npm artifacts. +cargo build --locked --release -p crowdb-kv-client --features ffi +cmake -S app/crowdb-diskio -B app/crowdb-diskio/build -DCMAKE_BUILD_TYPE=Release +cmake --build app/crowdb-diskio/build -j 4 --target crowdb-diskio +cargo build --locked --release \ + -p crowdb-monitor -p crowdb-kv-server -p crowdb-diskdb \ + -p crowdb-chunkdb -p crowdb-chunk-kv-server \ + -p crowdb-access-server -p crowdb-web +(cd app/crowdb-web/ui && npm ci --prefer-offline && npm run build) + +# Docker receives only the assembled runtime, never the source tree or data. +staging=$(mktemp -d "$PWD/target/container-runtime.XXXXXX") +trap 'rm -rf "$staging"' EXIT +bash container/single-node-container/collect-libs.sh "$staging" +cp -a app/crowdb-web/ui/dist "$staging/ui" +cp -a container/single-node-container/templates "$staging/templates" +cp container/single-node-container/{Dockerfile,profile.toml,entrypoint.sh} "$staging/" +git rev-parse HEAD > "$staging/SOURCE_REVISION" +cp VERSION "$staging/VERSION" +rm -rf target/container-runtime +mv "$staging" target/container-runtime +trap - EXIT +[[ "$mode" == image ]] || exit 0 + +proxy_args=() +for name in http_proxy https_proxy all_proxy no_proxy; do + value="${!name:-}" + if [[ "$value" == *'@'* ]]; then + echo 'Credential-bearing proxy settings are not accepted by the image build' >&2 + exit 1 + fi + if [[ "$name" != no_proxy && -n "$value" && "$value" != *://* ]]; then value="http://$value"; fi + if [[ -n "$value" ]]; then proxy_args+=(--build-arg "$name=$value"); fi +done +DOCKER_BUILDKIT=1 docker build --platform linux/amd64 \ + "${proxy_args[@]}" \ + --build-arg SOURCE_REVISION="$(cat target/container-runtime/SOURCE_REVISION)" \ + --build-arg PREVIEW_VERSION="$(cat VERSION)" \ + --tag "${CROWDB_CONTAINER_IMAGE:-crowdb-iceberg-single-node:dev}" target/container-runtime diff --git a/container/single-node-container/collect-libs.sh b/container/single-node-container/collect-libs.sh new file mode 100644 index 000000000..4276ad68a --- /dev/null +++ b/container/single-node-container/collect-libs.sh @@ -0,0 +1,70 @@ +#!/bin/bash +set -euo pipefail + +build_root=$(git rev-parse --show-toplevel) +output=${1:?runtime staging directory is required} +mkdir -p "$output/bin" "$output/lib" + +for binary in \ + crowdb-monitor crowdb-kv-server crowdb-diskdb crowdb-diskio \ + crowdb-chunkdb crowdb-chunk-kv-server crowdb-access-server \ + crowdb-iceberg crowdb-web; do + if [[ "$binary" == crowdb-diskio ]]; then + source="$build_root/app/crowdb-diskio/build/crowdb-diskio" + else + source="$build_root/target/release/$binary" + fi + if [[ ! -x "$source" ]]; then + echo "preview binary is missing: $source" >&2 + exit 1 + fi + echo "packing $binary" + cp "$source" "$output/bin/$binary" +done + +for binary in "$output"/bin/*; do + if ! LD_LIBRARY_PATH="$build_root/.pixi/envs/default/lib:$build_root/target/release" ldd "$binary" > "$output/dependencies.txt"; then + cat "$output/dependencies.txt" >&2 + exit 1 + fi + if grep -q 'not found' "$output/dependencies.txt"; then + cat "$output/dependencies.txt" >&2 + exit 1 + fi + while read -r name arrow path remainder; do + if [[ "$arrow" != '=>' ]]; then + continue + fi + resolved_path=$(readlink -f "$path") + case "$resolved_path" in + "$build_root/.pixi/envs/default/lib/"*|"$build_root/target/release/"*) + if ! cp -L "$resolved_path" "$output/lib/$name"; then + echo "cannot package dependency $name from $resolved_path" >&2 + exit 1 + fi + ;; + esac + done < "$output/dependencies.txt" + patchelf --set-rpath '/opt/crowdb/lib' "$binary" +done +if [[ ! -f "$output/lib/libcrowdb_kv_client.so" ]]; then + echo 'DiskIO FFI library was not collected' >&2 + exit 1 +fi +for library in "$output"/lib/*; do + patchelf --set-rpath '/opt/crowdb/lib' "$library" +done +rm "$output/dependencies.txt" + +for artifact in "$output"/bin/* "$output"/lib/*; do + strip --strip-debug "$artifact" +done + +for binary in "$output"/bin/*; do + LD_LIBRARY_PATH="$output/lib" ldd "$binary" > "$output/dependencies.txt" + if grep -q 'not found' "$output/dependencies.txt"; then + cat "$output/dependencies.txt" >&2 + exit 1 + fi +done +rm "$output/dependencies.txt" diff --git a/container/single-node-container/entrypoint.sh b/container/single-node-container/entrypoint.sh new file mode 100644 index 000000000..220c5c483 --- /dev/null +++ b/container/single-node-container/entrypoint.sh @@ -0,0 +1,11 @@ +#!/bin/sh +set -eu + +if ! grep -q ' /opt/crowdb/data ' /proc/self/mountinfo; then + echo 'CROWDB preview requires one volume mounted at /opt/crowdb/data' >&2 + exit 1 +fi + +echo 'CROWDB preview data: Docker creates an anonymous volume when none is specified. For data you want to keep across container recreation, use --mount type=volume,source=crowdb-data,target=/opt/crowdb/data.' + +exec /opt/crowdb/bin/crowdb-monitor run --profile /opt/crowdb/etc/profile.toml diff --git a/container/single-node-container/profile.toml b/container/single-node-container/profile.toml new file mode 100644 index 000000000..92ebcda5e --- /dev/null +++ b/container/single-node-container/profile.toml @@ -0,0 +1,223 @@ +version = 1 +name = "single-node-container" +display_name = "CROWDB Single-Node Container" +placement_mode = "unsafe-colocated" +s3_tenant = "preview" +iceberg_catalog = "preview" + +[paths] +install_root = "/opt/crowdb" +bin_root = "/opt/crowdb/bin" +template_root = "/opt/crowdb/etc/templates" +data_root = "/opt/crowdb/data" +run_root = "/opt/crowdb/run" +log_root = "/opt/crowdb/data/log" + +[logs] +max_file_bytes = 31457280 +max_files = 5 +mirror_warnings_to_stderr = true + +[[nodes]] +node_id = 1 +rack_id = 1 + +[[groups]] +store_id = 0 +group_id = 0 +replica_id = 1 +node_id = 1 +rpc_endpoint = "127.0.0.1:10100" +role = "system" + +[[groups]] +store_id = 0 +group_id = 1 +replica_id = 2 +node_id = 1 +rpc_endpoint = "127.0.0.1:10100" +role = "data" + +[[disks]] +disk_id = "00000000000000010000000000000001" +disk_group_id = 101 +node_id = 1 +path = "/opt/crowdb/data/disks/disk-0001.img" +capacity_bytes = 17179869184 +zone_size_bytes = 17179869184 + +[[disks]] +disk_id = "00000000000000010000000000000002" +disk_group_id = 101 +node_id = 1 +path = "/opt/crowdb/data/disks/disk-0002.img" +capacity_bytes = 17179869184 +zone_size_bytes = 17179869184 + +[[disks]] +disk_id = "00000000000000010000000000000003" +disk_group_id = 101 +node_id = 1 +path = "/opt/crowdb/data/disks/disk-0003.img" +capacity_bytes = 17179869184 +zone_size_bytes = 17179869184 + +[[disks]] +disk_id = "00000000000000010000000000000004" +disk_group_id = 101 +node_id = 1 +path = "/opt/crowdb/data/disks/disk-0004.img" +capacity_bytes = 17179869184 +zone_size_bytes = 17179869184 + +[[public_endpoints]] +id = "s3" +bind = "0.0.0.0" +port = 81 + +[[public_endpoints]] +id = "iceberg" +bind = "0.0.0.0" +port = 80 + +[[public_endpoints]] +id = "web" +bind = "0.0.0.0" +port = 8080 + +[[services]] +id = "kv" +program = "/opt/crowdb/bin/crowdb-kv-server" +args = ["--root", "/opt/crowdb/data/kv/node-1", "--config", "/opt/crowdb/run/config/kv.toml", "--management-port", "10000", "--ports", "10100,10101", "--instance-id", "1", "--node-id", "1", "--binding-monitor-interval", "1", "--log-dir", "/opt/crowdb/data/log/kv", "--log"] +dependencies = [] +fence_listeners = ["127.0.0.1:10000", "127.0.0.1:10100", "127.0.0.1:10101"] +config_template = "/opt/crowdb/etc/templates/kv.toml" +[services.probe] +kind = "http" +target = "http://127.0.0.1:10000/health" +timeout_ms = 1000 +failure_threshold = 5 +[services.restart] +max_attempts = 5 +backoff_base_ms = 250 +backoff_max_ms = 5000 + +[[services]] +id = "diskdb" +program = "/opt/crowdb/bin/crowdb-diskdb" +args = ["--config", "/opt/crowdb/run/config/diskdb.toml", "--log-dir", "/opt/crowdb/data/log/diskdb", "--log"] +dependencies = ["kv"] +fence_listeners = ["127.0.0.1:11000", "127.0.0.1:11100", "127.0.0.1:11200"] +config_template = "/opt/crowdb/etc/templates/diskdb.toml" +[services.probe] +kind = "http" +target = "http://127.0.0.1:11100/health" +timeout_ms = 1000 +failure_threshold = 5 +[services.restart] +max_attempts = 5 +backoff_base_ms = 250 +backoff_max_ms = 5000 + +[[services]] +id = "diskio" +program = "/opt/crowdb/bin/crowdb-diskio" +args = ["--config", "/opt/crowdb/run/config/diskio.toml"] +dependencies = ["kv", "diskdb"] +fence_listeners = ["127.0.0.1:13000"] +config_template = "/opt/crowdb/etc/templates/diskio.toml" +[services.probe] +kind = "rpc-ping" +target = "127.0.0.1:13000" +timeout_ms = 1000 +failure_threshold = 5 +[services.restart] +max_attempts = 5 +backoff_base_ms = 250 +backoff_max_ms = 5000 + +[[services]] +id = "chunkdb" +program = "/opt/crowdb/bin/crowdb-chunkdb" +args = ["--config", "/opt/crowdb/run/config/chunkdb.toml", "--log-dir", "/opt/crowdb/data/log/chunkdb", "--log"] +dependencies = ["kv", "diskdb", "diskio"] +fence_listeners = ["127.0.0.1:12100", "127.0.0.1:12200"] +config_template = "/opt/crowdb/etc/templates/chunkdb.toml" +[services.probe] +kind = "http" +target = "http://127.0.0.1:12100/ready" +timeout_ms = 1000 +failure_threshold = 5 +[services.restart] +max_attempts = 5 +backoff_base_ms = 250 +backoff_max_ms = 5000 + +[[services]] +id = "chunk-kv" +program = "/opt/crowdb/bin/crowdb-chunk-kv-server" +args = ["--config", "/opt/crowdb/run/config/chunk-kv.toml", "--log-dir", "/opt/crowdb/data/log/chunk-kv", "--log"] +dependencies = ["kv", "chunkdb", "diskio"] +fence_listeners = ["127.0.0.1:15100", "127.0.0.1:15200"] +config_template = "/opt/crowdb/etc/templates/chunk-kv.toml" +[services.probe] +kind = "http" +target = "http://127.0.0.1:15100/ready" +timeout_ms = 1000 +failure_threshold = 5 +[services.restart] +max_attempts = 5 +backoff_base_ms = 250 +backoff_max_ms = 5000 + +[[services]] +id = "s3" +program = "/opt/crowdb/bin/crowdb-access-server" +args = [] +env = { CROWDB_S3_LISTEN = "0.0.0.0:81", CROWDB_S3_PUBLIC_URI = "http://localhost:81", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000", CROWDB_S3_TENANT = "preview", CROWDB_S3_REGION = "us-east-1", CROWDB_S3_EC_DATA = "2", CROWDB_S3_EC_CODE = "1" } +dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] +fence_listeners = ["127.0.0.1:81"] +[services.probe] +kind = "http" +target = "http://127.0.0.1:81/_crowdb/health/ready" +timeout_ms = 1000 +failure_threshold = 5 +[services.restart] +max_attempts = 5 +backoff_base_ms = 250 +backoff_max_ms = 5000 + +[[services]] +id = "iceberg" +program = "/opt/crowdb/bin/crowdb-iceberg" +args = ["serve"] +env = { CROWDB_ICEBERG_LISTEN = "0.0.0.0:80", CROWDB_ICEBERG_PUBLIC_URI = "http://localhost", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000", CROWDB_ICEBERG_GC_ENABLED = "0" } +dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] +fence_listeners = ["127.0.0.1:80"] +[services.probe] +kind = "http" +target = "http://127.0.0.1:80/v1/config" +bearer_env = "CROWDB_ICEBERG_READ_TOKEN" +timeout_ms = 1000 +failure_threshold = 5 +[services.restart] +max_attempts = 5 +backoff_base_ms = 250 +backoff_max_ms = 5000 + +[[services]] +id = "web" +program = "/opt/crowdb/bin/crowdb-web" +args = ["--config", "/opt/crowdb/run/config/crowdb-web.toml"] +dependencies = ["s3", "iceberg"] +fence_listeners = ["127.0.0.1:8080"] +config_template = "/opt/crowdb/etc/templates/crowdb-web.toml" +[services.probe] +kind = "http" +target = "http://127.0.0.1:8080/healthz" +timeout_ms = 1000 +failure_threshold = 5 +[services.restart] +max_attempts = 5 +backoff_base_ms = 250 +backoff_max_ms = 5000 diff --git a/container/single-node-container/templates/chunk-kv.toml b/container/single-node-container/templates/chunk-kv.toml new file mode 100644 index 000000000..8e9770005 --- /dev/null +++ b/container/single-node-container/templates/chunk-kv.toml @@ -0,0 +1,25 @@ +instance_id = 1 +rpc_listen_addr = "127.0.0.1:15200" +rpc_advertise_addr = "127.0.0.1:15200" +http_listen_addr = "127.0.0.1:15100" +group0_mgmt_seeds = ["http://127.0.0.1:10000"] +catalog_refresh_interval_ms = 200 + +[balance] +enabled = false +target_partitions_per_owner = 1 +target_partition_bytes = 9223372036854775807 +minimum_weighted_improvement_percent = 100 +cooldown_ms = 9223372036854775807 +max_owner_request_rate = 0 + +[storage] +metadata_store_id = 0 +stream_mirror_copies = 1 + +[bootstrap_partition] +partition_id = { high = 1, low = 1 } +tree_id = 1 +stream_name = { high = 2, low = 1 } +owner_epoch = 1 +metadata_group_id = {{group.1.id}} diff --git a/container/single-node-container/templates/chunkdb.toml b/container/single-node-container/templates/chunkdb.toml new file mode 100644 index 000000000..1f5db6f6f --- /dev/null +++ b/container/single-node-container/templates/chunkdb.toml @@ -0,0 +1,27 @@ +[server] +rpc_workers = 2 +http_listen_addr = "127.0.0.1:12100" +rpc_listen_addr = "127.0.0.1:12200" +instance_id = "1" +kv_server_mgmt_seeds = ["http://127.0.0.1:10000"] +keepalive_interval_secs = 1 +kv_pool_size = 1 +kv_rpc_workers = 2 +diskdb_pool_size = 1 +diskdb_rpc_workers = 2 + +[topology] +refresh_interval_secs = 1 + +[range_guard] +allow_all_when_empty = false + +[lifecycle] +cache_capacity = 10000 +sweep_chunk_lock_interval_secs = 60 +lock_hold_warn_threshold_ms = 1000 + +[placement] +mode = "unsafe_colocated" +allow_unsafe_ec = true +allow_degraded_failure_domains = true diff --git a/container/single-node-container/templates/crowdb-web.toml b/container/single-node-container/templates/crowdb-web.toml new file mode 100644 index 000000000..770b0c472 --- /dev/null +++ b/container/single-node-container/templates/crowdb-web.toml @@ -0,0 +1,10 @@ +version = 1 +mode = "docker" +bind = "0.0.0.0" +port = 8080 +group0_management_seeds = ["http://127.0.0.1:10000"] +ui_root = "{{install_root}}/ui" +monitor_status = "{{run_root}}/status/monitor.json" +log_dir = "{{log_root}}/web" +log_max_file_mb = 30 +log_max_files = 5 diff --git a/container/single-node-container/templates/diskdb.toml b/container/single-node-container/templates/diskdb.toml new file mode 100644 index 000000000..6ff07eac6 --- /dev/null +++ b/container/single-node-container/templates/diskdb.toml @@ -0,0 +1,63 @@ +[server] +rpc_workers = 2 +listen_addr = "127.0.0.1:11000" +http_listen_addr = "127.0.0.1:11100" +rpc_listen_addr = "127.0.0.1:11200" +instance_id = "{{node.0.id}}" +kv_server_mgmt_seeds = ["http://127.0.0.1:10000"] + +[storage] +zone_size_bytes = {{disk.0.capacity_bytes}} +block_size_bytes = 1048576 +allocate_granularity = 1048576 +zone_rotate_count = 4 +cas_retry_limit = 100 + +[heartbeat] +interval_secs = 10 +miss_threshold = 3 +temp_failure_timeout_secs = 900 + +[persistence] +free_batch_enabled = false +free_flush_max_batch = 256 +compaction_cadence_secs = 300 +snapshot_compaction_threshold = 4096 +load_concurrency = 16 + +[scanner] +scan_interval_secs = 600 +tentative_owner_scan_interval_secs = 600 +tentative_owner_zone_delay_secs = 3 +tentative_owner_grace_secs = 86400 +reverify_delay_ms = 1000 + +[scanner.ghost] +detect = true +auto_correct = false + +[scanner.integrity] +verify = true +detect_owner_mismatch = false + +[rebalance] +enabled = true +plan_interval_secs = 300 +imbalance_threshold_pct = 20 +max_jobs_per_cycle = 1 +zone_delay_secs = 3 + +[allocator] +load_aware = true +load_aware_weight = "free_bytes" + +[sync] +group0_store_id = 0 +group0_group_id = 0 +sync_interval_secs = 10 + +[reporting] +interval_secs = 10 + +[notify] +notify_enabled = false diff --git a/container/single-node-container/templates/diskio.toml b/container/single-node-container/templates/diskio.toml new file mode 100644 index 000000000..c8f8e8989 --- /dev/null +++ b/container/single-node-container/templates/diskio.toml @@ -0,0 +1,43 @@ +[server] +bind_address = "127.0.0.1" +listen_port = 13000 +rpc_workers = 4 +node_id = {{node.0.id}} +dummy_disk_type = "null" +o_direct = true + +[engine] +thread_pool_size = 4 +sq_entries = 256 + +[group0] +kv_seeds = ["http://127.0.0.1:10000"] +instance_id = {{node.0.id}} +rack_id = {{node.0.rack_id}} +disk_group_id = 101 +sync_interval_ms = 5000 +auto_discover_disks = false + +[metrics] +log_dir = "{{log_root}}/diskio" +interval_secs = 5 + +[[disk]] +id = "1:1" +path = "{{disk.0.path}}" +zone_capacity = {{disk.0.capacity_bytes}} + +[[disk]] +id = "1:2" +path = "{{disk.1.path}}" +zone_capacity = {{disk.1.capacity_bytes}} + +[[disk]] +id = "1:3" +path = "{{disk.2.path}}" +zone_capacity = {{disk.2.capacity_bytes}} + +[[disk]] +id = "1:4" +path = "{{disk.3.path}}" +zone_capacity = {{disk.3.capacity_bytes}} diff --git a/container/single-node-container/templates/kv.toml b/container/single-node-container/templates/kv.toml new file mode 100644 index 000000000..0187ac8f6 --- /dev/null +++ b/container/single-node-container/templates/kv.toml @@ -0,0 +1,5 @@ +wal_early_ack = true +async_engine_apply = true + +[server] +rpc_workers = 2 diff --git a/container/single-node-container/tests/container-e2e.sh b/container/single-node-container/tests/container-e2e.sh new file mode 100644 index 000000000..daf21126f --- /dev/null +++ b/container/single-node-container/tests/container-e2e.sh @@ -0,0 +1,345 @@ +#!/bin/bash +set -euo pipefail + +image=${CROWDB_CONTAINER_IMAGE:-crowdb-iceberg-single-node:dev} +root=$(mktemp -d /tmp/crowdb-preview-e2e.XXXXXX) +name="crowdb-preview-e2e-$$" +chmod 0777 "$root" + +cleanup() { + local status=$? + if (( status != 0 )); then + echo "container E2E failed; diagnostic logs follow" >&2 + docker logs "$name" >&2 || true + if [[ -f "$root/log/monitor/monitor.log" ]]; then + cat "$root/log/monitor/monitor.log" >&2 + fi + if [[ -n "${CROWDB_PREVIEW_TEST_ARTIFACTS:-}" ]]; then + mkdir -p "$CROWDB_PREVIEW_TEST_ARTIFACTS" + docker logs "$name" >"$CROWDB_PREVIEW_TEST_ARTIFACTS/container.log" 2>&1 || true + if [[ -d "$root/log" ]]; then + cp -R "$root/log" "$CROWDB_PREVIEW_TEST_ARTIFACTS/service-logs" + fi + fi + fi + docker rm -fv "$name" >/dev/null 2>&1 || true + docker run --rm --network none --user root \ + --mount "type=bind,source=$root,target=/data" \ + --entrypoint /bin/chmod "$image" -R 0777 /data >/dev/null 2>&1 || true + rm -rf "$root" +} +trap cleanup EXIT + +start_container() { + local storage_mode=${1:-bind} + local mount_args=() + if [[ "$storage_mode" == bind ]]; then + mount_args=(--mount "type=bind,source=$root,target=/opt/crowdb/data") + fi + docker run -d --name "$name" \ + "${mount_args[@]}" \ + -p 127.0.0.1::80 -p 127.0.0.1::81 -p 127.0.0.1::8080 \ + "$image" >/dev/null + for attempt in $(seq 1 240); do + state=$(docker inspect --format '{{.State.Status}}' "$name") + if [[ "$state" != running ]]; then + docker logs "$name" + return 1 + fi + health=$(docker inspect --format '{{.State.Health.Status}}' "$name") + if [[ "$health" == healthy ]]; then + return 0 + fi + sleep 1 + done + docker logs "$name" + return 1 +} + +port() { + local published + published=$(docker port "$name" "$1/tcp") + printf '%s\n' "${published##*:}" +} + +verify_public_services() { + local iceberg_port s3_port web_port token + iceberg_port=$(port 80) + s3_port=$(port 81) + web_port=$(port 8080) + curl --fail --silent --show-error --max-time 5 \ + "http://127.0.0.1:$s3_port/_crowdb/health/ready" >/dev/null + curl --fail --silent --show-error --max-time 5 \ + "http://127.0.0.1:$web_port/api/authority" | jq -e '.source == "group0" and .available == true' >/dev/null + curl --fail --silent --show-error --max-time 5 \ + "http://127.0.0.1:$web_port/api/preview" | jq -e '.source == "group0" and (.services | length) > 0' >/dev/null + token=$(printf '%s\n' "$client_env" | sed -n 's/^ICEBERG_TOKEN=//p') + [[ -n "$token" ]] + printf 'header = "Authorization: Bearer %s"\nurl = "http://127.0.0.1:%s/v1/config"\n' "$token" "$iceberg_port" | + curl --config - --fail --silent --show-error --max-time 5 | jq -e '.defaults != null' >/dev/null + for internal in 10000 10100 11000 13000 15100 15200; do + if docker port "$name" "$internal/tcp" >/dev/null 2>&1; then + echo "internal port $internal is published" >&2 + return 1 + fi + done +} + +verify_clients() { + local operation=$1 + export AWS_DEFAULT_REGION AWS_ACCESS_KEY_ID AWS_SECRET_ACCESS_KEY ICEBERG_TOKEN + AWS_DEFAULT_REGION=$(printf '%s\n' "$client_env" | sed -n 's/^AWS_DEFAULT_REGION=//p') + AWS_ACCESS_KEY_ID=$(printf '%s\n' "$client_env" | sed -n 's/^AWS_ACCESS_KEY_ID=//p') + AWS_SECRET_ACCESS_KEY=$(printf '%s\n' "$client_env" | sed -n 's/^AWS_SECRET_ACCESS_KEY=//p') + ICEBERG_TOKEN=$(printf '%s\n' "$client_env" | sed -n 's/^ICEBERG_TOKEN=//p') + export CROWDB_PREVIEW_S3_ENDPOINT="http://127.0.0.1:$(port 81)" + export CROWDB_PREVIEW_ICEBERG_URI="http://127.0.0.1:$(port 80)" + pixi run -e s3-e2e python container/single-node-container/tests/s3-client.py "$operation" + pixi run -e iceberg-e2e python container/single-node-container/tests/iceberg-client.py "$operation" +} + +verify_web_logical() { + local web_port manage_token status + web_port=$(port 8080) + manage_token=$(docker exec "$name" sed -n 's/^CROWDB_ICEBERG_MANAGE_TOKEN=//p' /opt/crowdb/data/secrets/server.env) + [[ -n "$manage_token" ]] + status=$(curl --silent --show-error --output /dev/null --write-out '%{http_code}' \ + --header 'Content-Type: application/json' --data '{"store_id":7,"nodes":[1]}' \ + "http://127.0.0.1:$web_port/api/stores") + [[ "$status" == 401 ]] + local create_response create_status + create_response=$(printf 'header = "Authorization: Bearer %s"\nheader = "Content-Type: application/json"\nurl = "http://127.0.0.1:%s/api/stores"\n' "$manage_token" "$web_port" | + curl --config - --silent --show-error --max-time 10 \ + --write-out '\n%{http_code}' --data '{"store_id":7,"nodes":[1]}') + create_status=${create_response##*$'\n'} + if [[ "$create_status" != 201 ]]; then + echo "Web store create returned $create_status: ${create_response%$'\n'*}" >&2 + return 1 + fi + jq -e '.store_id == 7 and .nodes == [1]' <<<"${create_response%$'\n'*}" >/dev/null + curl --fail --silent --show-error --max-time 10 \ + "http://127.0.0.1:$web_port/api/stores" | jq -e 'any(.[]; .store_id == 7)' >/dev/null + status=$(printf 'header = "Authorization: Bearer %s"\nurl = "http://127.0.0.1:%s/api/racks"\n' "$manage_token" "$web_port" | + curl --config - --silent --show-error --output /dev/null --write-out '%{http_code}' --request POST) + [[ "$status" == 503 ]] + printf 'header = "Authorization: Bearer %s"\nurl = "http://127.0.0.1:%s/api/stores/7"\n' "$manage_token" "$web_port" | + curl --config - --fail --silent --show-error --max-time 10 --request DELETE >/dev/null + curl --fail --silent --show-error --max-time 10 \ + "http://127.0.0.1:$web_port/api/stores" | jq -e 'all(.[]; .store_id != 7)' >/dev/null +} + +verify_child_recovery() { + local service=$1 signal=$2 expected_event=$3 old_pid old_generation new_pid new_generation state + state=$(docker exec "$name" cat /opt/crowdb/run/status/monitor.json) + old_pid=$(jq -er --arg service "$service" '.services[$service].pid' <<<"$state") + old_generation=$(jq -er --arg service "$service" '.services[$service].generation' <<<"$state") + docker exec "$name" kill -"$signal" "$old_pid" + for attempt in $(seq 1 90); do + if [[ $(docker inspect --format '{{.State.Status}}' "$name") != running ]]; then + echo "container exited while recovering $service" >&2 + return 1 + fi + state=$(docker exec "$name" cat /opt/crowdb/run/status/monitor.json) + new_pid=$(jq -er --arg service "$service" '.services[$service].pid // empty' <<<"$state") || true + new_generation=$(jq -er --arg service "$service" '.services[$service].generation' <<<"$state") + if [[ $(jq -r '.phase' <<<"$state") == ready && "$new_pid" != "$old_pid" && -n "$new_pid" ]] && + (( new_generation > old_generation )); then + docker exec "$name" cat /opt/crowdb/data/log/monitor/monitor.log | + jq -se --arg service "$service" --arg kind "$expected_event" \ + 'any(.[]; .service == $service and .kind == $kind)' >/dev/null + docker exec "$name" crowdb-monitor readiness + return 0 + fi + sleep 1 + done + echo "$service did not recover after $signal" >&2 + return 1 +} + +verify_restart_exhaustion() { + local old_pid state exit_code + for attempt in $(seq 1 5); do + verify_child_recovery web KILL child_exited + done + state=$(docker exec "$name" cat /opt/crowdb/run/status/monitor.json) + old_pid=$(jq -er '.services.web.pid' <<<"$state") + docker exec "$name" kill -KILL "$old_pid" + for attempt in $(seq 1 40); do + if [[ $(docker inspect --format '{{.State.Status}}' "$name") == exited ]]; then + exit_code=$(docker inspect --format '{{.State.ExitCode}}' "$name") + [[ "$exit_code" != 0 ]] + docker logs "$name" 2>&1 | grep -F '"kind":"restart_exhausted"' >/dev/null + return 0 + fi + sleep 1 + done + echo 'container stayed running after restart budget exhaustion' >&2 + return 1 +} + +verify_recovery_identity_rejection() { + local old_pid ready_before ready_after exit_code + ready_before=$(docker exec "$name" cat /opt/crowdb/data/log/monitor/monitor.log | + jq -s '[.[] | select(.kind == "ready")] | length') + old_pid=$(docker exec "$name" cat /opt/crowdb/run/status/monitor.json | jq -er '.services.web.pid') + docker exec --user root "$name" /bin/sh -c 'printf "invalid credentials\n" > /opt/crowdb/data/secrets/server.env' + docker exec "$name" kill -KILL "$old_pid" + for attempt in $(seq 1 40); do + if [[ $(docker inspect --format '{{.State.Status}}' "$name") == exited ]]; then + exit_code=$(docker inspect --format '{{.State.ExitCode}}' "$name") + [[ "$exit_code" != 0 ]] + ready_after=$(docker run --rm --network none --user root \ + --mount "type=bind,source=$root,target=/data" \ + --entrypoint /bin/sh "$image" -c \ + 'cat /data/log/monitor/monitor.log' | + jq -s '[.[] | select(.kind == "ready")] | length') + [[ "$ready_after" == "$ready_before" ]] + docker logs "$name" 2>&1 | grep -F 'server credentials are incomplete' >/dev/null + return 0 + fi + sleep 1 + done + echo 'container restored readiness with changed durable credentials' >&2 + return 1 +} + +verify_invalid_manifest_rejected() { + docker run --rm --network none --user root \ + --mount "type=bind,source=$root,target=/data" \ + --entrypoint /bin/sh "$image" -c \ + 'printf "invalid manifest" > /data/bootstrap/manifest.json' + docker run -d --name "$name" \ + --mount "type=bind,source=$root,target=/opt/crowdb/data" "$image" >/dev/null + for attempt in $(seq 1 30); do + if [[ $(docker inspect --format '{{.State.Status}}' "$name") == exited ]]; then + [[ $(docker inspect --format '{{.State.ExitCode}}' "$name") != 0 ]] + docker logs "$name" 2>&1 | grep -F 'Manifest(' >/dev/null + return 0 + fi + sleep 1 + done + echo 'container accepted a corrupt bootstrap manifest' >&2 + return 1 +} + +verify_invalid_profile_rejected() { + printf 'invalid = true\n' >"$root/bad-profile.toml" + docker run -d --name "$name" \ + --mount "type=bind,source=$root/bad-profile.toml,target=/opt/crowdb/etc/profile.toml,readonly" \ + "$image" >/dev/null + for attempt in $(seq 1 30); do + if [[ $(docker inspect --format '{{.State.Status}}' "$name") == exited ]]; then + [[ $(docker inspect --format '{{.State.ExitCode}}' "$name") != 0 ]] + docker logs "$name" 2>&1 | grep -F 'Profile(' >/dev/null + return 0 + fi + sleep 1 + done + echo 'container accepted an invalid deployment profile' >&2 + return 1 +} + +verify_interrupted_bootstrap() { + local manifest deployment_id completed_steps recovered + docker run -d --name "$name" \ + --mount "type=bind,source=$root,target=/opt/crowdb/data" \ + "$image" >/dev/null + manifest= + for attempt in $(seq 1 400); do + if [[ $(docker inspect --format '{{.State.Status}}' "$name") != running ]]; then + echo 'container exited during bootstrap interruption setup' >&2 + return 1 + fi + manifest=$(docker exec "$name" cat /opt/crowdb/data/bootstrap/manifest.json 2>/dev/null) || true + if jq -e '.state == "initializing" and any(.steps[]; .complete)' <<<"$manifest" >/dev/null 2>&1; then + break + fi + sleep 0.1 + done + if ! jq -e '.state == "initializing" and any(.steps[]; .complete)' <<<"$manifest" >/dev/null; then + echo 'bootstrap did not expose a completed step before readiness' >&2 + return 1 + fi + deployment_id=$(jq -er '.deployment_id' <<<"$manifest") + completed_steps=$(jq -c '[.steps[] | select(.complete) | .name]' <<<"$manifest") + docker kill --signal=KILL "$name" >/dev/null + [[ $(docker wait "$name") != 0 ]] + docker rm "$name" >/dev/null + start_container + recovered=$(docker exec "$name" cat /opt/crowdb/data/bootstrap/manifest.json) + jq -e --arg deployment_id "$deployment_id" --argjson completed_steps "$completed_steps" \ + '. as $manifest | .state == "ready" and .deployment_id == $deployment_id and + all(.steps[]; .complete) and + all($completed_steps[]; . as $name | any($manifest.steps[]; .name == $name and .complete))' \ + <<<"$recovered" >/dev/null + docker exec "$name" crowdb-monitor readiness +} + +echo "checking interrupted bootstrap recovery" +verify_interrupted_bootstrap +echo "checking empty-volume boot" +docker exec "$name" crowdb-monitor readiness +client_env=$(docker exec "$name" crowdb-monitor credentials show --format env) +[[ "$client_env" == *'AWS_ACCESS_KEY_ID='* && "$client_env" == *'ICEBERG_TOKEN='* ]] +[[ $(docker exec "$name" stat -c %a /opt/crowdb/data/secrets/server.env) == 600 ]] +[[ $(docker exec "$name" stat -c %a /opt/crowdb/data/secrets/client.env) == 600 ]] +docker exec "$name" cat /opt/crowdb/data/bootstrap/manifest.json | jq -e '.state == "ready"' >/dev/null +verify_public_services +node container/single-node-container/tests/web-ui.cjs "http://127.0.0.1:$(port 8080)" "$name" +echo "checking S3 and Iceberg client writes" +verify_clients write +echo "checking Web logical writes" +verify_web_logical +for service in kv diskdb diskio chunkdb chunk-kv s3 iceberg web; do + echo "checking $service crash recovery" + verify_child_recovery "$service" KILL child_exited +done +for service in kv diskdb diskio chunkdb chunk-kv s3 iceberg web; do + echo "checking $service hang recovery" + verify_child_recovery "$service" STOP probe_failed +done +verify_public_services +node container/single-node-container/tests/web-ui.cjs "http://127.0.0.1:$(port 8080)" +verify_clients read +sleep 12 +docker exec "$name" crowdb-monitor readiness +if grep -Rq 'local split planned' "$root/log/kv"; then + echo 'disabled chunk-KV balance planned a split' >&2 + exit 1 +fi + +docker stop --time 15 "$name" >/dev/null +docker rm "$name" >/dev/null +start_container +echo "checking persisted-volume restart" +docker exec "$name" crowdb-monitor readiness +[[ "$(docker exec "$name" crowdb-monitor credentials show --format env)" == "$client_env" ]] +docker exec "$name" cat /opt/crowdb/data/bootstrap/manifest.json | jq -e '.state == "ready"' >/dev/null +verify_public_services +verify_clients read +echo "checking restart budget exhaustion" +verify_restart_exhaustion +python container/single-node-container/tests/logs.py "$root" "$name" +docker rm "$name" >/dev/null +start_container +verify_public_services +verify_clients read +echo "checking recovery rejects changed durable identity" +verify_recovery_identity_rejection +docker rm -v "$name" >/dev/null +echo "checking corrupt manifest rejection" +verify_invalid_manifest_rejected +docker rm -v "$name" >/dev/null +echo "checking invalid profile rejection" +verify_invalid_profile_rejected +docker rm -v "$name" >/dev/null +start_container anonymous +echo "checking default anonymous-volume boot" +docker inspect "$name" | jq -e '.[0].Mounts | any(.Destination == "/opt/crowdb/data" and .Type == "volume")' >/dev/null +docker exec "$name" crowdb-monitor readiness +docker exec "$name" cat /opt/crowdb/data/bootstrap/manifest.json | jq -e '.state == "ready"' >/dev/null +docker logs "$name" 2>&1 | grep -F 'For data you want to keep across container recreation' >/dev/null +echo "checking monitor death exits the container" +docker kill --signal=KILL "$name" >/dev/null +[[ $(docker wait "$name") != 0 ]] +echo "container E2E passed" diff --git a/container/single-node-container/tests/iceberg-client.py b/container/single-node-container/tests/iceberg-client.py new file mode 100644 index 000000000..5dec8524e --- /dev/null +++ b/container/single-node-container/tests/iceberg-client.py @@ -0,0 +1,35 @@ +import os +import sys + +from pyiceberg.catalog import load_catalog +from pyiceberg.schema import Schema +from pyiceberg.types import LongType, NestedField + + +NAMESPACE = ("crowdb-preview-e2e",) +TABLE = NAMESPACE + ("events",) + + +def main(): + catalog = load_catalog( + "crowdb-preview", + type="rest", + uri=os.environ["CROWDB_PREVIEW_ICEBERG_URI"], + token=os.environ["ICEBERG_TOKEN"], + ) + if sys.argv[1] == "write": + catalog.create_namespace(NAMESPACE, {"preview": "persisted"}) + table = catalog.create_table( + TABLE, + Schema(NestedField(field_id=1, name="id", field_type=LongType(), required=True)), + ) + table.transaction().set_properties({"preview": "persisted"}).commit_transaction() + assert catalog.namespace_exists(NAMESPACE) + assert NAMESPACE in catalog.list_namespaces() + assert catalog.load_namespace_properties(NAMESPACE) == {"preview": "persisted"} + assert TABLE in catalog.list_tables(NAMESPACE) + assert catalog.load_table(TABLE).properties["preview"] == "persisted" + + +if __name__ == "__main__": + main() diff --git a/container/single-node-container/tests/image-smoke.sh b/container/single-node-container/tests/image-smoke.sh new file mode 100644 index 000000000..584d1507b --- /dev/null +++ b/container/single-node-container/tests/image-smoke.sh @@ -0,0 +1,50 @@ +#!/bin/bash +set -euo pipefail + +image=${CROWDB_CONTAINER_IMAGE:-crowdb-iceberg-single-node:dev} +docker image inspect "$image" >/dev/null +image_bytes=$(docker image inspect --format '{{.Size}}' "$image") +if ((image_bytes > 325000000)); then + echo "single-node container image exceeds 325 MB: $image_bytes bytes" >&2 + exit 1 +fi +test "$(docker image inspect --format '{{.Architecture}}' "$image")" = amd64 +test "$(docker image inspect --format '{{.Config.User}}' "$image")" = crowdb:crowdb +test "$(docker image inspect --format '{{index .Config.Labels "org.opencontainers.image.version"}}' "$image")" = "$(cat VERSION)" +test "$(docker image inspect --format '{{index .Config.Labels "org.opencontainers.image.revision"}}' "$image")" = "$(git rev-parse HEAD)" +volumes=$(docker image inspect --format '{{json .Config.Volumes}}' "$image") +jq -e 'has("/opt/crowdb/data")' <<<"$volumes" >/dev/null +exposed=$(docker image inspect --format '{{json .Config.ExposedPorts}}' "$image") +for port in 80 81 8080; do + jq -e --arg port "$port/tcp" 'has($port)' <<<"$exposed" >/dev/null +done +for port in 10000 13000 15200; do + jq -e --arg port "$port/tcp" 'has($port) | not' <<<"$exposed" >/dev/null +done + +docker run --rm --network none --entrypoint /opt/crowdb/bin/crowdb-monitor "$image" validate /opt/crowdb/etc/profile.toml +docker run --rm --network none --entrypoint /bin/sh "$image" -ec ' + for tool in pixi cargo rustc gcc g++ cmake npm; do + if command -v "$tool" >/dev/null 2>&1; then + echo "Build tool was packaged into the runtime image: $tool" >&2 + exit 1 + fi + done + for library in /opt/crowdb/lib/libboost_regex* /opt/crowdb/lib/libicu*; do + if [ -e "$library" ]; then + echo "Unused Boost.Regex/ICU dependency was packaged: $library" >&2 + exit 1 + fi + done +' +for binary in crowdb-iceberg crowdb-access-server; do + capability=$(docker run --rm --network none --entrypoint /sbin/getcap "$image" "/opt/crowdb/bin/$binary") + [[ "$capability" == *'cap_net_bind_service=ep' ]] +done + +iceberg_output=$(docker run --rm --network none --entrypoint /opt/crowdb/bin/crowdb-iceberg "$image" 2>&1) && { + echo "Iceberg started without required configuration" >&2 + exit 1 +} +printf '%s\n' "$iceberg_output" +[[ "$iceberg_output" == *'Error: NotPresent'* ]] diff --git a/container/single-node-container/tests/logs.py b/container/single-node-container/tests/logs.py new file mode 100644 index 000000000..619b83c3c --- /dev/null +++ b/container/single-node-container/tests/logs.py @@ -0,0 +1,48 @@ +"""Check lifecycle coverage and secret exclusion in disposable container logs.""" + +import gzip +import json +import os +from pathlib import Path +import subprocess +import sys + +root = Path(sys.argv[1]) +container = sys.argv[2] +secrets = [] +for name in ["server.env", "client.env"]: + private = subprocess.run([ + "docker", "run", "--rm", "--network", "none", "--user", "root", + "--mount", f"type=bind,source={root},target=/data,readonly", + "--entrypoint", "/bin/cat", + os.environ.get("CROWDB_CONTAINER_IMAGE", "crowdb-iceberg-single-node:dev"), + f"/data/secrets/{name}", + ], capture_output=True, check=True, timeout=30) + for line in private.stdout.splitlines(): + key, _, value = line.partition(b"=") + if value and any(word in key for word in [b"TOKEN", b"KEY"]): + secrets.append(value) + +assert secrets, "credential files contain no secret values to check" + +def check_secret_free(body): + assert all(secret not in body for secret in secrets), "secret leaked into diagnostic logs" + +events = set() +for path in (root / "log").rglob("*"): + if not path.is_file(): + continue + body = gzip.decompress(path.read_bytes()) if path.suffix == ".gz" else path.read_bytes() + check_secret_free(body) + if path.name.startswith("monitor.log"): + for line in body.splitlines(): + events.add(json.loads(line)["kind"]) +required = {"starting", "ready", "bootstrap_step_started", "bootstrap_step_completed", + "child_started", "child_stopped", "child_exited", "probe_failed", + "restarting", "draining", "stopped", "restart_exhausted"} +assert required <= events, f"missing lifecycle events: {required - events}" +logs = subprocess.run(["docker", "logs", container], capture_output=True, check=True, timeout=10) +check_secret_free(logs.stdout + logs.stderr) +print("Container lifecycle events and secret-free logs passed") +# Copyright 2026-present Gian +# Licensed under the Apache License, Version 2.0. diff --git a/container/single-node-container/tests/release-policy.sh b/container/single-node-container/tests/release-policy.sh new file mode 100644 index 000000000..cc75258f3 --- /dev/null +++ b/container/single-node-container/tests/release-policy.sh @@ -0,0 +1,49 @@ +#!/bin/bash +set -euo pipefail + +release=.github/workflows/release-container.yml +ci=.github/workflows/ci.yml + +events=$(sed -n '/^on:/,/^concurrency:/p' "$release") +[[ "$events" == *'workflow_dispatch:'* ]] +! grep -Eq '^ (push|pull_request|release|create):' <<<"$events" + +for required in \ + 'environment: DockerHub' \ + 'DOCKERHUB_TOKEN' \ + 'ref: ${{ inputs.tag }}' \ + 'git rev-parse --verify "refs/tags/$RELEASE_TAG^{commit}"' \ + '[[ "$revision" == "$(git rev-parse HEAD)" ]]' \ + 'gh release view "$RELEASE_TAG"' \ + '[[ "$status" == 404 ]]' \ + 'needs: verify' \ + 'docker.io/crowdb/crowdb-iceberg:${{ inputs.tag }}' \ + 'docker.io/crowdb/crowdb-iceberg:git-${{ needs.verify.outputs.revision }}' \ + 'provenance: mode=max' \ + 'sbom: true' \ + 'cosign sign --yes'; do + grep -Fq "$required" "$release" +done +! grep -Eq 'crowdb-iceberg:(preview|latest)' "$release" +[[ $(grep -c 'push: true' "$release") == 1 ]] +[[ $(grep -c 'id-token: write' "$release") == 1 ]] +[[ "$events" != *'schedule:'* ]] + +verify_job=$(sed -n '/^ verify:/,/^ publish:/p' "$release") +publish_job=$(sed -n '/^ publish:/,$p' "$release") +[[ "$verify_job" == *'name: verified-container-runtime'* ]] +[[ "$publish_job" == *'name: verified-container-runtime'* ]] +[[ "$publish_job" == *'context: target/container-runtime'* ]] +for gate in 'pixi run test-single-node-container' 'test-boto3-e2e' 'test-pyiceberg-e2e' \ + 'pixi run test-console' 'pixi run test-console-ui' 'pixi run rs-fmt-check && pixi run rs-lint'; do + [[ "$verify_job" == *"$gate"* ]] +done +! grep -Eq 'DOCKERHUB_|push: true|id-token: write' <<<"$verify_job" +[[ "$publish_job" == *'needs: verify'* && "$publish_job" == *'environment: DockerHub'* ]] +[[ "$publish_job" != *'RELEASE_ENABLED'* ]] +[[ "$publish_job" == *'[[ "$(git rev-parse HEAD)" == "$REVISION" ]]'* ]] + +ci_job=$(sed -n '/^ DockerPreview:/,$p' "$ci") +[[ "$ci_job" == *'contents: read'* && "$ci_job" == *'pixi run test-single-node-container'* ]] +[[ "$ci_job" == *'Upload preview failure logs'* && "$ci_job" == *'CROWDB_PREVIEW_TEST_ARTIFACTS'* ]] +! grep -Eq 'secrets\.|docker/login-action|docker/build-push-action' <<<"$ci_job" diff --git a/container/single-node-container/tests/s3-client.py b/container/single-node-container/tests/s3-client.py new file mode 100644 index 000000000..0f76e9538 --- /dev/null +++ b/container/single-node-container/tests/s3-client.py @@ -0,0 +1,46 @@ +import base64 +import os +from pathlib import Path +import re +import sys + +import boto3 +from botocore.config import Config + + +BUCKET = "crowdb-preview-e2e" +KEY = "objects/persisted.parquet" +fixture = Path(__file__).resolve().parents[3] / "lib/crowdb-access-iceberg/tests/common/parquet_scalar_official.rs" +encoded = re.search(r'pub const PARQUET_1_0_FALSE: &str = "([^"]+)"', fixture.read_text()).group(1) +BODY = base64.b64decode(encoded) +assert BODY.startswith(b"PAR1") and BODY.endswith(b"PAR1") + + +def main(): + client = boto3.client( + "s3", + endpoint_url=os.environ["CROWDB_PREVIEW_S3_ENDPOINT"], + region_name=os.environ["AWS_DEFAULT_REGION"], + aws_access_key_id=os.environ["AWS_ACCESS_KEY_ID"], + aws_secret_access_key=os.environ["AWS_SECRET_ACCESS_KEY"], + config=Config( + s3={"addressing_style": "path"}, + request_checksum_calculation="when_required", + response_checksum_validation="when_required", + ), + ) + if sys.argv[1] == "write": + client.create_bucket(Bucket=BUCKET) + client.put_object(Bucket=BUCKET, Key=KEY, Body=BODY) + assert BUCKET in {item["Name"] for item in client.list_buckets()["Buckets"]} + listed = client.list_objects_v2(Bucket=BUCKET, Prefix="objects/") + assert [item["Key"] for item in listed["Contents"]] == [KEY] + head = client.head_object(Bucket=BUCKET, Key=KEY) + assert head["ContentLength"] == len(BODY) + assert head["LastModified"] is not None + assert client.get_object(Bucket=BUCKET, Key=KEY)["Body"].read() == BODY + assert client.get_object(Bucket=BUCKET, Key=KEY, Range="bytes=5-13")["Body"].read() == BODY[5:14] + + +if __name__ == "__main__": + main() diff --git a/container/single-node-container/tests/web-ui.cjs b/container/single-node-container/tests/web-ui.cjs new file mode 100644 index 000000000..805704a04 --- /dev/null +++ b/container/single-node-container/tests/web-ui.cjs @@ -0,0 +1,66 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +const { existsSync, mkdirSync } = require('node:fs'); +const { execFileSync } = require('node:child_process'); +const { resolve } = require('node:path'); +const { chromium, expect } = require('../../../app/crowdb-web/ui/node_modules/@playwright/test'); + +async function main() { + const executablePath = process.env.PLAYWRIGHT_CHROMIUM_EXECUTABLE || [ + '/snap/bin/chromium', '/usr/bin/chromium', '/usr/bin/chromium-browser', + '/usr/bin/google-chrome', '/usr/bin/google-chrome-stable', '/usr/bin/microsoft-edge', + ].find(existsSync); + if (!executablePath) throw new Error('Container acceptance requires an installed system browser'); + const browser = await chromium.launch({ executablePath, headless: true }); + try { + const page = await browser.newPage(); + await page.goto(process.argv[2]); + await expect(page.getByTestId('managed-source')).toHaveText('Source: Group 0', { timeout: 3000 }); + await expect(page.getByTestId('managed-readonly')).toHaveText('Hardware topology is read-only', { timeout: 3000 }); + await expect(page.getByRole('region', { name: 'Preview summary' })).toBeVisible({ timeout: 3000 }); + await expect(page.getByTestId('managed-monitor-phase')).toContainText('Phase: ready', { timeout: 3000 }); + for (const service of ['kv', 'diskdb', 'diskio', 'chunkdb', 'chunk-kv', 's3', 'iceberg', 'web']) { + await expect(page.getByTestId(`managed-process-${service}`)).toContainText(/PID \d+ · generation \d+/, { timeout: 3000 }); + } + await expect(page.getByTestId('managed-unavailable')).toHaveCount(0, { timeout: 3000 }); + if (process.argv[3]) { + await verifyAuthorityOutage(page, process.argv[3]); + } + if (process.env.CROWDB_PREVIEW_TEST_ARTIFACTS) { + mkdirSync(process.env.CROWDB_PREVIEW_TEST_ARTIFACTS, { recursive: true }); + await page.screenshot({ path: resolve(process.env.CROWDB_PREVIEW_TEST_ARTIFACTS, 'managed-web.png'), fullPage: true }); + } + console.log('Container managed Web browser acceptance passed'); + } finally { + await browser.close(); + } +} + +async function verifyAuthorityOutage(page, container) { + const docker = (...args) => execFileSync('docker', args, { encoding: 'utf8', timeout: 10000 }); + const status = JSON.parse(docker('exec', container, 'cat', '/opt/crowdb/run/status/monitor.json')); + const pid = String(status.services.kv.pid); + await page.clock.install(); + docker('exec', container, 'kill', '-STOP', pid); + try { + const failedSnapshot = page.waitForResponse(response => + response.url().endsWith('/api/preview') && response.status() === 503, + { timeout: 10000 }); + await page.clock.runFor(3001); + const response = await failedSnapshot; + const body = await response.json(); + expect(body.reason).toBe('group0_unavailable'); + expect(body.monitor.services.kv.pid).toBe(Number(pid)); + await expect(page.getByTestId('managed-unavailable')).toContainText('Group 0 is unavailable', { timeout: 3000 }); + await expect(page.getByRole('region', { name: 'Preview summary' })).toHaveCount(0, { timeout: 3000 }); + await expect(page.getByRole('region', { name: 'Monitor status' })).toBeVisible({ timeout: 3000 }); + } finally { + docker('exec', container, 'kill', '-CONT', pid); + } + await page.clock.runFor(3001); + await expect(page.getByTestId('managed-unavailable')).toHaveCount(0, { timeout: 3000 }); + await expect(page.getByRole('region', { name: 'Preview summary' })).toBeVisible({ timeout: 3000 }); +} + +main().catch((error) => { console.error(error); process.exitCode = 1; }); diff --git a/doc/backlog/R177-access-iceberg-catalog-foundation.md b/doc/backlog/R177-access-iceberg-catalog-foundation.md deleted file mode 100644 index 0cb1e58bf..000000000 --- a/doc/backlog/R177-access-iceberg-catalog-foundation.md +++ /dev/null @@ -1,238 +0,0 @@ - - - -### R177: access server / Iceberg — Native Iceberg storage blueprint - -## Problem - -CROWDB has Chunk-KV, chunk storage, and an S3 protocol, but it does not yet have an -Iceberg authority. Treating Iceberg as ordinary S3 objects would lose the catalog -name hierarchy, atomic table commits, immutable metadata and data files, snapshot -reachability, and spec-defined conflict behavior. It would also allow general S3 -overwrite and delete rules to violate the Iceberg table specification. - -The implementation needs one program contract before catalog, namespace, table, -file, reclamation, REST, and cache work can proceed independently. This requirement -owns that contract and every decision shared by R178 through R185. The permanent -architecture is [Native Iceberg Storage](../design/access-server/iceberge/design-crowdb-iceberg.md), -and the backed-up Apache specifications under `doc/design/access-server/iceberge/` are the -normative protocol and format references. - -## Solution - -### 1. Core milestone - -The first milestone implements one active catalog, multipart namespaces, table -CRUD and rename, Iceberg format v1, v2, and v3 metadata and files, optimistic table -commits, snapshots and references, native immutable file storage, and an -Iceberg-owned S3-shaped FileIO surface. Each format version has separate -parse/read/create/write capabilities. The implementation supports the spec-defined -v1-to-v2 and v2-to-v3 upgrades only after validating every intermediate invariant. - -The format profile includes schema, partition-spec, and sort-order evolution; -sequence numbers and row-level deletes; row lineage, deletion vectors, default -values, and v3 types and encodings; snapshot references and retention metadata; -statistics and partition statistics; and the Avro, Parquet, ORC, and Puffin rules -needed by those features. Optional behavior is capability-gated where the table -spec permits it. A server must not advertise write support for a version while -ignoring a mandatory field, inheritance rule, validation, or file encoding. - -The milestone does not advertise views, multi-table transactions, register-table, -server-side scan planning, multiple active catalogs, tenants, or warehouses. -Unsupported endpoints and optional features return the precise standard -unsupported response and perform no mutation. - -### 2. Authority hierarchy - -```text -system root -> active CatalogId/activation epoch - -> catalog authority - -> namespace name index -> NamespaceId -> namespace authority - -> table name index -> TableId -> TableHead/current metadata generation - -> immutable metadata, manifest, data, delete, and statistics files -``` - -- `CatalogId`, `NamespaceId`, `TableId`, and `FileId` are random, non-zero, - non-reused 128-bit CROWDB identities. Iceberg's `table-uuid` remains a distinct - spec field in table metadata. -- Name mappings are ordered lookup indexes. Stable-ID records are authoritative; - list and load filter mappings whose ID, lifecycle, or name epoch is stale. -- All Iceberg keys use a versioned `ICE\0` protocol prefix and live below the - current CatalogId except the single active-catalog root. -- Chunk-KV stores bounded authorities, mappings, heads, operation state, and file - records. Chunk storage owns all non-inline file bytes. Disk, EC, placement, and - node identities never enter Iceberg metadata or locations. -- The immutable standard table metadata JSON plus the `TableHead` that selects it - are the recoverable table-state authority. Binary projections are disposable, - generation-qualified accelerators. - -### 3. Program invariants - -- **ICE-I1 — Stable identity:** rename never changes a CatalogId, NamespaceId, - TableId, FileId, metadata location, or committed bytes. -- **ICE-I2 — One authority:** an index, cache, projection, or notification cannot - publish or repair catalog state; it must validate against its stable authority. -- **ICE-I3 — Atomic generation:** one successful commit performs one `TableHead` - compare-exchange that selects one complete immutable metadata generation. -- **ICE-I4 — Immutable files:** a published canonical location always resolves to - the same length, digest, and bytes and cannot be overwritten. -- **ICE-I5 — Bounded work:** no authority value contains unbounded children; every - request, scan page, stream window, projection, operation, retry, and GC batch has - independent byte, item, and concurrency limits. -- **ICE-I6 — Recoverable mutation:** a durable request identity, request digest, - phase, and result make response-loss retry safe on another Access Server. Reuse - of one identity with different input fails. -- **ICE-I7 — Domain clear:** after clear completes, no new request, cache entry, - credential, location, or retry record can expose the retired CatalogId. -- **ICE-I8 — Spec honesty:** only implemented endpoints and format capabilities are - advertised; unknown, disabled, or lossy requirements and updates fail closed. -- **ICE-I9 — Protocol ownership:** the Iceberg FileIO surface shares low-level - storage clients with S3 but never uses general S3 bucket/object authority. -- **ICE-I10 — Lock-free hot path:** implementations add no catalog-wide or global - cache lock to lookup, load, commit, or file streaming. - -### 4. Requirement decomposition and order - -1. R178 establishes the library, active catalog domain, management safety, stable - key/value envelope, server lifecycle, and `/v1/config` baseline. -2. R179 implements namespace authority and standard namespace operations. -3. R180 implements native immutable files, streaming/range FileIO, multipart, and - generation-local metadata projections. It can proceed after R178 in parallel - with R179. -4. R181 implements table identity, v1/v2/v3 metadata validation, lifecycle, load, - list, rename, and drop on R179 and R180. -5. R182 implements atomic create/staged-create and update commits, requirements, - updates, format upgrades, idempotency, conflict classification, and recovery. -6. R183 implements snapshot-aware purge, orphan cleanup, retired catalog cleanup, - and bounded reclamation after R180 through R182 define reachability. -7. R184 completes public REST integration, authentication, endpoint discovery, - standard errors, and official-client conformance for the core profile. -8. R185 adds bounded caches and cross-server invalidation after all identities, - epochs, digests, and reclamation fences are stable. - -R178 through R184 form the correctness milestone. R185 is a later performance -milestone and cannot be required for correctness. - -### 5. Resolved open issues - -All open issues from the former R177-A through R177-D drafts and the former R178 -cache draft are answered here. Child requirements must reference these decisions -and must not carry independent open questions. - -1. **Clear boundary:** clear enters maintenance, stops new admission, publishes the - new active pointer, and returns only after the maximum old-root cache lease plus - bounded admitted-request and delegated-access grace. It does not wait for - acknowledgements from a possibly stale instance registry. Reclamation also - waits for durable reader and operator pins. -2. **Tenant:** the first milestone stores no default tenant and puts no fixed - TenantId in hot keys. A later tenant root may map to an active CatalogId without - changing catalog-scoped keys. -3. **Warehouse:** absent or empty `warehouse` selects the sole catalog. Any non-empty - value returns the spec-defined `NoSuchWarehouse` response; it is never ignored - or created implicitly. -4. **Retired-catalog safety:** a configurable minimum retention period, clear grace, - durable reader pins, and explicit operator pins all fence physical GC. -5. **Catalog management:** R178 owns authenticated initialize/status/rename/clear - management commands. Clear requires a dedicated privilege, explicit destructive - confirmation, request identity, and durable audit record; it is not an Iceberg - REST endpoint. -6. **Namespaces:** arbitrary multipart identifiers are supported within configured - maximum levels and encoded bytes. Parent listing is complete. Namespace rename - is not implemented because it is non-standard; table rename may move across - namespaces. -7. **Namespace drop:** CAS the namespace to `Dropping`, reject new children, then - perform bounded existence probes of child namespace and table indexes. Restore - `Ready` on a non-empty result; tombstone only an empty fenced authority. -8. **Namespace listing:** scan ordered mappings with bounded over-fetch, validate - targets in bounded batches, and bind the opaque continuation token to catalog, - parent, parameters, and last scanned key. Stale mappings are omitted. -9. **Namespace properties:** at most 256 entries; keys and values are UTF-8 without - NUL, at most 1 KiB and 8 KiB respectively; the encoded authority is at most - 64 KiB. Duplicate remove/update keys return the standard 422 response. -10. **Metadata projections:** metadata JSON gets a durable, disposable, - generation-local root/page/child projection. Other parsed format structures - stay in R185's memory cache until measurements justify a later requirement. -11. **Register table:** it is deferred and not advertised because external - locations could bypass native CROWDB file authority. -12. **Table identity:** CROWDB TableId and Iceberg `table-uuid` are distinct and - both validated. Neither is derived from a mutable name. -13. **Rename and drop:** durable operation records reserve destinations and drive - a recoverable state machine. `TableHead` name epoch/lifecycle decides validity; - stale source or target mappings are filtered. An old name never remains an - alias after rename. -14. **Snapshot loading:** both `ALL` and `REFS` are supported for declared endpoints; - each is generated from the same selected metadata generation. -15. **Format profile:** v1, v2, and v3 each support parse, read, create, and write. - Mandatory version-specific semantics are implemented, and v1-to-v2 and - v2-to-v3 upgrades are supported. Optional spec features remain separately - capability-gated and cannot be silently discarded. -16. **Canonical location:** use the table prefix - `s3://iceberg-/t//`. An exact client-created - relative key beneath it maps once to a server FileId. The reserved bucket and - prefix are decoded by the Iceberg-owned FileIO service; catalog, namespace, and - table names never participate in a location. -17. **FileIO operations:** the first writable milestone includes immutable PUT, - HEAD, one-range GET, create/upload/list/complete/abort multipart, and delegated - credentials. Bucket CRUD, overwrite, tagging, lifecycle, and unrestricted - DELETE are unsupported. -18. **Multipart:** multipart is required for the first writable milestone; all - sessions, parts, bytes, TTLs, completion, and abort work are durable and bounded. -19. **Reclamation:** R183 uses generation-indexed candidates plus traversal from - retained snapshot roots and metadata logs. It never relies on racing per-file - reference counts. -20. **Projection retention:** metadata projections are generation-local without - cross-generation content deduplication in the first milestone. Simple bounded - GC is preferred over reference-count and write-amplification complexity. -21. **Cache clear fencing:** R185 uses the lease-plus-grace clear boundary in item - 1; notification remains only a latency optimization. -22. **Cache limits:** R185 defines separate configurable hard caps for every class, - queue, batch, fill, and fanout dimension. Initial defaults come from its focused - benchmark gate and configuration tests; no class inherits an unbounded or - universal one-size value. - -## Dependencies - -- Depends on routed Chunk-KV compare-exchange and scans, chunk streaming and range - reads, stable request identities, Group-0 service discovery, and `crowdb-rpc`. -- The Apache Iceberg REST OpenAPI and table specification in - `doc/design/access-server/iceberge/` are normative. Apache Java, `iceberg-rust`, official - clients, and the REST Compatibility Kit are test oracles, not production - authorities. -- R178 through R185 depend on this requirement. A material decision change must - first update R177 and the affected acceptance contracts. -- General S3 requirements do not gate Iceberg correctness. Shared chunk and - transport improvements may be reused only below the protocol-authority boundary. - -## Acceptance - -- Given the eight child requirements, when their scopes and dependencies are - inspected, assert every core Catalog, Namespace, Table, commit, FileIO, - reclamation, REST, and cache concern has exactly one owner and R185 is not on the - correctness path. Invariant: ICE-I2 one authority. Integration test. -- Given every declared v1, v2, and v3 capability and upgrade, when metadata and - files are compared with the backed-up table spec and reference implementations, - assert mandatory semantics are preserved and unknown or disabled optional - features fail before mutation. Invariant: ICE-I8 spec honesty. Integration test. -- Given any supported or unsupported REST endpoint, when `/v1/config` and an - operation are compared with the backed-up OpenAPI, assert advertised behavior is - implemented and unadvertised behavior fails without mutation. Invariant: ICE-I8 - spec honesty. E2E test. -- Given rename, commit, clear, response loss, and crash injection, when another - Access Server resumes the operation, assert stable identities, one selected - generation, retry digest equality, and the clear boundary remain true. - Invariants: ICE-I1, ICE-I3, ICE-I6, and ICE-I7. Integration test. -- Given metadata from bytes to hundreds of MiB and data files to TiB scale, when - load, commit, range read, and GC execute, assert no KV value, request allocation, - stream window, page, or background batch grows with the complete table or file. - Invariant: ICE-I5 bounded work. Integration test. -- Given the canonical S3-shaped location and a general S3 object with a similar - textual key, when each is accessed, assert only the Iceberg authority can publish, - overwrite, authorize deletion, or reclaim the Iceberg file. Invariants: ICE-I4 - and ICE-I9. E2E test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-iceberg --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` diff --git a/doc/backlog/R178-access-iceberg-catalog-domain.md b/doc/backlog/R178-access-iceberg-catalog-domain.md deleted file mode 100644 index 79f34b52c..000000000 --- a/doc/backlog/R178-access-iceberg-catalog-domain.md +++ /dev/null @@ -1,107 +0,0 @@ - - - -### R178: access server / Iceberg — Catalog domain and service foundation - -## Problem - -Iceberg REST treats the catalog as configured service context and does not define -catalog create, rename, or clear endpoints. CROWDB still needs a stable root for -all namespace, table, file, retry, and reclamation state. Reading one unqualified -root, deriving identity from a display name, or synchronously deleting descendants -would create a hot key, make rename move data, and make clear unbounded. - -R177 defines one active catalog, a stable CatalogId, no tenant or warehouse in the -first milestone, and a lease-plus-grace clear boundary. This requirement builds the -library and service foundation on which all other Iceberg requirements depend. - -## Solution - -- **CAT-I1 — Single active root:** one `ActiveCatalogRecord` is the only authority - selecting the visible CatalogId and activation epoch. -- **CAT-I2 — Stable domain:** display-name and configuration changes never change - CatalogId or descendant key prefixes. -- **CAT-I3 — Root publication:** initialize and clear publish visibility with one - compare-exchange; candidates not selected by it remain unreachable. -- **CAT-I4 — Admission fence:** every admitted request carries immutable CatalogId - and activation epoch context; clear completion follows R177's lease-plus-grace - rule. -- **CAT-I5 — Safe management:** initialize, rename, and clear are authenticated - CROWDB management operations, not Iceberg REST endpoints. - -1. Create feature-gated `lib/crowdb-access-iceberg` modules for `catalog`, `key`, - `record`, `operation`, `wire`, and `error`. Keep REST wire models, Iceberg domain - models, and versioned FlatBuffer storage records separate. Unknown key or value - versions, malformed IDs, and oversized values fail closed. -2. Reserve the binary `ICE\0`, key-version, and scope prefix. Store the system root - outside CatalogId ranges and every descendant record inside a half-open - CatalogId range. Use big-endian fixed fields and binary-safe length-delimited - variable fields. -3. Store bounded `ActiveCatalogRecord` and `CatalogAuthority` values. The authority - holds display name, name/config generations, lifecycle, and the v1/v2/v3 - parse/read/create/write and upgrade capability matrix, but no child collection. -4. Implement idempotent initialize, epoch-checked display rename, and clear as - recoverable state machines in `catalog/repository.rs`. Clear creates a new empty - authority and publishes it by root CAS; old lifecycle marking is reconciliation, - not the commit point. -5. Add durable management request and audit records. Clear requires a distinct - privilege, exact active epoch, explicit confirmation material bound into the - request digest, and an operator-visible result. A retry with the same identity - and digest returns the original result. -6. Add Iceberg server configuration and lifecycle wiring in - `app/crowdb-access-server/src/iceberg/`. Startup connects routed Chunk-KV and - chunk clients, validates the active root, and only then opens the separate - external Iceberg HTTP listener. Shutdown stops admission before draining - mutations and background work. -7. Implement the baseline `GET /v1/config`. Absent or empty `warehouse` selects the - active catalog; non-empty warehouse returns `NoSuchWarehouse`. The response - advertises only endpoints landed by later requirements and the exact v1/v2/v3 - capability matrix. It does not derive a REST prefix from the display name. -8. Do not introduce a global lock. Before R185, request admission reads the active - root authoritatively. R185 may add a bounded lease-qualified cache without - changing this contract. - -## Dependencies - -- Depends on R177, routed Chunk-KV compare-exchange and scans, stable request - identity, Access Server configuration, authentication, and audit facilities. -- Produces CatalogId, activation epoch, key/value envelope, service lifecycle, and - capability types consumed by R179 through R185. -- Old-catalog physical cleanup is R183. Before R183 lands, retired domains remain - unreachable but are not erased. -- Cache fanout is R185. Before it lands, authoritative root reads preserve correct - behavior at higher latency. - -## Acceptance - -- Given an empty root and concurrent initialize requests, when all instances publish - candidates, assert exactly one active pointer wins and same-identity retries - return that result. Invariants: CAT-I1 and CAT-I3. Integration test. -- Given a ready catalog, when rename succeeds or loses a concurrent CAS, assert the - winning display name and name epoch are deterministic while CatalogId and every - descendant prefix remain unchanged. Invariant: CAT-I2. Integration test. -- Given crashes immediately before and after clear's root CAS, when reconciliation - resumes on another instance, assert the old or new CatalogId is uniquely active, - respectively, and no candidate is partially visible. Invariant: CAT-I3. E2E test. -- Given a clear operation and old admitted work, when the new pointer is published, - assert new admission cannot acquire old context and clear does not report complete - until the bounded lease and request/delegated-access grace expires. Invariant: - CAT-I4. E2E test. -- Given missing confirmation, stale epoch, insufficient privilege, response loss, - or a reused identity with a different digest, when clear is requested, assert no - unauthorized second mutation occurs and the durable audit result is exact. - Invariant: CAT-I5. Integration test. -- Given absent, empty, and non-empty warehouse parameters, when an official client - calls `/v1/config`, assert the sole catalog is selected for the first two and the - last receives `NoSuchWarehouse`; the advertised endpoints equal landed support. - Invariant: CAT-I1. E2E test. -- Given unknown key/value versions, zero IDs, mismatched IDs, oversized records, and - epoch overflow, when codecs and repositories process them, assert they fail closed - without mutation. Invariant: CAT-I3. Unit test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-iceberg --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` diff --git a/doc/backlog/R179-access-iceberg-namespace.md b/doc/backlog/R179-access-iceberg-namespace.md deleted file mode 100644 index 418ff2876..000000000 --- a/doc/backlog/R179-access-iceberg-namespace.md +++ /dev/null @@ -1,95 +0,0 @@ - - - -### R179: access server / Iceberg — Namespace authority and REST operations - -## Problem - -Iceberg namespaces are multipart identifiers with parent-scoped listing, -properties, idempotent mutation, and empty-only deletion. Storing a nested child -array in one catalog value is unbounded, while using names as authority makes stale -mappings, deletion races, and future table moves ambiguous. - -R177 resolves multipart support, property limits, stale-mapping filtering, and the -drop fence. R178 supplies the active catalog and storage envelope. This requirement -turns those decisions into a stable NamespaceId authority and the standard REST -surface. - -## Solution - -- **NS-I1 — Stable identity:** NamespaceId survives property changes and is never - derived from its identifier. -- **NS-I2 — Ordered index:** name mappings support bounded parent-scoped scans but - are not authority. -- **NS-I3 — Empty drop:** a namespace cannot be tombstoned while a valid child - namespace or table can still be created or resolved beneath it. -- **NS-I4 — Bounded namespace:** identifier, properties, pages, retries, and stale - filtering all obey explicit limits. - -1. Add `namespace/id.rs`, `key.rs`, `record.rs`, `repository.rs`, and - `wire.rs`. Encode multipart identifiers as a sequence of length-delimited UTF-8 - components with maximum levels and total encoded bytes; accept the advertised - separator and legacy unit separator at the REST boundary. -2. Store an ordered parent/name mapping to NamespaceId and a separate authority - containing the canonical identifier, authority epoch, lifecycle, and bounded - properties. Validate mapping CatalogId, NamespaceId, and epoch against the - authority on load and list. -3. Implement list, create, load, exists, property update, and drop endpoints from - the backed-up OpenAPI. Namespace rename is unsupported and unadvertised. -4. Enforce R177's property contract: 256 entries, 1 KiB key, 8 KiB value, 64 KiB - encoded authority, UTF-8 without NUL. Apply removals and updates atomically; - duplicate keys across both sets return 422. -5. List direct children only. Scan mappings with bounded over-fetch, validate - targets in bounded batches, omit stale mappings, and encode catalog, parent, - parameters, and last scanned key into an authenticated opaque continuation - token. Concurrent mutations have page-relative rather than global-snapshot - visibility. -6. Drop CASes the authority from `Ready` to `Dropping`, which fences namespace and - table creation. It then performs bounded first-entry probes in both child index - ranges. A non-empty result restores `Ready`; an empty result tombstones the - authority and removes the mapping through a recoverable operation record. -7. Persist idempotency identity, request digest, phase, and result for create, - property update, and drop so another Access Server can resume after response - loss. Repair stale mappings asynchronously with bounded work. - -## Dependencies - -- Depends on R177 and R178 for active context, key/value envelope, request identity, - and error mapping. -- Produces NamespaceId, name mapping, authority epoch, lifecycle fence, and listing - contracts consumed by R181 and R182. -- Table-child probes become effective when R181 lands. Until then that range is - empty by construction; the key range is reserved here. -- R185 may cache mappings and authorities but cannot alter list or drop semantics. - -## Acceptance - -- Given identifiers at every level and byte boundary plus malformed separators, - when they are encoded and decoded through REST and storage codecs, assert valid - identifiers round-trip and invalid ones fail before mutation. Invariant: NS-I4. - Unit test. -- Given concurrent creates with the same identifier and response-loss retries, when - operations finish on different instances, assert one NamespaceId is visible and - identical request identities return one result. Invariants: NS-I1 and NS-I2. - Integration test. -- Given properties at entry and byte limits plus overlapping removal/update keys, - when property update runs, assert the complete valid change is atomic and the - overlap returns 422 without changing the authority. Invariant: NS-I4. E2E test. -- Given stale, corrupt, and current mappings across multiple scan pages, when a - client lists a parent with continuation tokens, assert only direct current - children are emitted, work per page is bounded, and tokens cannot cross catalog - or parameter contexts. Invariants: NS-I2 and NS-I4. Integration test. -- Given concurrent child creation and namespace drop, when the drop fence is - installed at any crash point, assert either the child is valid and drop returns - not-empty or the namespace tombstones and no child becomes visible beneath it. - Invariant: NS-I3. Integration test. -- Given official REST clients invoking every declared namespace endpoint, when - success, not-found, conflict, not-empty, and pagination cases execute, assert - status and error payloads match the OpenAPI. Invariant: NS-I2. E2E test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-iceberg --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` diff --git a/doc/backlog/R180-access-iceberg-fileio.md b/doc/backlog/R180-access-iceberg-fileio.md deleted file mode 100644 index 44af195e8..000000000 --- a/doc/backlog/R180-access-iceberg-fileio.md +++ /dev/null @@ -1,115 +0,0 @@ - - - -### R180: access server / Iceberg — Native immutable files and FileIO - -## Problem - -Iceberg metadata, manifest lists, manifests, data files, delete files, deletion -vectors, and statistics files are immutable objects with format-specific range-read -patterns. General S3 authority permits overwrite and delete that Iceberg cannot -permit. Storing complete files, chunk vectors, footers, or metadata graphs in -Chunk-KV would also make records and reads unbounded. - -R177 requires native file authority, an Iceberg-owned S3-shaped surface, durable -multipart in the first writable milestone, and generation-local metadata -projections. This requirement implements that foreground storage contract. R183 -owns physical reclamation. - -## Solution - -- **FILE-I1 — Byte immutability:** a published location resolves forever to one - FileId, length, digest, and byte sequence. -- **FILE-I2 — Bounded records:** a file record contains bounded inline bytes or one - bounded chunk root plus fixed-size hints, never a growing chunk vector or footer. -- **FILE-I3 — Streaming scale:** upload, download, range read, multipart completion, - Avro decode, and format probing use bounded windows independent of file length. -- **FILE-I4 — Native authority:** the Iceberg FileIO surface authorizes a table - prefix and immutable files; it never reads or writes general S3 metadata. -- **FILE-I5 — Canonical fallback:** projections and format hints may avoid work but - canonical bytes are the only file authority. - -1. Add `file/id.rs`, `key.rs`, `record.rs`, `repository.rs`, `writer.rs`, - `reader.rs`, `location.rs`, `multipart.rs`, and `s3_compat.rs`. A table location - is `s3://iceberg-/t//`; an exact client-created - relative key below that prefix maps once to a server FileId. Names and namespace - paths never enter the location. Reject bucket, table, or path escape and never - normalize two different S3 keys into one identity. -2. Store metadata JSON, manifest lists, and manifests with an inline-or-chunk - variant. Stored inline payload is at most 16 KiB; only original input at most - 64 KiB may be tested for LZ4 compression. Data, position/equality delete, - deletion-vector, and statistics files always use chunk storage regardless of - size. -3. Publish only after complete bytes, digest, length, file kind, content format, - and fixed-size format hint are verified. A retry of the same location with the - same digest returns the existing result; different bytes return conflict. - Published overwrite is impossible. -4. Implement immutable PUT, HEAD, and GET with one contiguous range. PUT streams - directly into bounded chunk writers; GET retains one FileRecord and applies - response credits so slow clients bound prefetch. Unsupported S3 operations - return stable S3-shaped errors without mutation. -5. Implement durable create, upload-part, list-parts, complete, and abort multipart - state. Bound sessions, parts per session, part bytes, aggregate staged bytes, - TTL, reconciliation pages, completion work, and retries. Complete publishes one - immutable file or returns the prior result; abandoned parts are R183 candidates. -6. Issue short-lived delegated credentials restricted to CatalogId, TableId, exact - operation set, table prefix, byte limits, expiry, and nonce. File DELETE is not - delegated and has no public S3 route; only R183 can authorize physical removal. -7. Add `metadata_projection/` with generation-local root, bounded pages, and child - JSON objects qualified by TableId, metadata generation, JSON digest, and - projection version. Missing, partial, corrupt, or oversized projections fall - back to byte-identical metadata JSON and never block publication. -8. Stream manifest lists and manifests by Avro blocks. Validate v1/v2/v3 inheritance - rules, sequence and row-ID fields, data/delete content, deletion-vector - descriptors, and metrics without building an unbounded entry vector. -9. Store only fixed-size Parquet, ORC, Avro, and Puffin location hints verified at - seal time. Invalid hints trigger bounded probing of canonical bytes. Parsed - footers, stripe directories, block directories, and pages are memory-only R185 - cache entries until a separate measured requirement approves persistence. - -## Dependencies - -- Depends on R177 and R178 for identity, capability, key/value, active context, - authentication, and chunk clients. -- Supplies immutable file identities, canonical locations, metadata projection, - and delegated-access contracts to R181, R182, R183, and R184. -- R183 owns staged, orphan, expired, and unreachable physical cleanup. Before R183, - such data may leak but can never become visible through a published location. -- R185 owns decoded caches. All reads remain correct when every cache is disabled. - -## Acceptance - -- Given inline boundaries at 16 KiB and compression-input boundaries at 64 KiB, - when compressible and incompressible metadata, manifest, data, delete, deletion - vector, and statistics files are written, assert the required variant is selected - and every read returns identical bytes. Invariants: FILE-I1 and FILE-I2. Unit test. -- Given files spanning chunk boundaries and clients with arbitrary backpressure, - when full and one-range GETs run, assert returned bytes and status are correct and - retained memory and prefetch remain within configured windows. Invariant: - FILE-I3. Integration test. -- Given two PUTs to one location with equal or different bytes plus response loss, - when they retry across Access Servers, assert equal content returns one FileId and - different content conflicts without overwrite. Invariant: FILE-I1. E2E test. -- Given multipart upload crash points, duplicate parts, completion retries, abort, - and TTL expiry, when recovery resumes, assert at most one immutable file publishes - and all state and work stay within independent limits. Invariants: FILE-I1 and - FILE-I3. Integration test. -- Given valid, missing, partial, corrupt, wrong-version, and wrong-digest metadata - projections, when load requests need selected and complete metadata, assert valid - pages avoid full decode and every invalid case falls back to byte-identical JSON - without changing authority. Invariant: FILE-I5. Integration test. -- Given v1, v2, and v3 manifests containing sequence, row-lineage, position/equality - delete, and deletion-vector cases, when blocks stream across chunk boundaries, - assert entries follow the version rules and memory does not grow with total - entries. Invariant: FILE-I3. Integration test. -- Given official S3 FileIO behavior, when allowed operations, bucket CRUD, - overwrite, path escape, tagging, lifecycle, and DELETE are attempted, assert only - the declared table-prefix operations succeed and general S3 objects remain - isolated. Invariant: FILE-I4. E2E test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-iceberg --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` diff --git a/doc/backlog/R181-access-iceberg-table-lifecycle.md b/doc/backlog/R181-access-iceberg-table-lifecycle.md deleted file mode 100644 index 2e189983e..000000000 --- a/doc/backlog/R181-access-iceberg-table-lifecycle.md +++ /dev/null @@ -1,99 +0,0 @@ - - - -### R181: access server / Iceberg — Table metadata and lifecycle - -## Problem - -An Iceberg table is not a mutable object value. Its stable identity, current name, -selected metadata generation, immutable metadata JSON, format version, and -lifecycle must stay coherent across list, load, exists, rename, and drop. Name -mappings and multi-key rename steps can become stale after crashes, while large -metadata cannot be copied into `TableHead` or decoded without bounds. - -R177 separates CROWDB TableId from Iceberg `table-uuid`, resolves recoverable -rename/drop, and requires v1, v2, and v3. R179 supplies namespace fences and R180 -supplies immutable metadata files and projections. - -## Solution - -- **TABLE-I1 — Stable table:** TableId and Iceberg `table-uuid` do not change on - rename and are never interchangeable. -- **TABLE-I2 — Selected metadata:** `TableHead` selects one immutable metadata - generation and digest; the selected standard JSON contains complete table state. -- **TABLE-I3 — Name consistency:** load and list accept a mapping only when its - TableId, NamespaceId, canonical name, name epoch, and lifecycle match `TableHead`. -- **TABLE-I4 — Version fidelity:** v1, v2, and v3 metadata are validated and served - without dropping unknown optional fields or violating version-specific rules. -- **TABLE-I5 — Logical lifecycle:** rename and drop change visibility through - bounded durable state machines and never synchronously move or delete files. - -1. Add `table/id.rs`, `key.rs`, `record.rs`, `metadata.rs`, `repository.rs`, - `lifecycle.rs`, and `wire.rs`. Store ordered namespace/name mappings separately - from bounded `TableHead` values. -2. `TableHead` stores TableId, NamespaceId, canonical table name, name epoch, - lifecycle, metadata generation, current metadata FileId/location/digest, format - version, and operation fence. It stores no metadata JSON, snapshot graph, - manifests, or data-file children. -3. Parse and validate all mandatory table metadata, schema/type, partition, - sorting, snapshot, reference, statistics, encryption-key metadata, and - serialization rules for format v1, v2, and v3. Preserve the original immutable - JSON for full REST and FileIO responses; projections cannot re-encode authority. -4. Implement list, load, exists, rename, and drop. Support `snapshot-loading-mode` - `ALL` and `REFS` from one selected generation. Bind ETag and conditional loads to - TableId, generation, and metadata digest. -5. Rename, including a move across namespaces, reserves the destination mapping, - advances `TableHead` name epoch and canonical identifier by CAS, and tombstones - the source through a durable operation record. Source and destination namespace - lifecycle fences are checked at every transition. Reconciliation completes or - removes reservations after crashes. -6. The old name is never an alias. A known old-name cache may later produce an - authorization-filtered hint under R185, but the repository returns not-found - once the head selects the new name. List filters every stale reservation or - mapping using bounded validation. -7. Drop CASes the head into a tombstoned lifecycle and removes name visibility. - `purgeRequested=false` leaves files retained; `purgeRequested=true` schedules an - R183 proof task. Neither path traverses snapshots in request latency. -8. Reject register-table and every unadvertised endpoint. A location can enter - table authority only through the native create/commit flow in R182. - -## Dependencies - -- Depends on R177, R178, R179, and R180. -- Produces TableId, TableHead generation, name epoch, lifecycle, metadata validator, - and selected-generation load contract for R182 through R185. -- R182 owns create, staged create, and metadata updates. Tests here may install - valid fixture heads through a test utility but may not define a second publisher. -- R183 owns purge and orphan reclamation. Drop remains a correct logical operation - while physical cleanup is unavailable. - -## Acceptance - -- Given valid and invalid v1, v2, and v3 metadata with version-specific schemas, - types, snapshots, row lineage, delete representations, and serialization, when - parsed and loaded, assert valid bytes are preserved and every mandatory violation - fails closed. Invariant: TABLE-I4. Unit test. -- Given `ALL` and `REFS` loads plus conditional ETags, when one selected metadata - generation is served, assert each response is derived from that generation and a - later commit cannot mix fields into it. Invariant: TABLE-I2. E2E test. -- Given same-namespace and cross-namespace rename crashes at every transition, when - reconciliation and concurrent list/load run, assert one canonical name resolves, - the old name is not an alias, and TableId, table UUID, and file locations do not - change. Invariants: TABLE-I1, TABLE-I3, and TABLE-I5. Integration test. -- Given stale mappings, reservations, tombstones, and valid entries over multiple - pages, when list and exists run, assert only head-qualified tables are exposed and - work per page remains bounded. Invariant: TABLE-I3. Integration test. -- Given logical drop with and without purge requested plus response loss, when the - request retries, assert name visibility disappears exactly once, no foreground - snapshot traversal occurs, and purge only creates a durable R183 task. Invariant: - TABLE-I5. E2E test. -- Given register-table and other unadvertised operations, when clients call them, - assert the standard unsupported response is returned and no mapping, head, or - file authority changes. Invariant: TABLE-I2. E2E test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-iceberg --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` diff --git a/doc/backlog/R182-access-iceberg-table-commit.md b/doc/backlog/R182-access-iceberg-table-commit.md deleted file mode 100644 index d23efba02..000000000 --- a/doc/backlog/R182-access-iceberg-table-commit.md +++ /dev/null @@ -1,111 +0,0 @@ - - - -### R182: access server / Iceberg — Atomic table commits and recovery - -## Problem - -Iceberg create and update operations validate requirements against one current -metadata state, apply an ordered update list, write a new immutable metadata file, -and atomically select it. A partial implementation that ignores an unknown update, -publishes files before validation, repeats a successful mutation after response -loss, or treats compare-exchange failure as generic server error would violate the -REST and table specifications. - -R181 deliberately leaves `TableHead` publication to one owner. This requirement -implements create, staged create, v1/v2/v3 updates and upgrades, deterministic -conflicts, idempotency, and crash recovery without a table-wide lock. - -## Solution - -- **COMMIT-I1 — One input generation:** all requirements and updates in a request - are evaluated against one retained `TableHead` and metadata generation. -- **COMMIT-I2 — Ordered atomic update:** either the complete ordered update list is - selected by one head CAS or none of it is visible. -- **COMMIT-I3 — Format fidelity:** every requirement, update, inheritance rule, and - upgrade follows the selected v1, v2, or v3 specification; unknown or disabled - variants fail before candidate publication. -- **COMMIT-I4 — Retry identity:** one request identity and digest has one durable - final result across instances and response loss. -- **COMMIT-I5 — Orphan safety:** a CAS-losing or abandoned candidate is unreachable - and only becomes an R183 reclamation candidate. - -1. Add `commit/requirement.rs`, `update.rs`, `evaluator.rs`, `operation.rs`, - `create.rs`, and `repository.rs`. Wire types decode into bounded domain enums; - no unknown tagged union is ignored or passed through as JSON. -2. Implement immediate create and staged create. Reserve the table name under the - namespace fence, validate initial schema/spec/order/properties and target format, - persist immutable metadata, then publish one initial `TableHead`. Staged state is - durable, expires, and can be completed only by its bound commit identity. -3. For update, retain one head revision and canonical metadata input; validate all - requirements; apply updates in request order to a bounded builder; revalidate - the complete output; serialize one canonical standard metadata JSON file; then - compare-exchange the head from the retained revision to generation plus one. -4. Cover the complete requirement and update union needed by the backed-up OpenAPI - and table specification for v1, v2, and v3. This includes schemas and defaults, - partition specs, sort orders, properties, locations, snapshots and references, - statistics, sequence and row-ID inheritance, row lineage, delete semantics, - encryption-key metadata, and version-specific fields. -5. Support v1-to-v2 and v2-to-v3 upgrades as explicit transitions. Validate the - source before applying transition rules and validate the result under the target - version. Reject downgrades, skipped transitions, and any upgrade that would lose - active metadata semantics. -6. Classify a failed requirement, stale generation, name/lifecycle fence, duplicate - create, unsupported operation, malformed metadata, and head CAS loss into their - precise REST conflict or validation response. A CAS loser never retries against - a new generation inside the same request. -7. Persist an `OperationRecord` before mutation with request identity, canonical - digest, table/name context, input generation, phase, candidate FileId, and final - response. Phase transitions use CAS. Same identity plus a different digest - conflicts; same identity plus the same digest resumes or returns the result. -8. Bound request bytes, update and requirement counts, metadata input/output bytes, - projection work, serialization buffers, candidate writes, and concurrent commits - independently. Stream large canonical JSON where possible and fail admission - before exceeding a hard cap. - -## Dependencies - -- Depends on R177 through R181 for namespace fences, immutable files, metadata - validation, TableHead, REST error types, and request identity. -- Produces selected metadata generations, candidate/orphan records, operation - histories, and upgrade results consumed by R183 through R185. -- R183 is not required to make CAS losers safe; before it lands candidates may leak - storage but remain unreachable. -- R185 may accelerate input loading and evaluation, but every mutation still - validates the authoritative head revision before publication. - -## Acceptance - -- Given two commits based on one generation, when they publish concurrently, assert - one head CAS selects one complete output, the loser receives the precise conflict, - and no partial update is visible. Invariants: COMMIT-I1 and COMMIT-I2. Integration test. -- Given every declared requirement and update for v1, v2, and v3 plus unknown tagged - variants, when evaluated against reference fixtures, assert supported results - match the spec and unknown or disabled input fails before candidate publication. - Invariant: COMMIT-I3. Unit test. -- Given valid and invalid v1-to-v2 and v2-to-v3 upgrades, when committed, assert all - transition defaults and inheritance rules are applied, invalid or lossy upgrades - fail, and downgrade or skipped-version requests do not mutate the head. Invariant: - COMMIT-I3. Integration test. -- Given crashes at every create, staged-create, candidate-write, operation-phase, - and head-CAS boundary, when another server resumes with the same request identity, - assert one table/generation/result is visible and different input under that - identity conflicts. Invariant: COMMIT-I4. E2E test. -- Given a failed requirement, stale generation, duplicate name, lifecycle fence, - malformed metadata, unsupported update, and CAS loss, when official clients commit, - assert each receives the standard status and error type and no case is collapsed - into a successful no-op. Invariants: COMMIT-I2 and COMMIT-I3. E2E test. -- Given a candidate whose publisher loses or crashes, when load and list execute - before reclamation, assert the candidate is unreachable and the current head - still resolves to complete canonical bytes. Invariant: COMMIT-I5. Integration test. -- Given requests at every byte/count limit and over each hard cap, when commit - admission and evaluation run, assert accepted resource use stays bounded and - rejected requests leave no operation or candidate leak. Invariant: COMMIT-I2. - Integration test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-iceberg --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` diff --git a/doc/backlog/R183-access-iceberg-reclamation.md b/doc/backlog/R183-access-iceberg-reclamation.md deleted file mode 100644 index c5f54531d..000000000 --- a/doc/backlog/R183-access-iceberg-reclamation.md +++ /dev/null @@ -1,99 +0,0 @@ - - - -### R183: access server / Iceberg — Reachability and bounded reclamation - -## Problem - -Catalog clear, table purge, snapshot expiration, failed commits, staged uploads, -multipart aborts, and projection replacement all create unreachable state. Deleting -on request latency or using per-file reference counts would race retained snapshots, -branches, tags, metadata logs, readers, and crash recovery. Scanning a full table or -catalog into memory would fail at Iceberg scale. - -R177 selects generation-indexed candidates plus reachability traversal, mandatory -retention and pins, and no racing reference counts. This requirement implements the -durable background proof and deletion workflow. - -## Solution - -- **GC-I1 — Invisibility first:** physical deletion is considered only after the - owning catalog, table generation, operation, or upload state is unreachable. -- **GC-I2 — Positive proof:** a file or record is removed only after a proof against - retained metadata roots, snapshots, refs, metadata logs, operations, credentials, - leases, readers, and operator pins. -- **GC-I3 — Bounded traversal:** discovery, graph traversal, sorting, retry, and - deletion use durable continuations and independent limits. -- **GC-I4 — Restart safety:** duplicate, reordered, or resumed work can leak but - cannot erase reachable state or restore visibility. -- **GC-I5 — Foreground isolation:** cleanup has separate CPU, memory, KV, chunk I/O, - bandwidth, and concurrency admission from catalog and FileIO requests. - -1. Add `gc/candidate.rs`, `reachability.rs`, `task.rs`, `repository.rs`, - `worker.rs`, and `pins.rs`. Store tasks and generation-indexed candidate pages - under their CatalogId/TableId; do not create one key per file in Group 0. -2. Emit candidates for failed/abandoned metadata generations, expired staged table - creates, multipart sessions and parts, orphan projections, expired snapshots, - purge-requested dropped tables, and retired catalog ranges. Candidate creation - never performs physical deletion. -3. Traverse standard metadata JSON, metadata logs, retained snapshots and refs, - manifest lists, manifests, data/delete files, deletion vectors, and statistics - files according to the owning format version. Spill bounded sorted mark pages to - durable task state instead of retaining the graph in memory. -4. Compare candidate pages with the retained mark set under a captured table or - catalog fence. Revalidate the fence, retention deadline, active operations, - delegated credentials, reader leases, and operator pins immediately before - scheduling deletion. -5. For catalog clear, wait for R178's maintenance publication, lease-plus-grace - completion, minimum retention, and pins; scan the retired CatalogId half-open - range with restartable continuations. Never scan the active range by name. -6. Delete file records and chunk roots idempotently only after proof. Delete derived - projections before or with their owning unreachable generation. A partial chunk - failure leaves durable retry state and never reconstructs a removed authority. -7. Expose pause, resume, inspect, pin, unpin, rate, progress, stalled reason, and - retry controls. Validate every configured item, byte, time, and concurrency cap; - use bounded exponential backoff and terminal quarantine for repeated corruption. - -## Dependencies - -- Depends on R177, R178, R180, R181, and R182 for all roots, locations, pins, - candidates, lifecycle fences, and v1/v2/v3 reachability semantics. -- Snapshot-expiration commits remain R182 mutations; this requirement performs only - the resulting physical cleanup. -- R184 exposes only authenticated operator status/control, not a public object - delete endpoint. -- R185 cache entries and invalidation never constitute reachability. R183 waits for - the durable lease boundary defined in R177, not for physical cache eviction. - -## Acceptance - -- Given retained v1, v2, and v3 snapshots, branches, tags, metadata logs, data and - delete files, deletion vectors, and statistics, when reachability runs, assert all - referenced files are marked and no task memory or KV value grows with the graph. - Invariants: GC-I2 and GC-I3. Integration test. -- Given a failed commit candidate, expired stage, aborted multipart upload, and - orphan projection, when cleanup runs after deadlines, assert only unreachable - state is removed and repeated execution is idempotent. Invariants: GC-I1 and - GC-I4. Integration test. -- Given a drop with purge and concurrent reader, credential, commit operation, and - operator pin, when each fence expires or releases in every order, assert deletion - starts only after the last valid fence and never affects the reader's bytes. - Invariant: GC-I2. E2E test. -- Given clear of a catalog containing billions of simulated keys across partitions, - when workers crash and resume, assert foreground clear does not scan children, - continuation makes progress, every batch stays bounded, and the new CatalogId is - untouched. Invariants: GC-I3 and GC-I4. Integration test. -- Given cleanup saturation and simultaneous namespace, commit, and FileIO load, - when resource limits are reached, assert cleanup throttles or pauses while - foreground admission retains its configured budget. Invariant: GC-I5. Integration test. -- Given a corrupt manifest, digest mismatch, missing candidate page, and repeated - chunk-delete error, when workers process them, assert they fail closed into - inspectable retry or quarantine state without guessing reachability. Invariants: - GC-I2 and GC-I4. Integration test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-iceberg --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` diff --git a/doc/backlog/R184-access-iceberg-rest-conformance.md b/doc/backlog/R184-access-iceberg-rest-conformance.md deleted file mode 100644 index c51c4ea93..000000000 --- a/doc/backlog/R184-access-iceberg-rest-conformance.md +++ /dev/null @@ -1,104 +0,0 @@ - - - -### R184: access server / Iceberg — REST integration and core conformance - -## Problem - -Component repositories can be locally correct while the public catalog remains -incompatible: `/v1/config` may advertise unimplemented routes, identifiers may be -decoded differently between handlers, error types may not match the OpenAPI, -authentication may disclose renamed resources, and a real Spark, Flink, Trino, or -Iceberg client may exercise a different sequence from unit tests. - -R178 through R183 define the native authority and operations. This requirement -owns the single public REST composition, capability discovery, common protocol -behavior, and conformance evidence for the first usable milestone. - -## Solution - -- **REST-I1 — Honest discovery:** `/v1/config` advertises exactly the enabled and - verified endpoint and format capability set. -- **REST-I2 — One protocol boundary:** all handlers share bounded decoding, - authentication, authorization, request identity, error serialization, admission, - deadlines, cancellation, and metrics. -- **REST-I3 — Standard semantics:** declared behavior matches the backed-up OpenAPI - and v1/v2/v3 table spec rather than one client implementation's quirks. -- **REST-I4 — Failure isolation:** invalid, unauthorized, oversized, timed-out, or - cancelled requests do not leave an ambiguous mutation. -- **REST-I5 — Interoperability:** official clients can create, evolve, write, commit, - load, time-travel, read, rename, expire, and drop core tables through CROWDB. - -1. Complete `app/crowdb-access-server/src/iceberg/` routing and - `lib/crowdb-access-iceberg/src/rest/`. Use one generated-or-verified wire schema - model tied to the backed-up OpenAPI; domain repositories never parse raw HTTP. -2. Advertise config, namespace CRUD/properties/exists, table list/create/load/update/ - drop/exists/rename, credentials, and metrics only when their requirements and - runtime dependencies are enabled. Do not advertise register-table, views, - transactions, or scan planning. -3. Implement common decoding for prefix, multipart namespace, table identifier, - pagination, idempotency key, data-access, snapshot-loading-mode, ETag, warehouse, - and purge parameters. Enforce header, URI, query, JSON, and response bounds - before allocating domain work. -4. Apply configured bearer/OAuth authentication before namespace or table lookup and - authorize each catalog, namespace, table, management, credential, and file - action separately. Only advertise the token endpoint if token issuance is - configured and implemented. Error details never disclose a destination rename, - location, credential, or existence to an unauthorized principal. -5. Map domain outcomes to the exact standard status and Iceberg error type. Preserve - conflict categories needed for client retry; never turn unknown updates, - unsupported operations, corruption, or expired authority into success. -6. Add a conformance harness that runs the Apache REST Compatibility Kit, official - Java and Rust clients, and supported Spark, Flink, and Trino smoke profiles - against one and multiple Access Servers with fault injection. Treat the backed-up - specs as authority when test oracles disagree. -7. Publish an executable v1/v2/v3 capability matrix. Cover create/read/write and - v1-to-v2/v2-to-v3 upgrades with version-specific fixtures, including row-level - deletes, row lineage, deletion vectors, defaults, types, statistics, and format - encodings required by the declared profile. -8. Add protocol metrics for endpoint, outcome class, latency, admitted bytes, - response bytes, retry/conflict class, and selected format version without logging - credentials, payloads, or unbounded identifiers. - -## Dependencies - -- Depends on R177 through R183. R184 is the integration gate for the core - correctness milestone. -- Reuses the Access Server HTTP runtime and authentication infrastructure but keeps - an independent listener, routes, admission budgets, metrics, and shutdown drain. -- R185 is deliberately not a dependency. Conformance must pass with caches disabled. -- Client/version combinations selected for release must be pinned in the test - environment; oracle updates do not silently change the specification contract. - -## Acceptance - -- Given every enabled and disabled endpoint combination, when `/v1/config` is - queried and each route is called, assert discovery lists exactly callable routes - and unadvertised routes return unsupported without mutation. Invariant: REST-I1. - E2E test. -- Given malformed identifiers, separators, tokens, headers, JSON unions, oversized - bodies, deadlines, and cancellation at mutation crash points, when requests run, - assert common errors are stable and durable operations are absent or recoverable. - Invariants: REST-I2 and REST-I4. E2E test. -- Given principals with catalog, namespace, table, file, management, and no access, - when all route classes and rename hints are exercised, assert only authorized - information and credentials are returned. Invariant: REST-I2. E2E test. -- Given the Apache compatibility kit and official Java and Rust clients, when the - declared endpoint matrix runs against multiple servers with response loss, assert - standard successes, conflicts, retries, pagination, and errors pass. Invariants: - REST-I3 and REST-I5. E2E test. -- Given supported Spark, Flink, and Trino profiles, when each creates, evolves, - writes, commits, loads, time-travels, reads row-level deletes, renames, expires, - and drops tables, assert results agree across engines and remain valid after an - Access Server restart. Invariant: REST-I5. E2E test. -- Given v1, v2, and v3 fixture matrices and valid upgrades, when differential tests - run against reference implementations, assert metadata and visible rows agree; - any oracle disagreement is resolved against the backed-up spec and recorded in - the fixture. Invariant: REST-I3. Integration test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-iceberg --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` diff --git a/doc/backlog/R185-access-iceberg-cache-invalidation.md b/doc/backlog/R185-access-iceberg-cache-invalidation.md index 58289639e..cf2f3d28d 100644 --- a/doc/backlog/R185-access-iceberg-cache-invalidation.md +++ b/doc/backlog/R185-access-iceberg-cache-invalidation.md @@ -17,7 +17,7 @@ multiply memory budgets and stale-data rules. Cross-server notification can redu staleness but cannot be authority because instances disconnect, register late, and receive duplicated or reordered messages. -R177 resolves lease-plus-grace clear semantics and requires per-class limits chosen +The native Iceberg design resolves lease-plus-grace clear semantics and requires per-class limits chosen by focused benchmarks. This requirement adds one Iceberg-owned cache manager while preserving correct behavior when notifications or the complete cache are disabled. @@ -63,8 +63,12 @@ preserving correct behavior when notifications or the complete cache are disable idempotently. Older generations are ignored. Table rename converts an existing old-name entry to an authorization-neutral tombstone; only request-time current authorization may disclose the destination. -8. Integrate clear with R177: notification prompts eviction, but completion waits - for the maximum root lease and admitted/delegated grace. Retired-catalog GC uses +8. Integrate clear with the native Iceberg contract: notification prompts eviction, but completion waits + for R178's persisted maintenance deadline and admitted/delegated grace. Start + lease age before the authoritative root read, never when a delayed reply arrives; + maintenance prevents fresh leases, while an existing lease may admit old-context + requests only until its original expiry. New-domain admission opens after the + grace proof. Retired-catalog GC uses durable fences and never waits for physical cache eviction acknowledgements. 9. Expose per-class hit, miss, stale, fill, bypass, bytes, entries, eviction, expiry, rebuild, notification, fanout, drop, and refresh-failure metrics. Benchmark hit @@ -106,6 +110,10 @@ preserving correct behavior when notifications or the complete cache are disable instances, and a late server, when fanout runs, assert work remains bounded, newest generations converge, committed mutation latency is unaffected, and the late server loads authority before readiness. Invariant: CACHE-I4. E2E test. +- Given a disconnected old-root lease holder and delayed cache fills, when clear + enters maintenance and publishes a new pointer, assert no lease extension and + no old-context response after the persisted completion boundary, even across a + recovering clear coordinator. Invariants: CACHE-I3 and CACHE-I4. E2E test. - Given concurrent hits, fills, invalidations, expiry, and eviction across classes, when contention benchmarks run, assert hot lookup takes no global lock and its latency is independent of unrelated class activity. Invariant: CACHE-I5. diff --git a/doc/backlog/R186-access-iceberg-orc-validation.md b/doc/backlog/R186-access-iceberg-orc-validation.md new file mode 100644 index 000000000..a2c64ef14 --- /dev/null +++ b/doc/backlog/R186-access-iceberg-orc-validation.md @@ -0,0 +1,68 @@ + + + +### R186: access server / Iceberg — Selected ORC validation + +## Status + +Retained by user decision as an independent, unimplemented ORC follow-up. The +Parquet catalog path is complete. R189 owns container client and engine workflows +and does not absorb this requirement; ORC does not block the container or client +ecosystem acceptance. Selected-file validation continues to reject ORC explicitly +rather than representing an unchecked file as validated. + +## Problem + +Native FileIO can retain immutable ORC bytes, but container recognition does not +prove selected schema, row counts or delete semantics. Accepting a container hint +as proof would violate the canonical-file contract in +[the root design](../design/access-server/iceberge/design-crowdb-iceberg.md). +An official client configured to write ORC needs a distinct, tested validation +capability rather than an implicit fallback to Parquet checks. + +## Solution + +1. Extend `crowdb-access-iceberg::file` with bounded canonical ORC decoding. Use + the official ORC protobuf and pinned Iceberg mappings and SDK as authorities; + decode compression framing correctly and reject unsupported codecs or + encryption explicitly. Independently limit encoded/decoded bytes, protobuf + work, type depth/count and stripe count. +2. Extend `crowdb-access-iceberg::manifest` selected-file validation with field-ID, + historical schema, row-count and supported delete checks. Bind the manifest's + semantic kind without changing immutable file authority. Never trust hints + instead of canonical bytes. +3. Integrate ORC into complete selected snapshot validation only after its format + gates pass. Failure or cancellation publishes no validation result or head. + Keep unsupported selected ORC explicit before this capability lands; ordinary + immutable upload is not a table-selection or commit proof. + +## Dependencies + +- R180 supplies immutable files and canonical range reads. +- R181/R182 supply trusted table metadata and selected snapshot validation. +- Extend the completed REST/SDK conformance profile with ORC after this + requirement passes; initial Parquet-only acceptance is already complete. +- R185 caches are optional; uncached canonical reads remain correct. + +## Acceptance + +- Given official SDK ORC fixtures and supported compression variants, when + canonical metadata is decoded, assert schemas and row counts match the writer. + Invariant: canonical authority. Integration test. +- Given malformed framing, excessive expansion, deep types, invalid stripes, + encryption or unsupported codecs, when decoded, assert bounded failure with no + successful proof. Invariant: independent resource limits. Unit test. +- Given data and delete manifests with historical schemas and false counts, when + selected, assert compatible files pass and mismatched kind, identity, fields or + counts fail. Invariant: manifest-to-file binding. Integration test. +- Given a complete snapshot containing ORC and a failure or cancellation during + validation, when publication is attempted, assert no head advances; before ORC + support, assert selection fails explicitly. Invariant: fail-closed publication. + Integration test. + +Required gates: + +- `pixi run -- cargo test -p crowdb-access-iceberg --all-targets` +- `pixi run -- cargo test -p crowdb-access-server --all-targets` +- `pixi run rs-fmt-check` +- `pixi run rs-lint` diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md new file mode 100644 index 000000000..6b6650239 --- /dev/null +++ b/doc/backlog/R188-console-group0-authority.md @@ -0,0 +1,157 @@ + + + +### R188: console — Group 0 authority and deployment configuration cleanup + +## Problem + +The unreleased `ConsoleConfig` in +[`design-crowdb-console.md`](../design/console/design-crowdb-console.md) +currently mixes cluster topology with host, SSH, binary, port, PID, and local +launch information. CLI and bare-metal Web can write the local file before a +Group 0 mutation succeeds, and some reads and restart paths still accept that +file or a monitor cache as topology authority. A second console can therefore +observe a different cluster, while an outage can resurrect stale topology. + +R187's Docker monitor already owns container process supervision; Group 0 owns +CROWDB system metadata, not container IDs, images, mounts, PIDs, restart +generations, or machine-local launch policy. Completing a cross-mode console +rewrite is not a prerequisite for packaging that monitor and the single-node +profile. The remaining boundary cleanup belongs in this separate requirement. + +At the user's request, this requirement also owns the deferred container crash +diagnostics work: core collection, bounded retention and source-line +symbolization. Existing crash recovery is implemented, but usable diagnostic +dumps depend on the host collector and exact-build symbols. This follow-up does +not block R187 completion. + +## Solution + +1. Keep Group 0 as the durable authority for CROWDB hardware hierarchy, + ownership and binding maps, KV store/group/replica metadata, and service + registration. Do not add Docker or bare-metal process deployment records to + Group 0. Docker process state comes from `crowdb-monitor`; bare-metal launch + policy remains local. Deployment mode changes which lifecycle and hardware + controls are allowed, not the meaning of Group 0 records. +2. Replace the mixed `ConsoleConfig` persistence in + `crowdb-console-shared::config`, `crowdb-web`, and `crowdb-cli` with a + versioned Web process configuration and a separate bare-metal launch-only + registry. Finish wiring the existing `LaunchRegistry` parser to actual + bare-metal deploy/restart operations; remove the unreleased mixed + parser/writer, topology fields, restore path, fixtures, and fallback rather + than adding a compatibility reader. Retain SSH credential references, + binary/config paths, workspace, and auto-start policy locally; never persist + inline secrets or runtime PID as topology. Docker Web rejects a launch + registry and keeps its monitor-owned process path. +3. Unify CLI and bare-metal Web hardware mutations through Group 0-backed + operations in `crowdb-console-shared::ops::hardware`. Confirm writes before + updating a read model; preserve conflicts and uncertain results. Docker Web + continues to reject hardware and process mutations. +4. Complete the common Group 0-backed logical store/group/replica flow in + `crowdb-console-shared::ops::kv_logical` for CLI and both Web modes. Reconcile + lost responses by reading confirmed authority, test multi-node fan-out and + rollback, and remove local logical-topology commits. A failed node-side + deletion must not erase surviving Group 0 membership. +5. Replace config-backed monitor refresh, KV endpoint fallback, and + physical/deployment topology reads with Group 0 membership and live service + registration. A missing or ambiguous live endpoint fails unavailable; a + stopped service may still have local launch policy but is not reported as + live. Neither a local launch registry nor a monitor cache is an authority + fallback during a Group 0 outage. +6. Keep pre-Group-0 bootstrap intent separate. After creation, verify every + committed hardware and logical record and delete the local topology copy. + Persist enough bootstrap identity to resume an interrupted transfer, prove + already committed content, and reject conflict. Destroy/clean must use + confirmed Group 0 state. If nonmember KV processes were launched before + Group 0 exists, propagate usable Group 0 seed hints after initialization + before treating their registration as live; seed hints are not topology. +7. Audit the S3 mini-cluster's local `console.toml` and restart path under the + same authority boundary. Retain only launch inputs and bootstrap seeds + locally after Group 0 cutover; do not replay a local topology copy. +8. Migrate the verified bare-metal deployment and operations material from the + old `doc/user-manual/user-guide.md` into + `doc/user-manual/bare-metal-user-guide.md`, organized by KV cluster, chunk + layer, and data access servers. State that bare-metal is not yet + production-ready. Remove the old combined guide only after its supported + material and links are migrated; Docker documentation remains independent. +9. Complete container crash diagnostics without changing host-wide collector + policy. Respect file-based core patterns, Ubuntu Apport, systemd-coredump and + Docker Desktop's Linux VM; document where dumps actually go or why collection + is unavailable. Where file dumps are supported, retain them in a bounded, + private data-volume location. Provide an exact-build source-line + symbolization workflow for child and monitor crashes. Dumps can contain + secrets and user data; diagnostics must not expose them in ordinary logs. + Host acceptance and symbol-distribution choices remain open in the execution + plan; no image-size increase or host configuration change is assumed. + +## Dependencies + +- R187 provides the working single-node Docker profile, monitor-owned process + state, managed Web baseline, and Group 0-backed system metadata. R187 image + verification does not depend on this cross-mode cleanup. +- The existing Group 0 schema and `crowdb-kv-client` service APIs remain the + authority. If a live registration is absent, operations fail unavailable or + wait for registration; local launch policy never substitutes for it. +- The old mixed console file is unreleased. No on-disk compatibility promise or + migration tool is required, but bootstrap replay must not overwrite a + confirmed initialized cluster. + +## Acceptance + +- Given a Docker process restart and a bare-metal process restart, when runtime + state is queried, assert Docker PID/restart state comes from the monitor and + bare-metal launch policy stays local, while neither appears as Group 0 + topology. Invariant: deployment state is not sysdata. Integration test. +- Given a mixed legacy config and valid/invalid launch registries, when Web and + CLI start, assert only versioned process and launch inputs are accepted, no + local topology is restored, Docker rejects the registry, and inline secrets + or topology fields fail validation. Invariant: separated configuration. + Integration test. +- Given two bare-metal consoles and one ready Group 0, when each mutates racks, + nodes, disk groups, or disks and a write conflicts or loses its response, + assert both read one confirmed result and neither commits a local-first + topology change. Invariant: hardware authority. Integration test. +- Given CLI, Docker Web, and bare-metal Web with the same Group 0, when each + performs authenticated logical store/group/replica operations, assert one + shared result, correct fan-out/rollback, and no local logical copy. + Invariant: common logical authority. Integration test. +- Given missing, duplicated, or expired registrations and then a Group 0 + outage, when topology, endpoint, or deployment status is read, assert no + stale local endpoint or monitor snapshot is presented as authoritative. + Invariant: fail-closed discovery. Integration test. +- Given a crash before and after each bootstrap commit and before local + deletion, when startup resumes, assert it proves identity and committed + content, writes only safely missing records, and rejects conflict without + overwriting Group 0. Invariant: replay-safe cutover. Integration test. +- Given nonmember KV processes launched before Group 0 initialization, when + Group 0 is created and seed hints are propagated, assert each process + registers exactly one live node identity before logical operations use it. + Invariant: registration readiness. E2E test. +- Given a persisted S3 mini-cluster and a Group 0 outage, when it restarts or + tears down, assert local launch data cannot recreate or mask old cluster + topology. Invariant: no secondary authority. Integration test. +- Given the two deployment guides and a reader following bare-metal steps, + when the reader deploys KV, chunk services, and Iceberg or S3 access servers, + assert each layer has a verified setup and health check, the non-production + boundary is explicit, and no link targets the removed combined guide. + Invariant: deployment guidance follows its implementation. E2E test. +- Given a disposable container on a supported file-based core collector, when + a child or PID 1 crashes, assert the dump has private ownership, bounded + retention and cleanup, and resolves to source lines using exact-build symbols. + Assert ordinary logs disclose no dump contents or credentials and the + container does not change host-wide collector policy. Invariant: private, + bounded and reproducible crash diagnostics. E2E test. +- Given Apport, systemd-coredump or Docker Desktop collector policies, when + crash collection is attempted, assert the documented host export workflow + locates the dump or explicitly reports unsupported collection, without + claiming an absent data-volume core. Invariant: truthful collector boundary. + Integration test. + +Required gates: + +- `pixi run clean-env && pixi run test-console` +- `pixi run clean-env && pixi run test-console-ui` +- `pixi run test-monitor` +- `pixi run test-single-node-container` +- `pixi run rs-fmt-check` +- `pixi run rs-lint` diff --git a/doc/backlog/R189-access-iceberg-container-ecosystem.md b/doc/backlog/R189-access-iceberg-container-ecosystem.md new file mode 100644 index 000000000..2cb721da9 --- /dev/null +++ b/doc/backlog/R189-access-iceberg-container-ecosystem.md @@ -0,0 +1,138 @@ + + + +### R189: access server / Iceberg — Container client and engine workflows + +## Status + +Ready after R187 local single-node image verification. This is a +separate client-ecosystem project, not a gate for publishing the non-production +Docker preview and is independent of the completed REST/official-SDK acceptance. + +## Problem + +The completed REST conformance work proves the declared REST protocol with official SDKs and a supported +subset of the Apache compatibility kit. R187 proves a packaged container with +PyIceberg and S3 client fixtures. Neither proves that a developer can connect a +notebook, dataframe library, SQL engine, or distributed compute engine to the +same image and obtain correct table rows across commits and restarts. A catalog +endpoint alone is insufficient: clients may require different credential +delegation, file-location, format-version, and delete-file behavior. Advertising +untested client compatibility would mislead evaluators of the preview. + +The [access architecture](../design/access-server/design-crowdb-access-server.md) +keeps Iceberg table authority separate from general S3 buckets. This project +tests clients against that boundary rather than assuming an S3-shaped file URI +is a normal S3 object. + +## Solution + +1. Build an isolated, reproducible client harness around the R187 image and + `container/single-node-container/tests/`. Pin each client and dependency in + Pixi or a locked external image, record its exact version and operation + profile, publish only S3/Iceberg/Web ports, and use generated scoped + credentials. Tests must not modify the user's persistent volume or accept + arbitrary external object locations as native Iceberg files. +2. Prove the Python notebook/dataframe path first: PyIceberg discovers the + REST catalog, writes a small Arrow/Parquet-backed table through its supported + FileIO, and reads selected rows and a prior snapshot into Arrow batches and + pandas. Test Polars through PyIceberg's conversion separately. Include a + notebook-style analysis and a batch-oriented ML/data-processing consumer; + distinguish eagerly materialized dataframes from a bounded Arrow batch + reader. Keep version and feature claims limited to pinned passing fixtures. +3. Test a local SQL path with DuckDB's Iceberg REST catalog integration, not + only a static metadata-file scan or a PyIceberg-to-DuckDB in-memory copy. + Verify catalog discovery, delegated file access, SELECT and a supported + write if the pinned client and current CROWDB profile permit it. If its + required FileIO or authentication contract is unsupported, record the exact + first divergence and a separate implementation dependency; do not expose a + misleading success recipe. +4. In the separately provisioned engine project, pin Spark, Flink and Trino + profiles against the same container and run the capabilities each client + actually supports: namespace/table lifecycle, append/read, schema or + partition evolution, snapshots/time travel, row-level deletes, and restart. + Compare one engine's committed rows with another engine and PyIceberg; do + not claim an engine or format version compatible from catalog-only tests. + Failures must identify REST, FileIO, format, credential or client behavior + without weakening the server's authority and durability contracts. + Exercise a cross-tool handoff where one client writes, another reads, and a + third verifies the same selected snapshot after a container restart. +5. Evaluate streaming/ingest as a later scenario, starting with the official + Iceberg Kafka Connect sink only after its pinned connector can use the + supported REST and FileIO profile. Record its setup and first divergence + separately; Kafka infrastructure is not required for the Python/SQL/engine + acceptance above. Exercise a simple BI query through a tested SQL engine; + do not claim direct BI-tool or catalog compatibility without its own fixture. +6. Add only passing, reproducible recipes to + `doc/user-manual/docker-single-node-user-guide.md`. Maintain a client + capability matrix with tested versions, read/write scope, known exclusions, + and links to executable fixtures. Label the image and all examples as + development/test, not production data storage or upgrade-stable service. + A direct Parquet file read or generic S3 object operation does not establish + Iceberg catalog, snapshot, or table-row compatibility. + +## Dependencies + +- R187 supplies the image, volume/port contract, monitor, credentials, and + container test baseline. R184 supplies REST/SDK conformance and the declared + capability profile. This requirement does not block either one's completion. +- R177's deferred cross-engine acceptance moves here. R186 ORC, R185 caching, + production durability, and general S3 bucket semantics are not prerequisites; + unsupported operations remain explicitly excluded from published recipes. +- The official [PyIceberg API](https://py.iceberg.apache.org/api/) documents + Arrow batches and pandas conversion; the official + [DuckDB catalog guide](https://duckdb.org/docs/current/core_extensions/iceberg/catalogs) + documents direct REST attachment. These describe client capabilities, not + proven CROWDB compatibility. Pin versions before implementation. + +## Acceptance + +- Given a clean R187 image and isolated volume, when the client harness starts + and exits, assert fixed versions, generated scoped credentials, bounded test + data, no internal port publication, complete diagnostic artifacts on failure, + and no change to another volume. Invariant: reproducible isolation. E2E test. +- Given a PyIceberg-written Parquet table with multiple snapshots, when Arrow + batches, pandas and Polars read a filtered current and historical view before + and after container restart, assert identical rows, types and snapshot + selection; the batch path does not materialize the whole result at once. + Invariant: Python dataframe correctness. E2E test. +- Given a notebook-style query and a batch-oriented downstream consumer, when + each uses the PyIceberg catalog and Arrow batch reader, assert selected rows + and types agree with the table snapshot and memory use is bounded by the + chosen batch rather than the full table. Invariant: analysis and ML consume + catalog-selected data. E2E test. +- Given a pinned DuckDB Iceberg REST profile, when it attaches CROWDB and + queries a PyIceberg-created table, assert rows and catalog identity match; if + direct attachment cannot satisfy the declared CROWDB FileIO contract, assert + the failure is classified and no direct-DuckDB recipe is published. + Invariant: direct SQL claims require proof. E2E test. +- Given pinned Spark, Flink and Trino profiles with only supported operations, + when each writes or reads and another client verifies after restart, assert + committed rows, schema, snapshots and supported deletes agree, with every + unsupported operation recorded rather than counted as a pass. Invariant: + cross-engine interoperability. E2E test. +- Given a table written by one client and changed by another, when a third + client reads before and after container restart, assert the selected snapshot + and visible rows agree across clients; reading the underlying Parquet file + alone is not counted as a catalog pass. Invariant: cross-tool handoff follows + Iceberg authority. E2E test. +- Given a candidate Iceberg Kafka Connect sink and an isolated event stream, + when its REST/FileIO handshake and one append are attempted, assert either a + verified end-to-end row result or a documented first unsupported boundary; + neither outcome blocks the core client matrix. Invariant: honest ingest + compatibility. E2E test. +- Given a verified SQL engine and a small dashboard-style aggregate query, + when the query runs against a catalog table, assert its result matches the + selected snapshot; do not claim an untested BI tool connects directly to + CROWDB. Invariant: BI recipes use a proven SQL path. E2E test. +- Given the client recipes and matrix, when a user follows each published + example against the pinned image, assert every advertised operation passes + and the non-production, no-upgrade and format limits remain visible. + Invariant: documentation follows evidence. E2E test. + +Required gates: + +- `pixi run test-single-node-container` +- `pixi run test-iceberg-container-ecosystem` (new task to add with the harness) +- `pixi run rs-fmt-check` +- `pixi run rs-lint` diff --git a/doc/backlog/R190-access-iceberg-shared-streaming-io.md b/doc/backlog/R190-access-iceberg-shared-streaming-io.md new file mode 100644 index 000000000..10b976b79 --- /dev/null +++ b/doc/backlog/R190-access-iceberg-shared-streaming-io.md @@ -0,0 +1,113 @@ + + + +### R190: access — Shared S3 and Iceberg streaming data path + +Status: Ready after R187 completion, at the user's request. Begin with a +complete read/write/delete/GC flow review before implementation. + +## Problem + +Iceberg FileIO treats each roughly 64 KiB leaf as a separately durable small +object. Each leaf registers physical ownership through six catalog reads and +one conditional write, then waits for the Chunk readable cursor. Receiving the +next leaf waits for that entire chain. Reads fetch individual leaves and copy +returned bytes. S3 already receives into 1 MiB owners, frames at 64 KiB, and uses +whole-object Chunk writers and lazy read streams. + +A measured 5 MiB release-build upload on the native null-DiskIO stack took +2.68–2.79 seconds for ordinary PUT and 2.09–2.18 seconds for UploadPart, excluding +multipart completion. These are API measurements, not NVMe throughput. + +Root designs: [Iceberg](../design/access-server/iceberge/design-crowdb-iceberg.md), +[S3](../design/access-server/s3/design-crowdb-access-s3.md), and +[Chunk I/O](../design/chunkio/design-crowdb-chunkio.md). + +## Solution + +The user-selected data path is shared streaming infrastructure with independent +S3 and Iceberg metadata semantics: + + HTTP receive owner (1 MiB) -> Chunk writer (64 KiB frames) -> DiskIO + durable complete locations -> one atomic file/part publication point + published locations -> Chunk read stream -> owner-backed HTTP response + +1. Share deferred HTTP receive-provider installation and bounded native owner + allocation. Authenticate and admit before reading bodies. Preserve signed + AWS-chunked decoding, checksums and unknown-length bounded streaming. +2. Write a whole file or multipart part through the Chunk writer selected by + object size. Do not register catalog intents or await a durable cursor per + frame. A frame is transport/integrity granularity, not a catalog transaction. +3. Publish complete immutable file or part metadata only after data completion + and validation. Keep fencing, conflicting-path rejection and ambiguous-result + resolution. Readers cannot observe partial data. +4. Stream GET and Range through the same Chunk read machinery as S3, retaining + owner-backed buffers and bounded backpressure. Keep Iceberg credentials, + generation checks, format validation, full-file integrity and GC protection. +5. Make multipart completion consume complete part references without restoring + the per-leaf write/commit path. Preserve ordering, replay, format validation + and atomic final-file visibility. +6. Retain crash-safe allocation ownership and reclamation below the per-frame + catalog path. Use durable Chunk allocation/lifecycle ownership rather than + deleting protection and assuming S3 already implements all orphan GC. + Drain submitted writes before reclaim; never free published or pinned data. +7. Review delete and GC end to end alongside reads and writes: logical + invisibility, reader pins, owner discovery, grace periods, cancelled writes, + shared ranges, compaction and physical reuse must form one coherent model. +8. Keep existing stored file descriptors readable, or implement an explicit + migration within this work; do not silently invalidate persisted volumes. + +## Dependencies + +- Existing native receive owners, prepared Chunk writers and Chunk read streams. +- Existing Iceberg file, multipart and GC contracts remain acceptance obligations. +- R168/R169/R147 contain deferred shared-storage reclamation work. Do not claim + those are implemented or weaken Iceberg recovery to bypass them; implement + any ownership support required for this path within this requirement. +- R188 remains a separate console-authority follow-up. + +## Acceptance + +- Given ordinary PUT bodies of 10 KiB, 1 MiB, 12 MiB and 100 MiB, upload + through S3 and Iceberg -> both use bounded + 1 MiB owners and 64 KiB frames; no catalog operation is issued per frame. + **Bounded shared ingress. Integration test.** +- Given signed chunks, corrupted signatures/checksums, short bodies and cancelled + requests, upload -> reject without publishing metadata; release owner credits + and drain in-flight writes. **No partial visibility. Integration test.** +- Given completed bytes, publish with a conflicting path or a lost reply -> keep + one complete authoritative outcome without overwriting different content. + **Atomic publication. Integration test.** +- Given full, cross-frame and cross-chunk ranges, read -> exact bytes, bounded + retained buffers, integrity checks and cancellation propagation. + **Shared bounded reads. Integration test.** +- Given a 100 MiB multipart upload (twenty 5 MiB parts), complete/replay/restart + -> one correct immutable file, + no per-leaf rewrite/commit loop and no premature part reclamation. + **Multipart correctness. E2E test.** +- Given process loss before/after data completion and metadata publication, + recover -> unpublished allocations remain discoverable and eventually reclaim; + published/pinned data remain readable. **Crash-safe ownership. E2E test.** +- Given published, pinned and unpublished data, delete and run GC across restart + -> logical deletion precedes physical reclamation; retained reads remain valid; + ownership is discoverable and freed ranges are not reused before writes drain. + **Delete/GC consistency. E2E test.** +- Given existing file descriptors, restart and read -> preserve bytes and ranges. + **Persisted-data readability. Integration test.** +- Given the same 5 MiB fixtures and build/storage profile, measure PUT, UploadPart + and GET -> record elapsed time and dependency-operation counts against the + baseline without raising deadlines or weakening assertions. + **Measured operation reduction. E2E test.** + +Commands: + +```sh +pixi run test-access-iceberg +pixi run test-access-s3 +pixi run test-access-server +pixi run -e s3-e2e test-boto3-e2e +pixi run -e iceberg-e2e test-iceberg-native +pixi run -e iceberg-e2e test-iceberg-sdk +pixi run rs-fmt-check +pixi run rs-lint +``` diff --git a/doc/backlog/R60-tree-scan-sibling-leaf-readahead.md b/doc/backlog/R60-tree-scan-sibling-leaf-readahead.md index b665acf55..061f5fe7a 100644 --- a/doc/backlog/R60-tree-scan-sibling-leaf-readahead.md +++ b/doc/backlog/R60-tree-scan-sibling-leaf-readahead.md @@ -121,7 +121,7 @@ bench config is a prerequisite for validation). - Readahead memory is bounded (per-scan in-flight cap, default window = 1); a full-keyspace cold scan does not grow unbounded RSS — Integration test. -- No regression on `tools/bench-kv-scan-regression.sh` (mem-mode configs +- No regression on `tools/benchmark/bench-kv-scan-regression.sh` (mem-mode configs unchanged — readahead is a no-op when leaves are resident) — Integration test. diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index e4436c9e3..6b28d7658 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -11,7 +11,14 @@ complexity, and dependency. Before implementation, follow the ## Item Index -**Next R number: R186** — Bump this line in the same commit when adding a new item. +**Next R number: R191** — Bump this line in the same commit when adding a new item. + +### Planned — Shared access streaming + +- **[R190](R190-access-iceberg-shared-streaming-io.md)** — align Iceberg PUT, + multipart and GET/Range with S3's bounded native receive and Chunk data path; + remove per-frame catalog transactions while preserving publication and recovery. + Ready for the full read, write, delete and GC review before implementation. ### Next Milestone — Chunk-backed range KV @@ -58,44 +65,36 @@ cuObject/RDMA acceleration after the TCP baseline is correct and measured. ### Planned — Native Iceberg storage -R177 is the program blueprint and resolves the shared design questions. R178 -through R184 form the correctness milestone; R185 is a later cache optimization. - -- **[R177](R177-access-iceberg-catalog-foundation.md)** — native Iceberg storage - blueprint — Area: access server / Iceberg / Chunk-KV / chunk I/O — Fix the - authority model, v1/v2/v3 core profile, program invariants, requirement order, - and all cross-cutting design decisions. -- **[R178](R178-access-iceberg-catalog-domain.md)** — catalog domain and service - foundation — Area: access server / Iceberg / Chunk-KV — Build the Iceberg - library, one active CatalogId domain, management lifecycle, key/value envelope, - server wiring, and `/v1/config` baseline. -- **[R179](R179-access-iceberg-namespace.md)** — namespace authority and REST - operations — Area: access server / Iceberg / Chunk-KV — Add stable NamespaceId, - multipart identifiers, properties, bounded listing, and fenced empty-only drop. -- **[R180](R180-access-iceberg-fileio.md)** — native immutable files and FileIO — - Area: access server / Iceberg / chunk I/O — Add immutable metadata, manifest, - data, delete, deletion-vector, and statistics files; streaming/range I/O; - durable multipart; delegated access; and metadata projections. -- **[R181](R181-access-iceberg-table-lifecycle.md)** — table metadata and lifecycle - — Area: access server / Iceberg / Chunk-KV — Add stable TableId, v1/v2/v3 - metadata validation, list/load, cross-namespace rename, and logical drop. -- **[R182](R182-access-iceberg-table-commit.md)** — atomic table commits and - recovery — Area: access server / Iceberg / Chunk-KV / chunk I/O — Add create, - staged create, complete requirements/updates, version upgrades, head CAS, - idempotency, conflict classification, and crash recovery. -- **[R183](R183-access-iceberg-reclamation.md)** — reachability and bounded - reclamation — Area: access server / Iceberg / chunk I/O — Prove v1/v2/v3 - snapshot and operation reachability before reclaiming candidates, purged tables, - staged files, or retired catalogs. -- **[R184](R184-access-iceberg-rest-conformance.md)** — REST integration and core - conformance — Area: access server / Iceberg — Compose the public REST service, - authentication, exact endpoint discovery and errors, compatibility kit, official - clients, and compute-engine smoke tests. +The native catalog correctness milestone is complete: catalog/service foundation, +namespace, immutable FileIO, atomic table commits, reclamation and REST/official-SDK +conformance. Its contract and executable profile are retained in +[Native Iceberg Storage](../design/access-server/iceberge/design-crowdb-iceberg.md). +Caches, selected ORC and container engine workflows remain separate. + - **[R185](R185-access-iceberg-cache-invalidation.md)** — bounded cache and invalidation — Area: access server / Iceberg / Group 0 / Chunk-KV — **Deferred - until R178–R184 stabilize and establish an uncached baseline.** Add one budgeted + pending focused cache measurements on the completed uncached baseline.** Add one budgeted cache manager, qualified entries, internal-RPC invalidation, and TTL safety nets. +- **[R186](R186-access-iceberg-orc-validation.md)** — selected ORC validation — + Area: access server / Iceberg — **Independent follow-up retained by user + decision; not absorbed by R189.** Add bounded canonical ORC schema, row-count + and delete validation with official-client fixtures. The Parquet catalog is + complete; ORC does not block container or client-ecosystem acceptance. +- **[R189](R189-access-iceberg-container-ecosystem.md)** — container client and + engine workflows — Area: Iceberg / clients / deployment — **Ready after local + container verification.** Verify Python dataframe, local SQL, distributed + engine and optional ingest scenarios against the single-node image; publish + only tested compatibility recipes. + +### Planned — Console authority and deployment + +- **[R188](R188-console-group0-authority.md)** — Group 0 authority and + deployment configuration cleanup — Area: console / CLI / KV — Separate bare-metal launch policy from + cluster sysdata, remove the mixed local topology fallback, and finish + cross-mode console consistency without moving Docker process state into + Group 0. + ### High Priority - **[R103](R103-chunkdb-range-migration.md)** — chunkdb range ownership diff --git a/doc/design/access-server/iceberge/design-crowdb-iceberg.md b/doc/design/access-server/iceberge/design-crowdb-iceberg.md index f24808b41..9095bbe0a 100644 --- a/doc/design/access-server/iceberge/design-crowdb-iceberg.md +++ b/doc/design/access-server/iceberge/design-crowdb-iceberg.md @@ -47,18 +47,421 @@ Standard Iceberg metadata is the recoverable table state. CROWDB may maintain derived indexes or projections for scale, but they are disposable and cannot become a second table authority. +Generation-local metadata projections preserve raw top-level JSON children in +bounded pages. REFS loads construct them only after canonical metadata passes +the complete bounded parser. A versioned validation receipt binds the selected +head, exact parser limits and projection root; a generic JSON projection alone +cannot stand in for metadata validation. REFS loads may reuse this validated +representation without decoding the complete object graph. They still read and +verify canonical storage and recheck the namespace and head before responding. +Missing, partial, corrupt, unknown-version or differently bounded projections +fall back to the canonical parser. ALL responses preserve exact canonical bytes. +Projection construction is optional and never participates in commit proofs or +head publication. No cross-generation cache or deduplication is implied. + Iceberg metadata stores bounded logical records and opaque data references. Physical chunk placement and storage topology remain below the access boundary. +### Catalog foundation + +One active root selects a random stable CatalogId and activation epoch. Display +rename updates its authority without moving descendant keys. System-scoped +management receipts, audit and retry bindings survive catalog replacement; +resource records and retained REST response bodies are catalog-scoped. + +Initialize, rename, capability activation and clear use bounded single-key CAS state machines, not a +global lock or a multi-key transaction. A root retains the operation identity +until its durable outcome and audit can be recovered by any instance. Clear +fences admission, records a maintenance observation after the durable fence, +publishes an empty replacement under maintenance, and persists the grace proof +before reopening admission. Completion uses persisted lease, request, delegated +access and clock-skew limits, never shorter restart configuration. Retired +authorities remain unreachable; physical deletion is not implemented. + +Format capability bits are durable catalog authority. Zero means no advertised +table format service; startup never rewrites or widens a legacy zero profile. +An authenticated management operation explicitly activates a validated profile +under the root fence. It preserves the catalog ID, activation epoch, table keys, +name generation and admission bounds, advances config generation, and may only +add support. A resumed operation replays its original profile and audit result. +Clear creates a new zero-profile catalog that requires separate activation. + +The baseline has no root lease. Each HTTP connection closes after five minutes +without network progress; active streamed file bodies and multipart completion +heartbeats extend the idle deadline. Request dispatch has a separate deadline +starting at connection acceptance; incomplete request headers close at that +deadline, while an active response remains governed by network idleness. REST +and FileIO admission reject configured request timeouts exceeding the persisted +catalog request bound. Newly initialized runtime catalogs use a five-minute +request bound; +new catalogs also persist a fifteen-minute delegated-access bound. Restart never +increases persisted bounds. Explicit catalog clear may expand them componentwise +under the maintenance fence and waits the resulting full grace before admission; +neither clear nor smaller restart settings can shorten existing bounds. +Listeners stop admission before bounded draining; +startup and periodic reconciliation resume interrupted management operations. + +Management, audit and shared REST retry ledgers each use 4096 deterministic +fast-hash slots with exact-identity overflow keys. An occupied slot does not +reject a different identity: it routes that identity to its own durable key. +Neither slot nor overflow records are evicted inside their retention window; +overflow storage is subject to normal disk capacity and physical reclamation. +Client identities use UUIDv7 issuance time with a 24-hour admission window and +30-second future-clock allowance. Retention starts at first admission and includes +grace. Principal, digest and catalog context must match before REST replay; +catalog replacement prevents old-body replay or rebinding. Terminal results are +immutable, while transient failures retain recoverable state. Requests without +client keys receive distinct internal identities, not cross-request deduplication. + +Immutable operation payloads use 32-KiB pages with a 2-MiB aggregate limit. Each +reference binds catalog, operation, content digest and total size; readers validate +every page and the complete digest. Small retry responses remain inline. Larger +responses publish one immutable manifest only after all pages are durable, then +complete the system retry binding. A lost reply resumes page writes or replays the +published manifest without changing the original response. + +Namespace operation journals occupy a separate key scope from HTTP responses. +They preserve request identity, principal, stable target and parent IDs, immutable +mutation snapshots, phase revisions and bounded child-probe cursors. Phase CAS +arbitrates publication versus abort; publishing cannot transition back to abort. +Snapshots cannot change after their write phase starts. Probe cursors advance +within one parent-scoped child range and reset when switching ranges. + +Authoritative namespace reads resolve each parent/name mapping against the +selected stable authority and full canonical identifier. Reservations, stale +epochs, missing targets and tombstones are not visible; corruption is an error. +Active-context checks bracket resolution so retirement cannot turn an old-domain +lookup into a response from the replacement catalog. + +Property updates persist their input and immutable before/after snapshots, then +CAS the whole namespace authority. Publication advances the property and mutation +revisions but preserves the name epoch and admission fence. An operation marker +protects uncertain publication evidence until the terminal result is durable; +another writer can finish that operation before replacing its marker. Cleanup +uses a conditional write and advances the mutation revision again. After a +definitive CAS conflict proves the input revision is obsolete, a property update +may return to preparation with fresh snapshots; unknown outcomes never take that +path. Work is bounded and exhaustion remains retryable, not a terminal conflict. +Property preparation uses the same holder-bound marker dispatcher as creation +and drop. It can finish interrupted child admission or a nonempty drop before +publishing properties; recursive helpers consume the caller's phase budget. +Namespace mutations use the shared HTTP retry ledger before executing their +durable operation driver; terminal client errors are retained alongside success. + +Namespace creation installs a recoverable parent/name reservation before a parent +authority CAS. Nested admission leaves a pending-operation marker and advances only +the mutation revision; a top-level admission conditionally validates the active +root without replacing its management operation. Helpers resolve uncertain parent +writes before allowing subsequent parent mutation. Definitive admission conflicts +may retry with fresh parent snapshots while retaining the name reservation. +Publication selects an initial authority and replaces the reservation with its +published mapping; a creation marker remains until the result is durable. Abort +records retain their exact failure outcome before conditional reservation cleanup. +Recursive creation helping shares one bounded phase budget. Root-admission +backend identities include the individual creation operation, so a different +creator cannot reuse a cached no-op CAS result from before its reservation. + +Namespace drop persists a Ready-to-Dropping fence before scanning its two child +index ranges. Durable cursors advance across bounded pages; reservations are +helped, stale namespace mappings are conditionally removed, and corruption blocks +the proof. A live child restores Ready without changing the name epoch or property +revision. Only completion of both ranges permits the fenced tombstone CAS. +Terminal replay and conditional cleanup cannot delete a recreated NamespaceId. +Table-child probes resolve published mappings against the selected table head. +Unpublished table reservations are helped through their creation or lifecycle +journal; an unadmitted creator or rename beneath the drop fence is aborted, while +an admitted publisher is completed before the parent can be fenced. Corrupt table authority +blocks the emptiness proof rather than being treated as absence. +Each listener runs a namespace-journal sweep with bounded pages, per-operation +phase budgets and a wall-clock deadline. The sweep resumes abandoned operations +and their conditional mapping cleanup without requiring a client retry. Catalog +changes invalidate its cursor; cancellation preserves durable recovery evidence. +Alternating mapping sweeps help durable reservations and conditionally remove +published bindings disproved by authoritative state. Corruption and unresolved +reservations never authorize deletion; no sweep physically removes file bytes. +The listener exposes authenticated namespace listing, load and exists routes. + +Namespace list pages scan bounded direct-child ranges and validate each published +mapping against its authority and canonical parent spelling. Reserved and stale +entries are omitted; corruption fails the page. HMAC-authenticated continuations +bind the catalog activation, stable parent identity, spelling, page size and last +scanned key. A stale-only page can therefore be empty while retaining a token. +Unpaginated lists build a complete in-memory spool before success headers, capped +independently at 2 MiB, 1024 results, 4096 scanned mappings and four concurrent +spools. Atomic admission rejects excess work without waiting. The request +deadline bounds construction before success headers. Dispatch stops before that +deadline, reserving the smaller of 100 ms or 10% of the request timeout for +emitting a bounded error response. Header receipt does not restart this budget. +A stalled transport closes after the independent idle timeout. Cancellation +drops the spool permit. Completed +responses stream in 16-KiB frames. Absent page tokens request complete results; +empty page tokens begin paginated mode. Tokens use a domain-separated signing key +derived from the configured credentials so equally configured listeners interoperate. + ## 3. HTTP and FileIO surfaces The REST Catalog is the portable control surface. It exposes only capabilities CROWDB implements with compliant Iceberg semantics. +The listener classifies complete method/path pairs before domain mutation and +uses the same fixed route set for discovery and protocol metric labels. Request +measurements use bounded atomic counters and follow response bodies through +completion or cancellation; streamed file bytes are measured when emitted. +Protocol counters never use principal, table name, token or raw path as a label. +An authenticated management-credential-only `GET /_crowdb/metrics` exposes a +bounded snapshot, including while catalog storage is unavailable. This local +diagnostic is not an Iceberg REST endpoint and is absent from `/v1/config`. + +The catalog listener exposes authenticated config and namespace REST. An absent or +empty warehouse selects the sole active catalog; other selectors fail with +`NoSuchWarehouseException`. Its endpoint list advertises installed namespace and +table read/create/commit/lifecycle/credential routes only when the persisted +format profile permits them. A zero-profile catalog returns unavailable config +rather than publishing a misleading set of false overrides; table routes reject +until management activation. Selected table versions gate load, HEAD, credential +refresh, create and commit. Version upgrades require each intermediate edge, +including direct v1-to-v3 requests. FileIO bytes alone do not identify a table +file's semantic kind or grant format-version authority. File grants intersect +the principal role with the selected version: published tables require write +support for upload permission, while staged drafts require create support. +Runtime table routes require a persisted delegation bound of at least fifteen +minutes. Legacy catalogs below that bound retain foundation-only service; activation +requires an explicit clear with expanded bounds and a listener restart after the +maintenance grace. Namespace and table mutations advertise a 24-hour UUIDv7 +idempotency window, bind canonical route, exact request input, principal and catalog +activation, and retain large results in immutable payload pages. Server errors +remain retryable, never terminal ledger outcomes. Exhausting the configured request +deadline leaves durable recovery evidence; subsecond completion is not guaranteed. +Static bearer credentials +separate reader, writer, management and clear roles; this is not an OAuth token +issuer. All four credentials are required and distinct. Writer has a separate +namespace-write capability and no catalog management or clear privilege; reader, +manager and clearer do not inherit namespace-write rights. All four can read the +configuration endpoint and namespaces. Only writer may invoke namespace or table mutations. +Management commands are separate from the Iceberg REST listener. Operational +configuration is in the [user guide](../../../user-manual/user-guide.md#9-iceberg-catalog-foundation). + Iceberg FileIO uses reserved S3-shaped locations so existing Iceberg clients can address immutable metadata and data files. The shape is a compatibility contract, not delegation to the general S3 authority. File publication, immutability, authorization, and deletion remain under Iceberg control. +Typed locations use lower-case unpadded base32 catalog IDs and lower-case hex +table IDs. Relative UTF-8 object keys preserve case, literal percent signs, plus +signs and repeated internal slashes; the whole object key is bounded to 1,024 +bytes. Dot traversal, leading slash, backslash, controls, query and fragment +delimiters are rejected rather than normalized. HTTP percent decoding belongs +only at the transport boundary, not in stored S3-shaped locations. + +Native file records bind FileId to exact location, kind, format, canonical length +and SHA-256 digest. Eligible metadata stores at most 16 KiB inline; bounded LZ4 +compression considers at most 64 KiB original input, and decoding verifies the +canonical length and digest. Other file kinds retain a fixed-size chunk root, +never a growing location vector. Hints are non-authoritative and out-of-bounds +hints are ignored. The publication primitive stages an immutable authority before +the exact-location CAS; equal-content retries return the selected FileId, while +conflicts retain losing candidates without overwriting or physical deletion. +An SDK upload supplies a path and bytes, not the eventual Iceberg data/delete +use. Sealing validates physical container bytes and records ambiguous Avro, +Parquet, ORC and Puffin uses as unbound. Selected metadata and manifests must +validate declared uses against these canonical records before table publication. +The isolated native HTTP surface exposes signed immutable object reads/writes +and multipart operations, but no general S3 bucket authority or file DELETE. + +Chunk-backed files use bounded leaf blocks and immutable chunk-resident directory +pages, with at most 256 children per page and eight directory levels. Each page +binds its catalog, table and file identity, child heights and covered byte count. +The writer retains only one partial leaf and bounded per-level frontiers. Native +block completion waits for the readable chunk cursor before publishing a root. +Pull readers retain one leaf and its current verified leaf-directory page, +bounded to 32 KiB independently of file length. Directory reuse is reader-local, +bound to the exact immutable root, and never shared across files or generations. +Readers verify directory/leaf digests and read no future block until requested; +full-file reads also verify the canonical digest. Range +parsing accepts one contiguous interval and rejects multiple ranges explicitly. +The HTTP pull-body adapter adds shared response admission and 16-KiB frames. +Only body polling starts a storage read; cancellation drops the in-flight read +before releasing admission. Exact remaining-byte hints track delivery, and storage +errors terminate the body rather than returning a successful truncated stream. +Writer checkpoints flush partial leaves and store the bounded directory frontier +plus resumable digest state in a chunk; durable journals need retain only one root. +Restoration checks owner identity, checksum, frontier heights and total byte +coverage. SHA-256 compression uses RustCrypto; versioned digest checkpoints retain +only chaining state, byte length and a partial block. They are trusted-storage +recovery records, not client authentication assertions. Failed checkpoint writes +poison the current writer without invalidating earlier durable checkpoints. +Native block writes persist an exact physical-range ownership intent in the +catalog before DiskIO. A shared-writer callback receives the assigned location; +uncertain catalog writes are read back before the physical batch proceeds. +This ledger also covers process loss before file publication and checkpoints +superseded by later assembly progress. Reclamation waits for the chunk readable +cursor or terminal state to settle any unconfirmed physical write. +Staged-tree readers validate physical roots, byte lengths and digests without +assigning a semantic file kind or declaring an incomplete multipart fragment to +be a valid complete-format file. Published-file reads retain record validation. +The assembly byte engine consumes a previously frozen part selection in ordinal +order. Each step copies at most one bounded window, persists target-writer progress +and checkpoints the current part digest. This verifies complete part digests even +when recovery spans many windows. Lost replies can repeat old progress without +duplicating bytes in the selected output; losing physical writes remain retained. +The engine requires a durable selection/progress journal and does not itself +authorize multipart operations or publish file locations. +Multipart session/part models retain independent resource limits and validate +phase coherence: publishing requires complete candidate bytes, published outcomes +require a selected FileId, and abort retains completion evidence without claiming +publication. Their FlatBuffers envelopes bind session and part identities to +separate catalog key scopes, retaining only bounded checkpoint references and +current-part digest state. Unknown phases and invalid revisions fail closed. +Catalog-scoped admission reserves an upload's entire staged-byte ceiling and one +session credit before creating its authority. Independent persisted limits cannot +be widened by another server's local configuration. A bounded CAS journal stores +immutable before/after session references; policy-bound sequence receipts make +create and terminal release recoverable without double accounting. Released +receipts remain in terminal sessions. These logical credits are not physical disk +reclamation or accounting for retained orphan bytes. +The native multipart repository reserves one part mutation in the session before +changing its part authority. A bounded before/after snapshot and monotonically +increasing revisions make the write and fence release recoverable across servers. +Counts and current staged bytes are reserved once at the session CAS. Abort cannot +bypass an unresolved mutation; stale helpers cannot restore an older part. Abort +retains parts and completion evidence rather than deleting physical storage. +Completion freezes an ordered part-number/revision/digest selection in immutable +payload pages, then changes the session phase by CAS to fence part replacement. +Selections are independently bounded to 10,000 entries and 420,007 encoded bytes. +Each completion step verifies that bounded selection and one selected part before +copying a bounded byte window and publishing its checkpoint by session CAS. Lost +replies reload progress without appending selected bytes twice. Assembled bytes +remain unexposed until semantic sealing and immutable location publication. +Foreground and recovery drivers use the same native-block-aligned byte window +below the one-MiB assembly ceiling. Equal windows prevent systematic CAS losses +to a smaller competing recovery step; alignment avoids checkpoint-only tiny leaves. +Within a step, one next 16-KiB frame read may overlap the current writer push. +There are no detached copy tasks or unbounded queues. Either IO failure cancels +the other future and returns no new checkpoint; prior durable progress stays valid. +A recovery page scans at most four session authorities and performs one pending +part settlement, logical expiry or assembly byte window per session. It validates +the complete scan page before session mutations, rejects foreign continuations and reports +finished assembly as awaiting semantic sealing. Expiry never deletes physical +parts and cannot bypass an unresolved part mutation or a publication fence. +Publication freezes a caller-validated sealed file record in immutable payload +pages before the publication phase CAS. Recovery replays that exact record through +the immutable file repository and persists the selected FileId. Equal preexisting +bytes retain their original identity. Only a proven incompatible immutable location +permits the terminal Conflicted phase; uncertain writes and context failures do not +become false aborts. Canonical format validation remains the seal caller's contract. +Each native listener schedules the multipart sweep independently of namespace +recovery. It observes at most four session revisions, then rechecks the same page +on the next tick. Byte-copy recovery defers revisions that advanced meanwhile; +unchanged revisions remain eligible. This bounded observation is only scheduling +advice, not a lock or lease: expiry, journal settlement, publication and all +context/session CAS checks remain authoritative. An observation does not retain +file bytes or survive a restart. It resets its cursor when the active context changes and bounds each +session by the persisted catalog request deadline. Timeout defers only that session, +allowing later entries in the page to progress. A separate outer budget bounds the +whole page and context/scan work. One separately bounded admission-journal recovery +step runs before scanning, including a reservation whose session is not yet present. +Terminal sessions return their credits on a later visit while retaining all parts. +The HTTP driver composes this durable state machine with physical sealing; +recovery remains the authority for abandoned or uncertain work. + +Multipart part listing uses one upload-scoped scan with at most 256 records per +page. Numeric markers preserve gaps and resume strictly after the returned part +number. Current-session checks bracket each scan; concurrent mutations invalidate +the page rather than mixing pending counters with old part records. Expired or +terminal sessions and malformed storage pages are not reported as successful lists. + +Native HTTP upload staging holds an independent concurrency +credit, slices each received frame into bounded writes and awaits storage before pulling more +input. Declared/actual byte limits, exact content length and optional signed SHA-256 +are checked before returning a tree. Failed or cancelled uploads retain orphan +blocks without publishing file authority. This transport adapter does not infer +semantic file kind, authorize grants or accept unchecked checksum trailers. + +Metadata JSON structural validation uses a bounded pull-reader bridge and an +ignored-value parser rather than retaining the metadata graph. A separate scanner +bounds nesting and verifies raw UTF-8 before parser scratch can grow. Admission +caps blocking workers; cancellation keeps its permit until the worker exits. +The validating full-file reader checks the canonical digest in the same storage +pass for chunked JSON; no independent preliminary full-file read is required. +This structural check does not replace Iceberg schema or commit validation. + +Avro writer-schema binary layouts compile to bounded named-reference graphs. +Decoded block validation checks datum widths, UTF-8, collection byte counts, +union/enum indexes and exact record consumption without retaining datum graphs. +Independent graph, recursion and visited-value limits also bound zero-byte values. +The record reader compiles its container's schema once, decodes one bounded block +per pull and permanently stops after failure or cancelled reads. Reader-schema +resolution and Iceberg logical/manifest semantics remain separate checks. + +Parquet and Puffin container probes derive footer ranges from canonical framing, +ignoring stored hints even when those hints happen to be in bounds. Their reads +retain one bounded leaf and only fixed-size framing bytes, independent of the +advertised footer size. Puffin probing also checks footer-start magic and reserved +flags. Container framing does not validate footer contents or data semantics. +Puffin metadata parsing separately caps encoded and decoded footer payloads at +1 MiB and bounds blob, field and property collections. It accepts plain JSON or +one sized, checksum-verified LZ4 frame and rejects overlapping blob ranges. Footer +deletion-vector descriptors validate their reserved snapshot/sequence markers, +uncompressed storage, referenced file and cardinality; manifest checks require +exact offset/length and referenced-file/cardinality agreement. The deletion-vector +reader then streams Roaring array, bitset and run containers, validates their +directories and cardinalities, and checks the blob's framing and CRC-32. It retains +one bounded container directory, not the deleted-position set; byte and bitmap +limits independently bound work. Snapshot-wide uniqueness and referenced data-file +row-count checks remain commit-level validation stages. +ORC probing retains at most 255 postscript bytes, checks protobuf wire framing +and resolves footer/metadata spans without decoding stripe directories. It accepts +legacy header-only magic and skips bounded unknown protobuf fields. + +Avro OCF framing uses a bounded header map and pull-based encoded blocks. Header +bytes, metadata count, block bytes and records per block have independent caps; +negative map blocks must match their declared byte lengths. Sync markers and +canonical block integrity are verified before a block returns. Errors or cancelled +reads poison the cursor rather than resuming at an ambiguous record boundary. +Null and raw-deflate block decoding enforce an independent decoded-byte cap; +truncated compressed data or unused suffixes fail closed. This layer does not +resolve Avro schemas or validate Iceberg manifest fields. + +Manifest inheritance is a separate constant-state semantic layer. It distinguishes +the manifest version from the containing table version: v1 sequences default to +zero, while new snapshots can assign row IDs to older manifests. Only added files +inherit missing sequence numbers; explicit file ages are preserved. Unassigned +data files advance the row-ID cursor in manifest order, including existing files +after an upgrade; delete files cannot carry row IDs. Invalid entries and arithmetic +overflow leave the cursor unchanged. Avro decoding and commit admission are not +yet connected to this resolver. + +Delegation tokens carry catalog activation epoch, table, principal fingerprint, +nonce, exact operation set, issue/expiry times and independent request/file byte +limits. Domain-separated HMAC authenticates bounded claims and derives per-grant +S3 credential material without a mutable credential registry. Verification requires +a freshly checked Ready context; file DELETE is not representable. These token +primitives feed native request-signature verification through a request-local +credential provider. Only the shared SigV4 algorithm is reused; general S3 +credentials and metadata are never consulted. Header and presigned requests have +bounded authentication input and reject duplicate authentication fields. Grant +expiry remains exact even when signature timestamps allow clock skew. Table +credential issuance requires a matching Ready catalog authority and rejects +lifetimes above its persisted delegated-access bound, independently of the +signer's configured maximum. A zero persisted delegation bound disables issuance. +Callers still must freshly authorize the root and exact live table or draft; +the serialization primitive does not perform those reads. The credential endpoint +checks the current namespace and published table or exact unbound draft. Draft +vending requires its original writer principal and a table-ID query selector; +same-name drafts cannot authorize each other. Expired drafts cannot refresh. Grants +last at most fifteen minutes and never outlive a draft. Published readers receive +read-only grants; only writers receive upload and multipart rights. Responses +configure the native S3 origin and SDK credential-refresh endpoint without embedding +long-lived secrets or changing canonical metadata bytes. Table-load ETags include +the SDK configuration as well as the selected-generation metadata representation; +changed endpoints cannot be hidden by a metadata-only conditional response. +Routed operation checks and streamed +request/response limits already enforce signed scopes and server budgets. +A session token alone never authenticates a request. +The native path-style request parser preserves decoded object-key bytes and limits +operations to immutable object reads/writes and multipart subresources. Unknown +query operations, duplicate parameters and general buckets fail closed. HTTP +DELETE can identify an upload abort only; it cannot identify physical file deletion. +These request primitives are attached to the native listener. Writes and reads stream through bounded CROWDB storage clients. Delegated FileIO access may move immutable ranges without an Access Server payload bounce, but @@ -76,11 +479,130 @@ Retries are idempotent across response loss. Any healthy Access Server can recover the durable operation outcome, so no server instance is a table leader or lock owner. +The library's immediate table creator records its immutable input, candidate +identity, canonical metadata and response before reserving the namespace/name. +It writes and verifies the initial metadata before acquiring a parent admission +marker. Parent helpers therefore resolve the remaining publication using catalog +records without requiring a file block reader. The initial head is selected once, +then the reservation becomes a published mapping. The durable terminal result +precedes conditional cleanup of parent and table markers. REST write admission +binds the principal, route and exact body in the shared retry ledger before invoking +these operations. Recovery reloads an existing operation before resolving the name +or current head and never rebases an uncertain request. Response headroom is checked +before publication so credential configuration fits the durable replay budget. + +Staged creation retains an invisible durable draft and metadata-only response. +Its native table location resolves the draft without a client-specific token. +The final assert-create request initializes an empty metadata builder using the +retained UUID, preserving the field IDs already used by staged files. One journal +CAS binds its request identity, input bytes, evaluation clock and candidate before +the ordinary name-reservation and parent-admission sequence begins. Initial file +validation is fenced by that exact reservation and journal revision, never by a +fabricated prior head. Publication checks all selected initial snapshots and the +enabled auxiliary-file profile before writing canonical table metadata. + +Draft expiry and final binding compete on the same phase CAS. Only an unbound +draft may expire; bound operations recover their original publication outcome +regardless of elapsed time. Known semantic file failures retain a terminal client +error and release their reservation. Uncertain storage outcomes remain recoverable. +The draft response and final commit response are retained separately for exact +replay. + +Bounded background scans rotate creation, update and lifecycle journals, four records per +page, with independent continuations reset on catalog activation changes. Recovery +expires only unbound drafts, reconstructs fixed candidate proofs, settles published +markers and retains uncertain storage errors. Known semantic validation failures +become durable client outcomes before any candidate is published. Recovery deadlines +preserve journal evidence rather than canceling the logical operation. + +Logical drop and same/cross-namespace rename use a bounded `TableLifecycleOperation` +journal. It fixes the original head and exact source mapping, request identity, +principal, input and candidate before publication. A single head CAS arbitrates +against metadata commits and other lifecycle operations. A losing operation keeps +its terminal conflict instead of rebasing onto a new generation or recreated name. +Rename changes the canonical identifier and name epoch, not table identity, UUID, +metadata generation, digest or file location. Destination reservation precedes a +namespace admission CAS. The admission marker remains until the head outcome and +destination mapping are durable; namespace-drop helpers finish or abort that exact +operation with a shared bounded work budget. The source stays head-qualified until +the move publishes, and the old name never becomes an alias. Cleanup conditionally +removes only the captured mapping, preserving names recreated with another identity. + +Drop tombstones the selected head without traversing snapshots or deleting files. +A purge request persists a `TablePurgeTask` containing the tombstoned head and +activation epoch, indexed by table, generation and metadata file. This is pending +reachability-proof work, not proof of deletion or permission to delete. Success is +retained before releasing rename head/namespace markers. Retrying after response +loss returns the original result without mutating a replacement table. The REST +drop/rename routes require independent writer credentials and return empty success +responses; stale table names fail normal load, exists, commit and credential refresh. + Drop, replacement, and snapshot expiration remove logical reachability first. Physical reclamation follows a proof that no live metadata, snapshot, reference, lease, or retained operation can reach the file. General S3 deletion and lifecycle rules cannot reclaim Iceberg-owned data. +The reclamation proof binds current and pinned historical metadata to their +captured heads. Its immutable traversal stack and compressed binary file-ID index +use content-addressed payload pages. A task CAS publishes the pending stack and +mark root together; a missing page is an error, including during a nonmembership +query. Physical deletion runs only for a tombstoned table or retired catalog; +the worker rechecks inactive authority before sweeping. A Ready table may retain +unreachable files until drop or clear rather than interrupt reads or commits. +Retained operations and table-wide credentials conservatively defer reclamation. +Background task advancement requires explicit activation; +it uses a separate storage client pool, one-step concurrency admission, bounded +KV and chunk request/byte budgets, and durable retry state. The enabled +scheduler admits persisted table purge markers and completed catalog clears; +management may also start inactive tasks. The scheduler is disabled by default +and requires explicit operator activation with validated resource limits. + +Provisioned disk capacity is the allocation boundary for both foreground files +and GC durable workspace. A failed GC workspace write retains the last durable +continuation and defers retry; it never substitutes an incomplete proof or +authorizes deletion. Committed files remain readable when new chunk allocation +fails. Progress resumes after capacity is restored through the normal storage +flow. Shared-chunk ranges remain pending while range deletion is unsupported. + +Metadata readers, direct FileIO, file publication and both published and staged +credentials persist pins before rechecking their authority. Pin expiry includes +the applicable persisted request and clock-skew bounds. Once a file's canonical +deletion intent has started, ordinary resolution and publication reject it even +if physical range reclamation is deferred. Legacy live tasks are retired without +further deletion, releasing an owned table fence. Retained and deferred +candidates remain durable work for later inactive passes. + +An already authorized FileIO GET or HEAD can pin a tombstoned table while its +credential remains valid. The pin is persisted before the exact head is +rechecked; a concurrent transition to `Reclaiming` rejects admission. Uploads +and new table credentials still require a Ready table. Logical drop therefore +does not invalidate retained file reads or bypass the physical deletion fence. + +Retired catalog recovery scans system retry and management ledgers before file +deletion and after the final file rescan. Pending or retained bindings stop the +pass; exact-identity overflow entries remain independent of occupied primary +slots. After files are reclaimed, the worker conditionally removes expired +bindings, audits, projections and non-GC catalog records while preserving the +active root's management operation. Multipart parts have their own durable tree +candidates. Assembly checkpoints have separate claims and a durable frontier-root +index. Abandoned frontiers are authenticated before traversal; a conflicted final +tree is traversed once instead of revisiting its shared frontier. Published +sessions reclaim only the checkpoint block, preserving the assembled data tree. +Each physical step rechecks the terminal session and retention. The checkpoint +block is deleted after its children, and session cleanup requires its completed +claim. Block intents are swept after tree candidates, preserving reachable owners +and any unfinished file or assembly cursor. An owner fence prevents publication +or new block writes once orphan deletion begins. Superseded intents of a reachable +owner are conservatively retained until that owner becomes unreachable. + +Final catalog cleanup verifies that all candidates are complete and that no owner +is paused or quarantined. A durable retirement marker selects the cleanup owner +and rejects stale GC mutations before bounded deletion of claims, candidates, +proof pages, write fences and old tasks. Only the retired authority, winning task +result and retirement marker remain. Late unfinished records stop cleanup. +Uncertain progress responses are resolved by reading durable state; an unfenced +live proof whose authority changed terminates without deleting files. + ## 5. Compatibility CROWDB covers the core Iceberg format semantics for v1, v2, and v3, including @@ -92,6 +614,48 @@ REST wire types, Iceberg domain state, and CROWDB storage records remain separate. Unknown or disabled requirements and updates fail before mutation. The backed-up specifications decide behavior when implementations differ. +### Executable conformance profile + +The official Java oracle is Apache Iceberg 1.11.0; the official Rust REST +client is 0.10.0. The declared selected-file profile uses Parquet data/deletes, +Avro manifests and Puffin deletion vectors/statistics. ORC bytes can be stored, +but selected ORC validation and compute-engine certification are separate work. + +- **v1/v2/v3 metadata and upgrades:** `official_java_metadata_roundtrips_without_rewriting`, + `official_catalog_creates_commits_upgrades_stages_and_refreshes_native_credentials`, + and the official create/update/snapshot fixtures compare canonical metadata. + `TestIcebergVersionRows` reads actual rows before and after adjacent upgrades, + reads historical snapshots, expires them logically, and reloads after restart. +- **Selected data and deletes:** `TestIcebergSelectedFiles` rejects mismatched + selected uses of identical uploaded bytes, reads the original row, verifies + equality-delete visibility and reads the historical snapshot. Canonical file + validators separately cover position deletes, v3 lineage, deletion vectors, + defaults, nested/variant types, integer encodings and nullable values. +- **Statistics:** `TestIcebergCatalogWrites` and the official partition-statistics + fixtures cover publication, replay, evolution and staged creation. Historical + omissions follow the explicit compatibility rules described above. +- **Discovery and authorization:** + `discovery_uses_installed_routes_and_unsupported_paths_leave_no_record` and + `explicit_partial_activation_limits_discovery_and_table_admission` exercise + installed/disabled route and version combinations. HTTP namespace, table, + lifecycle, credentials and body tests cover roles, malformed requests, limits, + cancellation, unchanged authority on rejection and bounded metrics. +- **Faults and retirement:** `official_rust_client_observes_lost_create_reply_on_another_listener`, + `official_rust_client_lost_reply_survives_native_storage_restart`, and the + Java response-loss/retired-catalog fixtures exercise listener changes and + restart. The SDKs do not automatically replay a lost mutation POST with the + same key; direct HTTP fault tests prove the server's same-key replay contract. +- **Apache REST Compatibility Kit:** the unmodified 1.11.0 runner verifies six + supported cases: namespace create, basic table create, rename, drop, + missing-table drop and table list. This is not a full-kit pass. Other cases + assume register/views or local filesystem locations that the native authority + deliberately rejects. Custom SDK fixtures are not described as kit results. +- **Large logical files:** `tib_address_space_range_reads_keep_fixed_windows_and_small_authority` + and `tib_reclamation_progress_serializes_a_bounded_resumable_cursor` use a + 1 TiB logical tree with repeated immutable blocks. They prove bounded windows + and resumable state, not physical TiB capacity. Oversized declared metadata + is rejected before I/O at the configured metadata budget. + ## 6. Relationship to other access models General S3 and Iceberg share chunk, Chunk-KV, transport, credential, and safe diff --git a/doc/design/chunkdb/chunkdb-allocate-flow-analysis.md b/doc/design/chunkdb/chunkdb-allocate-flow-analysis.md index 9f8c3d5e6..cd4ca4a9a 100644 --- a/doc/design/chunkdb/chunkdb-allocate-flow-analysis.md +++ b/doc/design/chunkdb/chunkdb-allocate-flow-analysis.md @@ -52,7 +52,7 @@ then committing or freeing the block. - Duration: 20 seconds per row. - Capacity: four 4-TiB logical disks per DiskDB; 256-GiB zones. - KV inflight/coalescing: 32/32. -- Command: `pixi run -- bash tools/bench-chunkdb-regression.sh`. +- Command: `pixi run -- bash tools/benchmark/bench-chunkdb-regression.sh`. | Workload | Groups | Threads | Strips | EC | Client conn | ChunkDB conn | DiskDB conn | KV conn | Workers | Chunk/s | Block/s | p50 us | p99 us | Errors | Space | |---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---| diff --git a/doc/design/chunkdb/design-crowdb-chunkdb.md b/doc/design/chunkdb/design-crowdb-chunkdb.md index e3c1c93d1..9dbe0257d 100644 --- a/doc/design/chunkdb/design-crowdb-chunkdb.md +++ b/doc/design/chunkdb/design-crowdb-chunkdb.md @@ -996,9 +996,11 @@ mutating RPC acquires the per-chunk lock before its RMW cycle: - `delete_chunk`: `check_range` → `acquire` → state check → persist Deleted with segments as cleanup intent → free segments → clear the segment list and persist the tombstone → `guard.refresh(chunk)`. -- `delete_chunk_range`: `check_range` → `acquire` → validate a nonzero, - nonoverflowing range → persist the retained strips → free the removed - strips' segments → `guard.refresh(chunk)`. +- `delete_chunk_range`: offset and length are independent u32 byte values, + describing an exact half-open object range without KiB rounding. The RPC + returns `Unimplemented` without changing storage until shared-object range + reclamation is supported. An intersecting strip is not permission to free + its blocks while neighbouring objects remain live. - `update_chunk_strip`: compatibility wrapper over the one-strip form of `replace_chunk_strip_range`. - `replace_chunk_strip_range`: `check_range` → `acquire` → validate Active or @@ -1276,7 +1278,7 @@ total to equal the DiskDB busy-space delta after compaction. Capacity exhaustion is a successful stop reason; any correctness error invalidates the sample. -`tools/bench-chunkdb-regression.sh` builds all four release binaries and uses +`tools/benchmark/bench-chunkdb-regression.sh` builds all four release binaries and uses a fresh timestamped combined cluster for mirror, EC 4+2, EC 8+4, lifecycle mix, concurrency, and capacity-exhaustion cases. It retains each case's logs, destroys each cluster, runs all later cases after a failure, and returns a diff --git a/doc/design/chunkds/design-crowdb-chunk-kv.md b/doc/design/chunkds/design-crowdb-chunk-kv.md index ff75f8514..1eaad720a 100644 --- a/doc/design/chunkds/design-crowdb-chunk-kv.md +++ b/doc/design/chunkds/design-crowdb-chunk-kv.md @@ -17,8 +17,9 @@ name, and one partition-local mutation sequence. Unbounded endpoints are allowed. A split key must be strictly inside the source range and its two child ranges must be adjacent and exactly cover the parent. -The lifecycle is `Closed`, `Recovering`, `WriteStalled`, `Prepared`, -`Serving`, `SplitPreparing`, `SplitFinalizing`, `Retired`, or `Faulted`. +The lifecycle is `Closed`, `Recovering`, `TransferQuiesced`, `Prepared`, +`Serving`, `TransferFencing`, `SplitPreparing`, `SplitFinalizing`, `Retired`, +or `Faulted`. Data-path admission reads atomics and reserves bounded request and byte capacity. Lifecycle control closes mutation admission and waits asynchronously for the admitted count to reach zero. The manager registry and @@ -96,11 +97,14 @@ WAL trim cannot cross the checkpoint's replay offset. Requests below the declared retained floor return `RequestExpired` and are not executed anew. The frontiers satisfy -`checkpoint_seq <= applied_seq <= journal_durable_seq`. A journal uncertainty -enters `WriteStalled` and preserves reads from the healthy applied prefix. A -non-OK tree result after durable append is `ApplyStateUnknown`, moves only that -partition to `Recovering`, and prevents later records from applying. The C ABI -catches C++ exceptions before they can cross into Rust. +`checkpoint_seq <= applied_seq <= journal_durable_seq`. The stream resolves +uncertain cursor and manifest outcomes against durable state before completing +the append. If a journal append nevertheless returns an error, the partition +enters `Recovering`; later writes and reads require recovery rather than +mistaking that failure for a quiesced transfer source. A non-OK tree result +after durable append is `ApplyStateUnknown`, also moves only that partition to +`Recovering`, and prevents later records from applying. The C ABI catches C++ +exceptions before they can cross into Rust. ## 5. Overlay Split and Writer Handoff @@ -224,7 +228,8 @@ assigning new records to the source WAL, drains only records already assigned there, and records their final durable sequence and byte offset as `C`. One short `TransferFencing` lifecycle closes public mutation admission while the existing mutation worker drains the already-admitted set; it is not a second -queue or an independent authority flag. The +queue or an independent authority flag. Only after that drain does the source +enter `TransferQuiesced`, which permits the exact handoff checkpoint. The target artifact is extended from `P` to `C`. The live target keeps its opened tree, memtable, and replay coroutine and reads only `P+1..C`; reopening the pinned base and replaying the complete retained suffix is crash recovery, not @@ -273,8 +278,8 @@ post-journal apply state require recovery of the affected partition only. Checkpoint and GC failures retain the prior manifest and WAL authority. Per-partition lock-free counters cover mutation requests and outcomes, ordered -seeks and scans, range and stale-epoch rejection, admission backpressure, write -stalls, unknown apply outcomes, recoveries, checkpoints, and split lifecycle +seeks and scans, range and stale-epoch rejection, admission backpressure, +journal failures, unknown apply outcomes, recoveries, checkpoints, and split lifecycle events. Split counters distinguish preparation and base-checkpoint time, tail records and bytes, catch-up lag and finalization duration, overlay replay records and bytes, and materialization duration. Snapshot frontiers expose lifecycle, diff --git a/doc/design/chunkds/design-crowdb-chunk-stream.md b/doc/design/chunkds/design-crowdb-chunk-stream.md index c2c406210..bfca20358 100644 --- a/doc/design/chunkds/design-crowdb-chunk-stream.md +++ b/doc/design/chunkds/design-crowdb-chunk-stream.md @@ -127,12 +127,15 @@ in queue order into one retained staging buffer and sent through one `write_mirrors` call. The worker then performs one fenced cursor advance. Completion occurs only after all mirror writes and the durable cursor update. -Each request receives its exact non-overlapping logical subrange. A failed -batch completes no member successfully and stalls subsequent writes until -reopen. An ambiguous cursor response is inspected without resubmission: the -worker accepts it only when the durable cursor equals the proposed end and the -last-advance checksum matches the staging checksum; an unchanged cursor proves -absence; every other state stalls. +Each request receives its exact non-overlapping logical subrange. Mirror-write +failures use the retained strip image to replace a failed block; a confirmed +absent append can rotate to another chunk and retry. An ambiguous cursor +response is inspected without blind resubmission: the worker accepts it only +when the durable cursor equals the proposed end and the last-advance checksum +matches the staging checksum; an unchanged cursor proves absence. If durable +state cannot be read, the current append remains pending for resolution. A +fencing violation or corrupt durable state requires reopening and recovery; +it is not a transfer-quiescence state. For a chunk-bound request, the worker adds the selected `ChunkId` after the caller's bytes before assembling the aggregate write. Admission, remaining diff --git a/doc/design/chunkio/chunkio-write-flow-analysis.md b/doc/design/chunkio/chunkio-write-flow-analysis.md index c7fd28c70..a5c636767 100644 --- a/doc/design/chunkio/chunkio-write-flow-analysis.md +++ b/doc/design/chunkio/chunkio-write-flow-analysis.md @@ -5,7 +5,7 @@ Large-object write flow from the benchmark workload through chunk preparation, fetch, EC encode, DiskIO RPC, and chunk seal. The benchmark -sentinel is `tools/bench-chunkio-write-regression.sh`. The write pipeline +sentinel is `tools/benchmark/bench-chunkio-write-regression.sh`. The write pipeline architecture is in [`design-crowdb-chunkio.md`](design-crowdb-chunkio.md); this doc traces the measured hot path and records benchmark results. diff --git a/doc/design/chunkio/design-crowdb-chunkio-small-object-writer.md b/doc/design/chunkio/design-crowdb-chunkio-small-object-writer.md index b663ee744..c09d12e06 100644 --- a/doc/design/chunkio/design-crowdb-chunkio-small-object-writer.md +++ b/doc/design/chunkio/design-crowdb-chunkio-small-object-writer.md @@ -32,8 +32,9 @@ and orphan sealing. One client-owned pool multiplexes small objects across a bounded set of pipelines. Each pipeline exclusively owns one active Repo chunk containing mirror strips and may own one empty prepared replacement. A pipeline batches -whole objects, writes the physical range to every mirror, durably advances the -chunk cursor, then returns an independent `Location` to each caller. +whole objects and writes the physical range to every mirror. Ordinary completion +returns independent locations while cursor progress runs asynchronously; callers +can opt into completion after the readable cursor is durably confirmed. The shared path can incrementally form 8+4 EC groups while writing mirrors. It retains one open-strip image and four parity accumulators, but never retains @@ -97,6 +98,14 @@ recovers the unique mutable shadow after completion without another steady-state copy. Each returned location covers only its object's exact bytes. +A caller can attach a `SmallWriteIntent` to durable completion. Once the batch +has assigned exact locations, every attached callback completes before any of +the batch's DiskIO. Failure aborts the batch and its pipeline; cancellation of +the caller does not detach ownership registration from the physical operation. +The callback has no default storage policy: FileIO uses it to persist its own +catalog-sharded block ledger. Range reclamation defers active shared-chunk ranges +that extend beyond the acknowledged cursor, preserving uncertain writes. + Chunk allocation reserves a bounded group of hidden strips. A reserved strip becomes `Consumed` immediately before its first DiskIO and is confirmed into the readable layout only after its mirror write succeeds. Confirmation and @@ -147,11 +156,25 @@ The response barrier is: 4. Coalesce cursor progress in the background metadata chain. Strip close and batched append execute there in revision order. +The physical mirror-strip flow is shared with chunk streams: it writes mirrors +in parallel, excludes failed disks, writes a prefix-complete replacement image, +and publishes the fenced strip swap. Each single-owner caller retains its own +current-strip shadow and controls its publication barrier. Journal streams also +fsync the final mirror set before advancing their durable cursor and resolve an +uncertain replacement result against chunk metadata before retrying it. + No location is visible before its complete physical range exists on every configured mirror. Cursor persistence is an asynchronous availability and orphan-recovery checkpoint; readers can transiently report `NotYetAvailable` until it catches up. +Callers publishing immediately readable immutable authorities use +`SharedObjectWriter::finish_durable`. A batch containing a durable-completion +request waits for the existing metadata chain, then confirms any remaining +cursor suffix before delivering locations. Metadata failure fails that batch +instead of exposing an unreadable reference. Ordinary `on_finish` retains its +asynchronous cursor behavior; no additional lock or reader-side retry is needed. + ## 6. Chunk Lifecycle and Recovery A pipeline allocates its first chunk and its initial strip batch before diff --git a/doc/design/config/design-crowdb-config.md b/doc/design/config/design-crowdb-config.md index 9fe34e0b2..bc910ef2f 100644 --- a/doc/design/config/design-crowdb-config.md +++ b/doc/design/config/design-crowdb-config.md @@ -112,6 +112,33 @@ vary, and restart reuses the same file path. Explicit deployment overrides are represented as explicit CLI options and therefore retain the standard precedence. +The single-node container uses a named, validated `DeploymentProfile` and +rendered service configurations. `crowdb-monitor` owns PID 1 supervision, +dependency order, probes, restart budgets, bounded logs and shutdown. A durable +bootstrap manifest binds the profile/configuration digest and generated +credential identity to the mounted data volume. Restart replays completed steps; +conflicting identity fails closed. Child recovery revalidates storage, catalog +and Web authority before readiness returns. + +Release programs and UI are compiled incrementally on a Linux amd64 host with +the locked repository toolchain. Docker receives a staged runtime directory +containing only those artifacts, required shared libraries and the deployment +profile. It performs no source compilation and contains no Pixi or compiler. +Runtime linkage and source/version metadata are checked while packaging into +the digest-pinned Ubuntu image. The publication job reuses the verified staged +artifacts rather than compiling a second set. + +Group 0 owns hardware/topology and service registration. It does not store +container mounts, process PIDs or restart policy. Docker Web reads live Group 0 +topology and fresh monitor process status independently; missing authority does +not produce an empty or cached topology. Hardware/process mutation is disabled, +while authenticated logical operations use the existing operation paths. + +Only the container profile overrides public listeners to Iceberg 80, S3 81 and +Web 8080. Bare-metal defaults remain independent. See the +[single-node container guide](../../user-manual/docker-single-node-user-guide.md) +for publication, volume and endpoint usage. + ## 7. Failure Handling Malformed TOML, an unreadable named file, a wrong known-field type, or failed diff --git a/doc/design/console/design-crowdb-console.md b/doc/design/console/design-crowdb-console.md index fe9ba40bc..3f927d880 100644 --- a/doc/design/console/design-crowdb-console.md +++ b/doc/design/console/design-crowdb-console.md @@ -488,12 +488,22 @@ For each multi-node operation in the logical tree, the backend obeys these rules: - **Plan first, act second.** Resolve every required node + replica id - from the monitor cache before issuing any upstream RPC. + from Group 0 membership and live service registrations before issuing + mutation RPCs. Missing or ambiguous registrations fail the operation. - **Built on physical primitives.** The orchestrator only calls the per-node physical mutators; it never invents a side channel. - **All-or-nothing where feasible.** On partial failure, attempt to undo successful sub-steps and surface the resulting state in the error body. +- **Confirmed membership publication.** Complete peer wiring before publishing + group or replica membership. A new group and its initial replica records + commit in one conditional batch. Create records conditionally; a concurrent + conflicting record is preserved. A lost write response is resolved only by + a linearizable read that confirms the intended record. +- **Deletion preserves authority on node failure.** Confirm deletion on every + hosting node before removing membership. Store hosts include nodes from + replica records as well as the store record. Remove descendants before + parents; an already absent node-side object permits retry. - **Idempotent retries.** A repeat of the same logical request must converge to the same state. - **Cache refresh on success.** Every successful mutation triggers an @@ -655,6 +665,12 @@ group-0 sysdata. After `cluster init` completes, subsequent commands use `--system-ip` / `--system-port` to connect to any node in the newly created system group. +Initialization also sends Group 0 management seeds to deployed KV processes +outside the selected member set and waits for exactly one live registration +per node. Those processes retain connection hints locally across restart; +they do not become Group 0 members. Logical operations use the confirmed live +registrations rather than treating launch configuration as a live endpoint. + ### 7.4 `cluster clean` — data wipe boundary `cluster clean` wipes user-layer data across all storage services, diff --git a/doc/design/diskdb/diskdb-allocate-flow-analysis.md b/doc/design/diskdb/diskdb-allocate-flow-analysis.md index cc6ce16a2..d0ce7fd99 100644 --- a/doc/design/diskdb/diskdb-allocate-flow-analysis.md +++ b/doc/design/diskdb/diskdb-allocate-flow-analysis.md @@ -161,7 +161,7 @@ and exact rollback semantics. The immediate-drain free coalescer was compared with the direct path using: ```bash -DISKDB_BENCH_DURATION=10 DISKDB_BENCH_CASES='free_batch_off_mem free_batch_on_mem' pixi run -- bash tools/bench-diskdb-regression.sh +DISKDB_BENCH_DURATION=10 DISKDB_BENCH_CASES='free_batch_off_mem free_batch_on_mem' pixi run -- bash tools/benchmark/bench-diskdb-regression.sh ``` Reference host: Intel Core i9-7960X (16 cores / 32 threads), x86_64, diff --git a/doc/design/kv/design-crowdb-kv-server.md b/doc/design/kv/design-crowdb-kv-server.md index 3907ceb61..244371f5e 100644 --- a/doc/design/kv/design-crowdb-kv-server.md +++ b/doc/design/kv/design-crowdb-kv-server.md @@ -132,6 +132,21 @@ Key endpoint groups: memtable into L1 in memory; used by the bench's `--flush-after-prepopulate` flag and as an admin drain). +`POST /system/group0-discovery` accepts one to sixteen HTTP management +origins for a process launched before Group 0 exists. The server atomically +persists these connection hints and its keepalive loop discovers Group 0 +before registering. Explicit startup seeds take precedence. Repeating the +same request is safe; malformed origins and embedded credentials are rejected. +These hints survive process restart and never create stores, groups, replicas, +or membership. Without hints or a local Group 0 replica, registration waits. + +The service instance identity is persisted with the node's fixed-layout +configuration before opening listeners. Restarts reuse it, including when a +previous process could not unregister. A changed explicit instance or node +identity, or a malformed identity file, fails startup. Different durable roots +receive different generated instance identities; duplicate live node identities +remain ambiguous rather than being merged by discovery. + ### 2.5 Group lifecycle Local replicas start as `Follower`; no role assignment needed at group diff --git a/doc/design/kv/kv-read-flow-analysis.md b/doc/design/kv/kv-read-flow-analysis.md index 412fa588c..369fd3e24 100644 --- a/doc/design/kv/kv-read-flow-analysis.md +++ b/doc/design/kv/kv-read-flow-analysis.md @@ -5,7 +5,7 @@ Point reads (`get`) from the client through crowdb-rpc, the Paxos read policy, and the storage engine. The benchmark sentinel is -`tools/bench-kv-read-regression.sh`. +`tools/benchmark/bench-kv-read-regression.sh`. ## 1. Flow diff --git a/doc/design/kv/kv-scan-flow-analysis.md b/doc/design/kv/kv-scan-flow-analysis.md index 55bb318dd..8a65c60f8 100644 --- a/doc/design/kv/kv-scan-flow-analysis.md +++ b/doc/design/kv/kv-scan-flow-analysis.md @@ -5,7 +5,7 @@ Range reads from the client through crowdb-rpc, the read policy, and the crowdb-tree cursors. The benchmark sentinel is -`tools/bench-kv-scan-regression.sh`. +`tools/benchmark/bench-kv-scan-regression.sh`. ## 1. Flow diff --git a/doc/design/kv/kv-write-flow-analysis.md b/doc/design/kv/kv-write-flow-analysis.md index 4945c1941..d23dfe7f6 100644 --- a/doc/design/kv/kv-write-flow-analysis.md +++ b/doc/design/kv/kv-write-flow-analysis.md @@ -5,7 +5,7 @@ Write flow from client request through proposal admission, Paxos, WAL, and engine apply. The benchmark sentinel is -`tools/bench-kv-write-regression.sh`. +`tools/benchmark/bench-kv-write-regression.sh`. ## 1. Flow @@ -272,7 +272,7 @@ space for 15 seconds, then repeats from clean group state three times on one three-node mem-block deployment. The command was: ```bash -KV_WRITE_BENCH_CASES=largeval_16k pixi run -- bash tools/bench-kv-write-regression.sh +KV_WRITE_BENCH_CASES=largeval_16k pixi run -- bash tools/benchmark/bench-kv-write-regression.sh ``` Reference host: Intel Core i9-7960X (16 cores / 32 threads), x86_64, Linux diff --git a/doc/design/tree/design-crowdb-tree-engine.md b/doc/design/tree/design-crowdb-tree-engine.md index 9e95781a5..22e1a3d88 100644 --- a/doc/design/tree/design-crowdb-tree-engine.md +++ b/doc/design/tree/design-crowdb-tree-engine.md @@ -460,7 +460,7 @@ the point of L0 is that it is no longer on the scan or get path. Scan is part of the read flow but a separate perf track from random point reads (different cost shapes: per-entry overhead vs leaf-chain traversal vs per-byte copy). The regression sentinel is -`tools/bench-kv-scan-regression.sh` driving `crowdb-cli bench run --workload +`tools/benchmark/bench-kv-scan-regression.sh` driving `crowdb-cli bench run --workload list` with `--scan-limit`, `--scan-prefix`, `--scan-start-after` flags against a 3-node mem-mode cluster, mirroring the write/read regression sentinels. The sync `scan` `start_after` pushdown correctness is @@ -567,7 +567,7 @@ guard). Design rules: - **Allocator seam.** `alloc()` routes owned allocations larger than `kInlineCap` through a single internal allocator hook (today: glibc `malloc`); a size-classed pool or RDMA-pinned allocator could slot in here - later with no call-site changes. See [`todo_code.md`](../todo_code.md) for + later with no call-site changes. See [`todo_code.md`](../../todo_code.md) for why that hasn't been done speculatively. - **MemTable = `absl::btree_map`.** The KEY stays `std::string`, the VALUE is a `cell_entry{slot, flags, cell}`. The diff --git a/doc/dev/env_setup.md b/doc/dev/env_setup.md index 528551f04..2a2a4924a 100644 --- a/doc/dev/env_setup.md +++ b/doc/dev/env_setup.md @@ -17,7 +17,7 @@ One-time host setup for perf counters and CROWDB benchmarks on Ubuntu ## Run ```bash -sudo bash tools/setup-perf.sh +sudo bash tools/profiling/setup-perf.sh ``` The script auto-detects AMD vs Intel, applies every setting below, and @@ -177,10 +177,10 @@ pixi run build-cpp Regression sentinels under `tools/`: -- `tools/bench-kv-read-regression.sh` -- `tools/bench-kv-write-regression.sh` -- `tools/bench-kv-scan-regression.sh` -- `tools/bench-rpc-regression.sh` -- `tools/bench-diskdb-regression.sh` -- `tools/bench-chunkdb-regression.sh` -- `tools/bench-chunkio-write-regression.sh` +- `tools/benchmark/bench-kv-read-regression.sh` +- `tools/benchmark/bench-kv-write-regression.sh` +- `tools/benchmark/bench-kv-scan-regression.sh` +- `tools/benchmark/bench-rpc-regression.sh` +- `tools/benchmark/bench-diskdb-regression.sh` +- `tools/benchmark/bench-chunkdb-regression.sh` +- `tools/benchmark/bench-chunkio-write-regression.sh` diff --git a/doc/doc_index.md b/doc/doc_index.md index 585fa9e0d..1fb23e13c 100644 --- a/doc/doc_index.md +++ b/doc/doc_index.md @@ -22,6 +22,7 @@ listed document or section needed by the task. | `doc/design/config/design-crowdb-config.md` | Configuration ownership, precedence, validation, reload. | | `doc/design/access-server/design-crowdb-access-server.md` | S3, Iceberg, native Dataset access, and GPU delivery. | | `doc/user-manual/user-guide.md` | Web UI, CLI, REST API, setup, operations, upgrade. | +| `doc/user-manual/docker-single-node-user-guide.md` | Single-node Docker preview, credentials, clients, recovery. | ## Backlog (`doc/backlog/`) @@ -50,12 +51,14 @@ Temporary plans live under `doc/working/`; flow analyses live under | ---------------------------- | ---------------------------------------------------------------------- | | `doc/dev/env_setup.md` | Benchmark commands, sentinels, prerequisites, and perf-counter setup. | | `doc/dev/hyper_fork.md` | Hyper fork branches, submodule, build, sync, validation, and recovery. | +| `tools/README.md` | Tool directories, Pixi task entry points, CI checks, and suite timing. | ## Project Files (repo root) | File | When to read | | -------------------- | ------------------------------------------- | | `AGENTS.md` | Always-on project rules and skill dispatch. | +| `VERSION` | Canonical project development version. | | `CONTRIBUTING.md` | PR setup, conventions, and process. | | `CHANGELOG.md` | Release history. | | `SECURITY.md` | Vulnerability handling. | diff --git a/doc/user-manual/build_html.py b/doc/user-manual/build_html.py index 46dcef7b4..8bc48d3d9 100644 --- a/doc/user-manual/build_html.py +++ b/doc/user-manual/build_html.py @@ -2,7 +2,7 @@ """Convert user-guide.md to a standalone HTML page with tabbed code sections. Usage: - python3 doc/user-manual/build_html.py + python3 doc/user-manual/build_html.py [source.md] Output: doc/user-manual/user-guide.html @@ -25,7 +25,6 @@ SCRIPT_DIR = Path(__file__).resolve().parent SOURCE = SCRIPT_DIR / "user-guide.md" -OUTPUT = SCRIPT_DIR / "user-guide.html" CSS = """ :root { @@ -340,6 +339,7 @@ def inline_format(text: str) -> str: text = escape(text) text = re.sub(r"`([^`]+)`", r"\1", text) text = re.sub(r"\*\*([^*]+)\*\*", r"\1", text) + text = re.sub(r'\[([^\]]+)\]\((https?://[^\s)"<>]+)\)', r'\1', text) return text @@ -529,7 +529,7 @@ def close_list(): def build_sidebar(toc: list[dict]) -> str: - lines = ['', "