diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9e7d3c7f1..20ef248c5 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -14,7 +14,7 @@ env: RUST_BACKTRACE: 1 RUST_LIB_BACKTRACE: 1 -# 10 parallel jobs. See doc/working/test.md § "Current CI Test Design" +# 9 parallel jobs. See doc/working/test.md § "Current CI Test Design" # for the assignment rule and how to add new test tasks. jobs: Lint: @@ -37,8 +37,8 @@ jobs: uses: Swatinem/rust-cache@v2 - name: Check formatting run: pixi run cargo fmt --all -- --check - - name: Check test-task coverage - run: pixi run test-task-coverage + - name: Check CI test tasks + run: pixi run check-ci-test-tasks - name: Run clippy run: pixi run cargo clippy --all-targets -- -D warnings @@ -280,8 +280,8 @@ jobs: name: runtime-icebergsdk-${{ github.run_attempt }} path: | .crowdb-runtime/ - target/iceberg-rck/open-api/build/reports/tests/ - target/iceberg-rck/open-api/build/test-results/ + ${{ runner.temp }}/iceberg-rck/open-api/build/reports/tests/ + ${{ runner.temp }}/iceberg-rck/open-api/build/test-results/ if-no-files-found: ignore retention-days: 7 - name: Clean subprocesses @@ -395,39 +395,3 @@ jobs: - name: Clean subprocesses if: always() run: pixi run clean-env - - DockerPreview: - runs-on: ubuntu-24.04 - permissions: - contents: read - steps: - - uses: actions/checkout@v4 - with: - submodules: true - - name: Free disk space - run: | - sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache - sudo apt-get clean - df -h / - - uses: prefix-dev/setup-pixi@v0.8.1 - with: - pixi-version: latest - - name: Build and test single-node preview image - env: - CROWDB_PREVIEW_TEST_ARTIFACTS: ${{ runner.temp }}/crowdb-preview-artifacts - run: pixi run test-single-node-container - - name: Capture Docker diagnostics on failure - if: failure() - run: | - docker ps -a - docker images - docker info - df -h - - name: Upload preview failure logs - if: failure() - uses: actions/upload-artifact@v4 - with: - name: docker-preview-${{ github.run_attempt }} - path: ${{ runner.temp }}/crowdb-preview-artifacts - if-no-files-found: ignore - retention-days: 7 diff --git a/.github/workflows/docker-preview.yml b/.github/workflows/docker-preview.yml new file mode 100644 index 000000000..fcb0bed03 --- /dev/null +++ b/.github/workflows/docker-preview.yml @@ -0,0 +1,41 @@ +name: DockerPreview + +on: + workflow_dispatch: + +jobs: + DockerPreview: + runs-on: ubuntu-24.04 + permissions: + contents: read + steps: + - uses: actions/checkout@v4 + with: + submodules: true + - name: Free disk space + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache + sudo apt-get clean + df -h / + - uses: prefix-dev/setup-pixi@v0.8.1 + with: + pixi-version: latest + - name: Build and test single-node preview image + env: + CROWDB_PREVIEW_TEST_ARTIFACTS: ${{ runner.temp }}/crowdb-preview-artifacts + run: pixi run test-single-node-container + - name: Capture Docker diagnostics on failure + if: failure() + run: | + docker ps -a + docker images + docker info + df -h + - name: Upload preview failure logs + if: failure() + uses: actions/upload-artifact@v4 + with: + name: docker-preview-${{ github.run_attempt }} + path: ${{ runner.temp }}/crowdb-preview-artifacts + if-no-files-found: ignore + retention-days: 7 diff --git a/.github/workflows/iceberg-rust-sdk.yml b/.github/workflows/iceberg-rust-sdk.yml new file mode 100644 index 000000000..9e0c265eb --- /dev/null +++ b/.github/workflows/iceberg-rust-sdk.yml @@ -0,0 +1,42 @@ +name: IcebergRustSDK + +on: + workflow_dispatch: + +env: + CARGO_TERM_COLOR: always + CARGO_INCREMENTAL: "0" + CARGO_PROFILE_DEV_DEBUG: line-tables-only + CARGO_PROFILE_TEST_DEBUG: line-tables-only + RUST_BACKTRACE: 1 + RUST_LIB_BACKTRACE: 1 + +jobs: + IcebergRustSDK: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + with: + submodules: true + - name: Free disk space + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache + sudo apt-get clean + df -h / + - uses: prefix-dev/setup-pixi@v0.8.1 + with: + pixi-version: latest + - uses: Swatinem/rust-cache@v2 + - name: Run Rust Iceberg SDK acceptance + run: pixi run -e iceberg-e2e test-rust-iceberg-e2e + - name: Upload failure logs + if: failure() + uses: actions/upload-artifact@v4 + with: + name: runtime-iceberg-rust-sdk-${{ github.run_attempt }} + path: .crowdb-runtime/ + if-no-files-found: ignore + retention-days: 7 + - name: Clean subprocesses + if: always() + run: pixi run clean-env diff --git a/.github/workflows/release-container.yml b/.github/workflows/release-container.yml index cc4efcbda..01d96adc7 100644 --- a/.github/workflows/release-container.yml +++ b/.github/workflows/release-container.yml @@ -3,10 +3,15 @@ name: Publish crowdb-iceberge docker container on: workflow_dispatch: inputs: - tag: - description: Existing Git release tag to publish - required: true + candidate_sha: + description: Expected main commit (optional when started manually) + required: false type: string + include_symbols: + description: Build and attach the large exact-build symbol archive + required: false + default: false + type: boolean concurrency: group: crowdb-iceberg-single-node-preview-release @@ -19,15 +24,19 @@ jobs: CARGO_INCREMENTAL: "0" CARGO_PROFILE_DEV_DEBUG: line-tables-only CARGO_PROFILE_TEST_DEBUG: line-tables-only + CROWDB_PACKAGE_SYMBOLS: ${{ inputs.include_symbols && '1' || '0' }} permissions: contents: read + actions: read outputs: version: ${{ steps.source.outputs.version }} revision: ${{ steps.source.outputs.revision }} + tag: ${{ steps.source.outputs.tag }} + runtime_sha256: ${{ steps.runtime_digest.outputs.sha256 }} steps: - uses: actions/checkout@v4 with: - ref: ${{ inputs.tag }} + ref: ${{ github.sha }} fetch-depth: 0 submodules: true - uses: prefix-dev/setup-pixi@v0.8.1 @@ -38,17 +47,17 @@ jobs: - name: Verify release source id: source env: - RELEASE_TAG: ${{ inputs.tag }} - GH_TOKEN: ${{ github.token }} + CANDIDATE_SHA: ${{ inputs.candidate_sha }} run: | pixi run bash -euc ' [[ "$GITHUB_REPOSITORY" == buzzcrow/crowdb ]] - [[ "$RELEASE_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+(-[0-9A-Za-z.-]+)?$ ]] - [[ "$RELEASE_TAG" == "v$(cat VERSION)" ]] - revision=$(git rev-parse --verify "refs/tags/$RELEASE_TAG^{commit}") - [[ "$revision" == "$(git rev-parse HEAD)" ]] - [[ "$(gh release view "$RELEASE_TAG" --json isDraft --jq .isDraft)" == false ]] - printf "version=%s\nrevision=%s\n" "${RELEASE_TAG#v}" "$revision" >> "$GITHUB_OUTPUT" + [[ "$GITHUB_REF" == refs/heads/main ]] + version=$(cat VERSION) + [[ "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]] + revision=$(git rev-parse HEAD) + [[ "$revision" == "$GITHUB_SHA" ]] + [[ -z "$CANDIDATE_SHA" || "$revision" == "$CANDIDATE_SHA" ]] + printf "version=%s\nrevision=%s\ntag=v%s\n" "$version" "$revision" "$version" >> "$GITHUB_OUTPUT" ' - name: Free disk space run: | @@ -59,39 +68,52 @@ jobs: env: CROWDB_PREVIEW_TEST_ARTIFACTS: ${{ runner.temp }}/crowdb-preview-artifacts run: pixi run test-single-node-container - - name: Free preview build artifacts before acceptance tests - run: | - pixi run docker image rm crowdb-iceberg-single-node:dev - pixi run docker builder prune --all --force - pixi run cargo clean --release - pixi run df -h / - - name: Run S3 client acceptance - run: pixi run clean-env && pixi run -e s3-e2e test-boto3-e2e - - name: Run Iceberg client acceptance - run: pixi run clean-env && pixi run -e iceberg-e2e test-pyiceberg-e2e - - name: Run console acceptance - run: pixi run clean-env && pixi run test-console - - name: Require installed system browser + - name: Require CI success for the release commit + timeout-minutes: 90 + env: + GH_TOKEN: ${{ github.token }} + REVISION: ${{ steps.source.outputs.revision }} run: | pixi run bash -euc ' - for browser in /snap/bin/chromium /usr/bin/chromium /usr/bin/chromium-browser /usr/bin/google-chrome /usr/bin/google-chrome-stable /usr/bin/microsoft-edge; do - [[ ! -x "$browser" ]] || exit 0 + for attempt in {1..180}; do + state=$(gh api "repos/$GITHUB_REPOSITORY/actions/workflows/ci.yml/runs?head_sha=$REVISION&branch=main&event=push&per_page=20" \ + --jq ".workflow_runs | map(select(.head_sha == \"$REVISION\" and .head_branch == \"main\" and .event == \"push\")) | sort_by(.run_number, .run_attempt) | last | if . == null then \"pending\" else .status + \"/\" + (.conclusion // \"pending\") end") + case "$state" in + completed/success) exit 0 ;; + completed/*) echo "CI failed for $REVISION: $state" >&2; exit 1 ;; + esac + echo "Waiting for CI on $REVISION: $state" + sleep 30 done - echo "Release runner requires an installed system browser" >&2 + echo "Timed out waiting for CI on $REVISION" >&2 exit 1 ' - - name: Run console UI acceptance - run: pixi run clean-env && pixi run test-console-ui - - name: Check Rust formatting and lint - run: pixi run rs-fmt-check && pixi run rs-lint + - name: Verify exact-build source-line symbols + if: inputs.include_symbols + run: pixi run -- python tools/ci-checks/check-container-symbols.py - name: Archive verified runtime files run: pixi run tar -C target/container-runtime -czf target/container-runtime.tar.gz . + - name: Record verified runtime digest + run: pixi run bash -euc 'printf "sha256=%s\n" "$(sha256sum target/container-runtime.tar.gz | cut -d " " -f1)" >> "$GITHUB_OUTPUT"' + id: runtime_digest + - name: Archive exact-build symbols + if: inputs.include_symbols + run: pixi run tar -C target/container-symbols -I zstd -cf "target/crowdb-symbols-${{ steps.source.outputs.tag }}-git-${{ steps.source.outputs.revision }}-linux-amd64.tar.zst" . - uses: actions/upload-artifact@v4 with: name: verified-container-runtime path: target/container-runtime.tar.gz compression-level: 0 retention-days: 7 + - uses: actions/upload-artifact@v4 + id: symbols_artifact + if: inputs.include_symbols + continue-on-error: true + with: + name: verified-container-symbols + path: target/crowdb-symbols-*.tar.zst + compression-level: 0 + retention-days: 7 - name: Upload preview failure logs if: failure() uses: actions/upload-artifact@v4 @@ -106,20 +128,20 @@ jobs: runs-on: ubuntu-24.04 environment: DockerHub permissions: - contents: read + contents: write id-token: write steps: - uses: actions/checkout@v4 with: - ref: ${{ inputs.tag }} + ref: ${{ needs.verify.outputs.revision }} fetch-depth: 0 submodules: true - uses: prefix-dev/setup-pixi@v0.8.1 with: pixi-version: latest - - name: Require publication credentials and unused immutable tags + - name: Require publication credentials and current release source env: - RELEASE_TAG: ${{ inputs.tag }} + RELEASE_TAG: ${{ needs.verify.outputs.tag }} REVISION: ${{ needs.verify.outputs.revision }} DOCKERHUB_USERNAME: ${{ vars.DOCKERHUB_USERNAME }} DOCKERHUB_TOKEN: ${{ secrets.DOCKERHUB_TOKEN }} @@ -127,27 +149,86 @@ jobs: pixi run bash -euc ' [[ -n "$DOCKERHUB_USERNAME" && -n "$DOCKERHUB_TOKEN" ]] [[ "$(git rev-parse HEAD)" == "$REVISION" ]] - for tag in "$RELEASE_TAG" "git-$REVISION"; do - status=$(curl --silent --show-error --output /dev/null --write-out "%{http_code}" \ - "https://hub.docker.com/v2/namespaces/crowdb/repositories/crowdb-iceberg/tags/$tag") - [[ "$status" == 404 ]] || { echo "Immutable tag $tag is present or registry unavailable (HTTP $status)" >&2; exit 1; } - done + [[ "$(git ls-remote origin refs/heads/main | cut -f1)" == "$REVISION" ]] + [[ "$RELEASE_TAG" == "v$(cat VERSION)" ]] ' - uses: docker/setup-buildx-action@v4 - uses: actions/download-artifact@v4 with: name: verified-container-runtime path: target + - uses: actions/download-artifact@v4 + id: symbols_download + if: inputs.include_symbols + continue-on-error: true + with: + name: verified-container-symbols + path: target - name: Extract verified runtime files + env: + RUNTIME_SHA256: ${{ needs.verify.outputs.runtime_sha256 }} run: | + pixi run bash -euc '[[ "$(sha256sum target/container-runtime.tar.gz | cut -d " " -f1)" == "$RUNTIME_SHA256" ]]' pixi run bash -euc 'mkdir -p target/container-runtime && tar -C target/container-runtime -xzf target/container-runtime.tar.gz' pixi run rm target/container-runtime.tar.gz - uses: docker/login-action@v4 with: username: ${{ vars.DOCKERHUB_USERNAME }} password: ${{ secrets.DOCKERHUB_TOKEN }} + - name: Check immutable registry tags for safe retry + id: registry + env: + RELEASE_TAG: ${{ needs.verify.outputs.tag }} + REVISION: ${{ needs.verify.outputs.revision }} + VERSION: ${{ needs.verify.outputs.version }} + RUNTIME_SHA256: ${{ needs.verify.outputs.runtime_sha256 }} + run: | + pixi run bash -euc ' + for tag in "$RELEASE_TAG" "git-$REVISION"; do + status=$(curl --silent --show-error --output /dev/null --write-out "%{http_code}" \ + "https://hub.docker.com/v2/namespaces/crowdb/repositories/crowdb-iceberg/tags/$tag") + case "$status" in + 200|404) printf "%s=%s\n" "$tag" "$status" ;; + *) echo "Registry check failed for $tag (HTTP $status)" >&2; exit 1 ;; + esac + if [[ "$tag" == "$RELEASE_TAG" ]]; then version_status=$status; else revision_status=$status; fi + done + if [[ "$version_status" == 404 && "$revision_status" == 404 ]]; then + echo "reuse=false" >> "$GITHUB_OUTPUT" + exit 0 + fi + [[ "$version_status" == 200 && "$revision_status" == 200 ]] || { + echo "Only one immutable registry tag exists" >&2; exit 1; + } + image=docker.io/crowdb/crowdb-iceberg + version_digest=$(docker buildx imagetools inspect "$image:$RELEASE_TAG" --format "{{json .Manifest}}" | jq -er .digest) + revision_digest=$(docker buildx imagetools inspect "$image:git-$REVISION" --format "{{json .Manifest}}" | jq -er .digest) + [[ "$version_digest" == "$revision_digest" ]] + docker pull --platform linux/amd64 "$image:$RELEASE_TAG" + [[ "$(docker image inspect --format "{{index .Config.Labels \"org.opencontainers.image.revision\"}}" "$image:$RELEASE_TAG")" == "$REVISION" ]] + [[ "$(docker image inspect --format "{{index .Config.Labels \"org.opencontainers.image.version\"}}" "$image:$RELEASE_TAG")" == "$VERSION" ]] + [[ "$(docker image inspect --format "{{index .Config.Labels \"org.crowdb.runtime.sha256\"}}" "$image:$RELEASE_TAG")" == "$RUNTIME_SHA256" ]] + printf "reuse=true\ndigest=%s\n" "$version_digest" >> "$GITHUB_OUTPUT" + ' + - name: Create release tag after verification + env: + RELEASE_TAG: ${{ needs.verify.outputs.tag }} + REVISION: ${{ needs.verify.outputs.revision }} + run: | + pixi run bash -euc ' + git config user.name github-actions[bot] + git config user.email 41898282+github-actions[bot]@users.noreply.github.com + if git ls-remote --exit-code origin "refs/tags/$RELEASE_TAG" >/dev/null; then + git fetch origin "refs/tags/$RELEASE_TAG:refs/tags/$RELEASE_TAG" + [[ "$(git rev-parse "refs/tags/$RELEASE_TAG^{commit}")" == "$REVISION" ]] + else + git tag -a "$RELEASE_TAG" -m "$RELEASE_TAG" "$REVISION" + git push origin "refs/tags/$RELEASE_TAG" + fi + ' - name: Build and publish signed-source image with attestations id: build + if: steps.registry.outputs.reuse != 'true' uses: docker/build-push-action@v7 with: context: target/container-runtime @@ -159,11 +240,37 @@ jobs: build-args: | SOURCE_REVISION=${{ needs.verify.outputs.revision }} PREVIEW_VERSION=${{ needs.verify.outputs.version }} + RUNTIME_SHA256=${{ needs.verify.outputs.runtime_sha256 }} tags: | - docker.io/crowdb/crowdb-iceberg:${{ inputs.tag }} + docker.io/crowdb/crowdb-iceberg:${{ needs.verify.outputs.tag }} docker.io/crowdb/crowdb-iceberg:git-${{ needs.verify.outputs.revision }} - uses: sigstore/cosign-installer@v4.1.2 - name: Sign published digest env: - DIGEST: ${{ steps.build.outputs.digest }} + DIGEST: ${{ steps.registry.outputs.reuse == 'true' && steps.registry.outputs.digest || steps.build.outputs.digest }} run: pixi run cosign sign --yes "docker.io/crowdb/crowdb-iceberg@$DIGEST" + - name: Publish verified GitHub Release + env: + GH_TOKEN: ${{ github.token }} + RELEASE_TAG: ${{ needs.verify.outputs.tag }} + run: | + pixi run bash -euc ' + if gh release view "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --json isDraft --jq .isDraft >/dev/null 2>&1; then + [[ "$(gh release view "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --json isDraft --jq .isDraft)" == true ]] + else + gh release create "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --verify-tag --generate-notes --draft + fi + gh release edit "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --draft=false + ' + - name: Attach exact-build symbols to GitHub Release + id: symbol_upload + if: inputs.include_symbols && steps.symbols_download.outcome == 'success' + continue-on-error: true + env: + GH_TOKEN: ${{ github.token }} + RELEASE_TAG: ${{ needs.verify.outputs.tag }} + REVISION: ${{ needs.verify.outputs.revision }} + run: pixi run gh release upload "$RELEASE_TAG" "target/crowdb-symbols-$RELEASE_TAG-git-$REVISION-linux-amd64.tar.zst" --repo "$GITHUB_REPOSITORY" + - name: Report optional symbol upload failure + if: inputs.include_symbols && (steps.symbols_download.outcome == 'failure' || steps.symbol_upload.outcome == 'failure') + run: echo '::warning::The image and release were published, but the optional symbol archive could not be uploaded.' diff --git a/Cargo.lock b/Cargo.lock index 6e4c7ca91..ae46c778c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -628,6 +628,7 @@ dependencies = [ "chrono", "crc32fast", "crowdb-access-iceberg", + "crowdb-access-multipart", "crowdb-chunk-client", "crowdb-chunk-kv-client", "crowdb-common", @@ -650,6 +651,16 @@ dependencies = [ "zstd", ] +[[package]] +name = "crowdb-access-multipart" +version = "0.1.0-dev" +dependencies = [ + "crowdb-protocol", + "md-5", + "serde", + "thiserror 2.0.18", +] + [[package]] name = "crowdb-access-s3" version = "0.1.0-dev" @@ -661,6 +672,7 @@ dependencies = [ "base64", "bincode", "chrono", + "crowdb-access-multipart", "crowdb-chunk-client", "crowdb-chunk-kv-client", "crowdb-protocol", @@ -672,6 +684,7 @@ dependencies = [ "md5", "percent-encoding", "rand 0.8.6", + "serde", "sha2", "subtle", "thiserror 2.0.18", diff --git a/Cargo.toml b/Cargo.toml index 0040296b1..3230db57c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -26,6 +26,7 @@ members = [ "container/crowdb-monitor", "lib/crowdb-access-s3", "lib/crowdb-access-iceberg", + "lib/crowdb-access-multipart", ] exclude = ["third-party/hyper"] # : crowdb-tree/ffi moved from `exclude` into `members` now that diff --git a/app/crowdb-access-server/src/iceberg/file_selection.rs b/app/crowdb-access-server/src/iceberg/file_selection.rs index 2bb6c4cf5..26525ffff 100644 --- a/app/crowdb-access-server/src/iceberg/file_selection.rs +++ b/app/crowdb-access-server/src/iceberg/file_selection.rs @@ -1,28 +1,11 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +pub use crate::multipart_complete::{CompletePart, CompleteRequestError, CompleteSelection}; use crowdb_access_iceberg::catalog::CatalogError; use crowdb_access_iceberg::file::{ MultipartPhase, MultipartRepository, MultipartSelection, MultipartSession, SelectedPart, }; -use quick_xml::events::Event; -use quick_xml::Reader; - -const MAX_COMPLETE_XML_BYTES: usize = 2 * 1024 * 1024; -const MAX_COMPLETE_PARTS: usize = 10_000; -const S3_NAMESPACE: &[u8] = b"http://s3.amazonaws.com/doc/2006-03-01/"; - -#[derive(Clone, Debug, Eq, PartialEq)] -pub struct CompletePart { - pub number: u16, - pub etag: String, -} - -#[derive(Clone, Debug, Eq, PartialEq)] -pub struct CompleteSelection { - parts: Vec, -} - -#[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] -#[error("invalid multipart completion XML")] -pub struct CompleteRequestError; #[derive(Debug, thiserror::Error)] pub enum CompleteResolveError { @@ -35,96 +18,6 @@ pub enum CompleteResolveError { } impl CompleteSelection { - /// Parses a bounded S3 `CompleteMultipartUpload` body. The caller must match - /// each selected digest to the current durable part revision before freezing. - /// # Errors - /// Rejects malformed XML, extra fields and unordered or duplicate parts. - pub fn parse(bytes: &[u8]) -> Result { - if bytes.is_empty() || bytes.len() > MAX_COMPLETE_XML_BYTES { - return Err(CompleteRequestError); - } - let mut reader = Reader::from_reader(bytes); - let mut state = State::Start; - let mut parts = Vec::new(); - let mut number = None; - let mut digest = None; - let mut etag = Vec::new(); - loop { - match reader.read_event().map_err(|_| CompleteRequestError)? { - Event::Decl(_) if state == State::Start => {} - Event::Start(event) if valid_attributes(state, &event)? => { - state = match (state, event.name().as_ref()) { - (State::Start, b"CompleteMultipartUpload") => State::Root, - (State::Root, b"Part") if parts.len() < MAX_COMPLETE_PARTS => State::Part, - (State::Part, b"PartNumber") if number.is_none() => State::Number, - (State::Part, b"ETag") if digest.is_none() => State::Etag, - _ => return Err(CompleteRequestError), - }; - } - Event::Text(event) => match state { - State::Number if number.is_none() => { - let value: &[u8] = event.as_ref(); - if value.is_empty() || !value.iter().all(u8::is_ascii_digit) { - return Err(CompleteRequestError); - } - number = Some( - std::str::from_utf8(value) - .map_err(|_| CompleteRequestError)? - .parse::() - .map_err(|_| CompleteRequestError)?, - ); - } - State::Etag => append_etag(&mut etag, &event)?, - State::Start | State::Root | State::Part | State::Done - if event.iter().all(u8::is_ascii_whitespace) => {} - _ => return Err(CompleteRequestError), - }, - Event::GeneralRef(event) if state == State::Etag => { - if event.len() > 16 { - return Err(CompleteRequestError); - } - let name = std::str::from_utf8(&event).map_err(|_| CompleteRequestError)?; - let encoded = format!("&{name};"); - let decoded = quick_xml::escape::unescape(&encoded).map_err(|_| CompleteRequestError)?; - append_etag(&mut etag, decoded.as_bytes())?; - } - Event::End(event) => { - state = match (state, event.name().as_ref()) { - (State::Number, b"PartNumber") if number.is_some() => State::Part, - (State::Etag, b"ETag") => { - digest = Some(parse_etag(&etag)?); - etag.clear(); - State::Part - } - (State::Part, b"Part") => { - let number = number.take().ok_or(CompleteRequestError)?; - let etag = digest.take().ok_or(CompleteRequestError)?; - if number == 0 - || number > 10_000 - || parts - .last() - .is_some_and(|part: &CompletePart| part.number >= number) - { - return Err(CompleteRequestError); - } - parts.push(CompletePart { number, etag }); - State::Root - } - (State::Root, b"CompleteMultipartUpload") if !parts.is_empty() => State::Done, - _ => return Err(CompleteRequestError), - }; - } - Event::Eof if state == State::Done => return Ok(Self { parts }), - _ => return Err(CompleteRequestError), - } - } - } - - #[must_use] - pub fn parts(&self) -> &[CompletePart] { - &self.parts - } - /// Resolves the selected parts against one current durable session snapshot. /// # Errors /// Rejects missing, replaced or differently hashed parts and storage failures. @@ -133,14 +26,14 @@ impl CompleteSelection { repository: &MultipartRepository, session: &MultipartSession, ) -> Result { - if self.parts.len() > usize::from(session.limits.max_parts) { + if self.parts().len() > usize::from(session.limits.max_parts) { return Err(CompleteResolveError::InvalidPart); } if session.completion.is_some() { let frozen = repository.load_selection(session).await?; if let Some(snapshots) = frozen.snapshots() { - if self.parts.len() != snapshots.len() - || self.parts.iter().zip(frozen.parts().iter().zip(snapshots)).any( + if self.parts().len() != snapshots.len() + || self.parts().iter().zip(frozen.parts().iter().zip(snapshots)).any( |(requested, (selected, snapshot))| { requested.number != selected.number || requested.etag != snapshot.etag }, @@ -151,9 +44,9 @@ impl CompleteSelection { return Ok(frozen); } } - let mut selected = Vec::with_capacity(self.parts.len()); - let mut parts = Vec::with_capacity(self.parts.len()); - for (index, requested) in self.parts.iter().enumerate() { + let mut selected = Vec::with_capacity(self.parts().len()); + let mut parts = Vec::with_capacity(self.parts().len()); + for (index, requested) in self.parts().iter().enumerate() { let part = if session.phase == MultipartPhase::Open { repository.part_for_upload(session, requested.number).await? } else { @@ -163,7 +56,7 @@ impl CompleteSelection { if part.etag() != requested.etag { return Err(CompleteResolveError::InvalidPart); } - if index + 1 < self.parts.len() && part.length() < 5 * 1024 * 1024 { + if index + 1 < self.parts().len() && part.length() < 5 * 1024 * 1024 { return Err(CompleteResolveError::EntityTooSmall); } selected.push(SelectedPart { @@ -180,59 +73,3 @@ impl CompleteSelection { } } } - -#[derive(Clone, Copy, Eq, PartialEq)] -enum State { - Start, - Root, - Part, - Number, - Etag, - Done, -} - -fn parse_etag(bytes: &[u8]) -> Result { - let hex = bytes - .strip_prefix(b"\"") - .and_then(|bytes| bytes.strip_suffix(b"\"")) - .ok_or(CompleteRequestError)?; - if hex.len() != 32 && hex.len() != 64 { - return Err(CompleteRequestError); - } - for pair in hex.chunks_exact(2) { - let _ = (hex_digit(pair[0])? << 4) | hex_digit(pair[1])?; - } - String::from_utf8(hex.to_vec()).map_err(|_| CompleteRequestError) -} - -fn append_etag(etag: &mut Vec, bytes: &[u8]) -> Result<(), CompleteRequestError> { - if etag.len().saturating_add(bytes.len()) > 66 { - return Err(CompleteRequestError); - } - etag.extend_from_slice(bytes); - Ok(()) -} - -fn hex_digit(byte: u8) -> Result { - match byte { - b'0'..=b'9' => Ok(byte - b'0'), - b'a'..=b'f' => Ok(byte - b'a' + 10), - _ => Err(CompleteRequestError), - } -} - -fn valid_attributes( - state: State, - event: &quick_xml::events::BytesStart<'_>, -) -> Result { - let mut attributes = event.attributes(); - let Some(attribute) = attributes.next() else { - return Ok(true); - }; - let attribute = attribute.map_err(|_| CompleteRequestError)?; - Ok(state == State::Start - && event.name().as_ref() == b"CompleteMultipartUpload" - && attribute.key.as_ref() == b"xmlns" - && attribute.value.as_ref() == S3_NAMESPACE - && attributes.next().is_none()) -} diff --git a/app/crowdb-access-server/src/lib.rs b/app/crowdb-access-server/src/lib.rs index d7b637f7c..3029821ef 100644 --- a/app/crowdb-access-server/src/lib.rs +++ b/app/crowdb-access-server/src/lib.rs @@ -6,6 +6,7 @@ pub mod config; mod http_receive; pub mod iceberg; +mod multipart_complete; #[cfg(feature = "s3")] pub mod credentials; diff --git a/app/crowdb-access-server/src/main.rs b/app/crowdb-access-server/src/main.rs index dfe8eb8f4..7c092373e 100644 --- a/app/crowdb-access-server/src/main.rs +++ b/app/crowdb-access-server/src/main.rs @@ -116,6 +116,7 @@ async fn run_s3(access_config: &AccessConfig) -> Result<(), Box Result<(), Box Result<(), Box) -> tokio::task::JoinHandle<()> { + tokio::spawn(async move { + let mut interval = tokio::time::interval(Duration::from_secs(60 * 60)); + loop { + interval.tick().await; + match operations.expire_multipart_uploads().await { + Ok(0) => {} + Ok(count) => tracing::debug!(count, "expired S3 multipart sessions"), + Err(error) => tracing::warn!(?error, "S3 multipart expiry sweep deferred"), + } + } + }) +} + #[cfg(feature = "s3")] async fn authenticate_s3( access: &AccessConfig, diff --git a/app/crowdb-access-server/src/multipart_complete.rs b/app/crowdb-access-server/src/multipart_complete.rs new file mode 100644 index 000000000..79025f25e --- /dev/null +++ b/app/crowdb-access-server/src/multipart_complete.rs @@ -0,0 +1,184 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Bounded S3 completion XML shared by the Iceberg and S3 protocol modules. + +use quick_xml::events::Event; +use quick_xml::Reader; + +const MAX_COMPLETE_XML_BYTES: usize = 2 * 1024 * 1024; +const MAX_COMPLETE_PARTS: usize = 10_000; +const S3_NAMESPACE: &[u8] = b"http://s3.amazonaws.com/doc/2006-03-01/"; + +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct CompletePart { + pub number: u16, + pub etag: String, +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct CompleteSelection { + parts: Vec, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] +pub enum CompleteRequestError { + #[error("invalid multipart completion XML")] + InvalidRequest, + #[error("multipart parts are not in ascending order")] + InvalidPartOrder, +} + +impl CompleteSelection { + /// Parses a bounded S3 `CompleteMultipartUpload` body. The caller must match + /// each selected digest to the current durable part revision before freezing. + /// # Errors + /// Rejects malformed XML, extra fields and unordered or duplicate parts. + pub fn parse(bytes: &[u8]) -> Result { + if bytes.is_empty() || bytes.len() > MAX_COMPLETE_XML_BYTES { + return Err(CompleteRequestError::InvalidRequest); + } + let mut reader = Reader::from_reader(bytes); + let mut state = State::Start; + let mut parts = Vec::new(); + let mut number = None; + let mut digest = None; + let mut etag = Vec::new(); + loop { + match reader + .read_event() + .map_err(|_| CompleteRequestError::InvalidRequest)? + { + Event::Decl(_) if state == State::Start => {} + Event::Start(event) if valid_attributes(state, &event)? => { + state = match (state, event.name().as_ref()) { + (State::Start, b"CompleteMultipartUpload") => State::Root, + (State::Root, b"Part") if parts.len() < MAX_COMPLETE_PARTS => State::Part, + (State::Part, b"PartNumber") if number.is_none() => State::Number, + (State::Part, b"ETag") if digest.is_none() => State::Etag, + _ => return Err(CompleteRequestError::InvalidRequest), + }; + } + Event::Text(event) => match state { + State::Number if number.is_none() => { + let value: &[u8] = event.as_ref(); + if value.is_empty() || !value.iter().all(u8::is_ascii_digit) { + return Err(CompleteRequestError::InvalidRequest); + } + number = Some( + std::str::from_utf8(value) + .map_err(|_| CompleteRequestError::InvalidRequest)? + .parse::() + .map_err(|_| CompleteRequestError::InvalidRequest)?, + ); + } + State::Etag => append_etag(&mut etag, &event)?, + State::Start | State::Root | State::Part | State::Done + if event.iter().all(u8::is_ascii_whitespace) => {} + _ => return Err(CompleteRequestError::InvalidRequest), + }, + Event::GeneralRef(event) if state == State::Etag => { + if event.len() > 16 { + return Err(CompleteRequestError::InvalidRequest); + } + let name = + std::str::from_utf8(&event).map_err(|_| CompleteRequestError::InvalidRequest)?; + let encoded = format!("&{name};"); + let decoded = quick_xml::escape::unescape(&encoded) + .map_err(|_| CompleteRequestError::InvalidRequest)?; + append_etag(&mut etag, decoded.as_bytes())?; + } + Event::End(event) => { + state = match (state, event.name().as_ref()) { + (State::Number, b"PartNumber") if number.is_some() => State::Part, + (State::Etag, b"ETag") => { + digest = Some(parse_etag(&etag)?); + etag.clear(); + State::Part + } + (State::Part, b"Part") => { + let number = number.take().ok_or(CompleteRequestError::InvalidRequest)?; + let etag = digest.take().ok_or(CompleteRequestError::InvalidRequest)?; + if number == 0 || number > 10_000 { + return Err(CompleteRequestError::InvalidRequest); + } + if parts + .last() + .is_some_and(|part: &CompletePart| part.number >= number) + { + return Err(CompleteRequestError::InvalidPartOrder); + } + parts.push(CompletePart { number, etag }); + State::Root + } + (State::Root, b"CompleteMultipartUpload") if !parts.is_empty() => State::Done, + _ => return Err(CompleteRequestError::InvalidRequest), + }; + } + Event::Eof if state == State::Done => return Ok(Self { parts }), + _ => return Err(CompleteRequestError::InvalidRequest), + } + } + } + + #[must_use] + pub fn parts(&self) -> &[CompletePart] { + &self.parts + } +} + +#[derive(Clone, Copy, Eq, PartialEq)] +enum State { + Start, + Root, + Part, + Number, + Etag, + Done, +} + +fn parse_etag(bytes: &[u8]) -> Result { + let hex = bytes + .strip_prefix(b"\"") + .and_then(|bytes| bytes.strip_suffix(b"\"")) + .ok_or(CompleteRequestError::InvalidRequest)?; + if hex.len() != 32 && hex.len() != 64 { + return Err(CompleteRequestError::InvalidRequest); + } + for pair in hex.chunks_exact(2) { + let _ = (hex_digit(pair[0])? << 4) | hex_digit(pair[1])?; + } + String::from_utf8(hex.to_vec()).map_err(|_| CompleteRequestError::InvalidRequest) +} + +fn append_etag(etag: &mut Vec, bytes: &[u8]) -> Result<(), CompleteRequestError> { + if etag.len().saturating_add(bytes.len()) > 66 { + return Err(CompleteRequestError::InvalidRequest); + } + etag.extend_from_slice(bytes); + Ok(()) +} + +fn hex_digit(byte: u8) -> Result { + match byte { + b'0'..=b'9' => Ok(byte - b'0'), + b'a'..=b'f' => Ok(byte - b'a' + 10), + _ => Err(CompleteRequestError::InvalidRequest), + } +} + +fn valid_attributes( + state: State, + event: &quick_xml::events::BytesStart<'_>, +) -> Result { + let mut attributes = event.attributes(); + let Some(attribute) = attributes.next() else { + return Ok(true); + }; + let attribute = attribute.map_err(|_| CompleteRequestError::InvalidRequest)?; + Ok(state == State::Start + && event.name().as_ref() == b"CompleteMultipartUpload" + && attribute.key.as_ref() == b"xmlns" + && attribute.value.as_ref() == S3_NAMESPACE + && attributes.next().is_none()) +} diff --git a/app/crowdb-access-server/src/s3.rs b/app/crowdb-access-server/src/s3.rs index 8c8201029..13e2a6563 100644 --- a/app/crowdb-access-server/src/s3.rs +++ b/app/crowdb-access-server/src/s3.rs @@ -23,6 +23,7 @@ mod dispatcher; mod operations; pub use crate::http_receive::install_body_receive_provider; +pub use crate::multipart_complete::{CompletePart, CompleteRequestError, CompleteSelection}; pub use dispatcher::S3Dispatcher; pub use operations::{ProductionS3Operations, S3Operations, S3OperationsFuture, S3ServiceConfig}; diff --git a/app/crowdb-access-server/src/s3/dispatcher.rs b/app/crowdb-access-server/src/s3/dispatcher.rs index 90cd14586..5459b8cdf 100644 --- a/app/crowdb-access-server/src/s3/dispatcher.rs +++ b/app/crowdb-access-server/src/s3/dispatcher.rs @@ -296,7 +296,10 @@ fn defer_body_provider( factory: Option DeferredBodyReceiveProvider + Send + Sync>>, request: &mut Request, ) { - if operation == crowdb_access_s3::route::S3Operation::PutObject { + if matches!( + operation, + crowdb_access_s3::route::S3Operation::PutObject | crowdb_access_s3::route::S3Operation::UploadPart + ) { if let Some(factory) = factory { request.extensions_mut().insert(factory()); } diff --git a/app/crowdb-access-server/src/s3/operations.rs b/app/crowdb-access-server/src/s3/operations.rs index 8e6616621..7a0a4752b 100644 --- a/app/crowdb-access-server/src/s3/operations.rs +++ b/app/crowdb-access-server/src/s3/operations.rs @@ -39,6 +39,8 @@ use crate::storage::S3StorageClients; use super::{error_response, full_body, install_body_receive_provider, BoxError, ResponseBody}; use crowdb_access_s3::wire; +mod multipart; + const DEFAULT_LIST_LIMIT: usize = 1_000; const DEFAULT_LIST_SCAN_BYTES: usize = 4 * 1024 * 1024; @@ -143,6 +145,12 @@ impl ProductionS3Operations { S3Operation::GetObject => self.get_object(route, &request, &request_id).await, S3Operation::ListObjectsV2 => self.list_objects(route, &request).await, S3Operation::DeleteObject => self.delete_object(route).await, + S3Operation::CreateMultipartUpload => self.create_multipart_upload(route, &request).await, + S3Operation::UploadPart => self.upload_part(route, request).await, + S3Operation::ListParts => self.list_parts(route, &request).await, + S3Operation::CompleteMultipartUpload => self.complete_multipart_upload(route, request).await, + S3Operation::AbortMultipartUpload => self.abort_multipart_upload(route).await, + S3Operation::ListMultipartUploads => self.list_multipart_uploads(route, &request).await, }; self.record_dependency_outcome(operation, &result); result.unwrap_or_else(|code| { @@ -159,7 +167,10 @@ impl ProductionS3Operations { let Some(health) = &self.health else { return; }; - let uses_chunks = matches!(operation, S3Operation::PutObject | S3Operation::GetObject); + let uses_chunks = matches!( + operation, + S3Operation::PutObject | S3Operation::GetObject | S3Operation::UploadPart + ); match result { Ok(_) => { health.set_metadata(DependencyHealth::Ready); diff --git a/app/crowdb-access-server/src/s3/operations/multipart.rs b/app/crowdb-access-server/src/s3/operations/multipart.rs new file mode 100644 index 000000000..c96aa520a --- /dev/null +++ b/app/crowdb-access-server/src/s3/operations/multipart.rs @@ -0,0 +1,439 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Authenticated S3 multipart operations over durable session and part records. + +use crowdb_access_s3::bucket; +use crowdb_access_s3::error::S3ErrorCode; +use crowdb_access_s3::integrity::{IntegrityError, SinglePartIntegrity}; +use crowdb_access_s3::metadata::{ + new_upload_id, CompletionPart, MultipartPartRecord, MultipartPhase, MultipartRepository, + MultipartRepositoryError, MultipartSessionRecord, +}; +use crowdb_access_s3::route::{S3Operation, S3Route}; +use crowdb_access_s3::streaming::{ + write_body_with_checksums_buffered, write_native_body_with_checksums_metered, +}; +use crowdb_access_s3::wire; +use crowdb_chunk_client::ChunkIoWriter; +use http_body_util::BodyExt as _; +use hyper::body::Bytes; +use hyper::body::Incoming; +use hyper::header::ETAG; +use hyper::{Request, Response, StatusCode}; + +use super::{ + content_length, full_body, install_body_receive_provider, map_put_outcome, required_bucket, required_key, + response, strict_header, unix_millis, xml_response, ProductionS3Operations, Query, ResponseBody, +}; +use crate::multipart_complete::{CompleteRequestError, CompleteSelection}; + +const MAX_PARTS: u16 = 10_000; +const MAX_PART_BYTES: u64 = 5 * 1024 * 1024 * 1024; +const MAX_OBJECT_BYTES: u64 = 5 * 1024 * 1024 * 1024 * 1024; +const MAX_STAGED_BYTES: u64 = 10_000 * MAX_PART_BYTES; +const UPLOAD_LIFETIME_MS: u64 = 7 * 24 * 60 * 60 * 1000; +const MAX_COMPLETE_BODY: usize = 2 * 1024 * 1024; + +impl ProductionS3Operations { + /// Walks bounded metadata pages and marks expired sessions terminal. + /// + /// # Errors + /// Defers the sweep when bucket or upload metadata is unavailable. + pub async fn expire_multipart_uploads(&self) -> Result { + let buckets = bucket::list_buckets(&self.storage.metadata, &self.config.tenant, 1_001) + .await + .map_err(|_| S3ErrorCode::ServiceUnavailable)?; + if buckets.len() > 1_000 { + return Err(S3ErrorCode::SlowDown); + } + let now = unix_millis(); + let mut expired = 0; + for bucket in buckets { + let mut cursor = None; + loop { + let page = self + .multipart() + .expire_page(bucket.bucket_id, cursor.as_deref(), now, 1_000) + .await + .map_err(|error| map_multipart_error(&error))?; + expired += page.expired; + let Some(next) = page.next else { + break; + }; + cursor = Some(next); + tokio::task::yield_now().await; + } + } + Ok(expired) + } + + fn multipart(&self) -> MultipartRepository { + MultipartRepository::new(self.storage.metadata.clone(), self.config.tenant.clone()) + } + + async fn multipart_identity(&self, route: &S3Route) -> Result { + let bucket = self.resolve_bucket(required_bucket(route)?).await?; + let key = required_key(route)?; + let upload_id = route.upload_id.ok_or(S3ErrorCode::InvalidRequest)?; + let session = self + .multipart() + .load_identity(bucket, key, &upload_id) + .await + .map_err(|error| map_multipart_error(&error))? + .ok_or(S3ErrorCode::NoSuchUpload)?; + if session.phase == MultipartPhase::Open && session.expires_ms <= unix_millis() { + let expired = self + .multipart() + .abort(&session) + .await + .map_err(|error| map_multipart_error(&error))?; + if route.operation != S3Operation::AbortMultipartUpload { + return Err(S3ErrorCode::NoSuchUpload); + } + return Ok(expired); + } + Ok(session) + } + + pub(super) async fn create_multipart_upload( + &self, + route: S3Route, + request: &Request, + ) -> Result, S3ErrorCode> { + let bucket_name = required_bucket(&route)?; + let key = required_key(&route)?; + let bucket_id = self.resolve_bucket(bucket_name).await?; + if content_length(request)?.is_some_and(|length| length != 0) { + return Err(S3ErrorCode::InvalidRequest); + } + let now = unix_millis(); + let session = MultipartSessionRecord { + bucket_id, + object_key: key.to_vec(), + upload_id: new_upload_id(now), + revision: 1, + phase: MultipartPhase::Open, + created_ms: now, + expires_ms: now + .checked_add(UPLOAD_LIFETIME_MS) + .ok_or(S3ErrorCode::InternalError)?, + content_type: strict_header(request, "content-type", S3ErrorCode::InvalidRequest)? + .unwrap_or("application/octet-stream") + .to_owned(), + max_parts: MAX_PARTS, + max_part_bytes: MAX_PART_BYTES, + max_object_bytes: MAX_OBJECT_BYTES, + max_staged_bytes: MAX_STAGED_BYTES, + part_count: 0, + staged_bytes: 0, + pending: None, + selection: None, + completion_request_digest: None, + publication_ms: None, + object_predecessor: None, + etag: None, + }; + self.multipart() + .begin(&session) + .await + .map_err(|error| map_multipart_error(&error))?; + xml_response(wire::create_multipart_upload( + bucket_name, + key, + &session.upload_id, + )) + } + + pub(super) async fn upload_part( + &self, + route: S3Route, + mut request: Request, + ) -> Result, S3ErrorCode> { + let session = self.multipart_identity(&route).await?; + let number = route.part_number.ok_or(S3ErrorCode::InvalidRequest)?; + let length = content_length(&request)?.ok_or(S3ErrorCode::InvalidRequest)?; + if length > session.max_part_bytes { + return Err(S3ErrorCode::InvalidRequest); + } + let content_md5 = + strict_header(&request, "content-md5", S3ErrorCode::InvalidDigest)?.map(str::to_owned); + let payload_sha256 = strict_header(&request, "x-amz-content-sha256", S3ErrorCode::InvalidRequest)? + .filter(|value| *value != "UNSIGNED-PAYLOAD") + .map(str::to_owned); + let mut route_key = self.config.tenant.as_bytes().to_vec(); + route_key.extend_from_slice(session.bucket_id.as_bytes()); + route_key.extend_from_slice(&session.upload_id); + route_key.extend_from_slice(&number.to_be_bytes()); + let mut writer = self.prepare_writer(Some(length), &route_key).await?; + let native_receiver = if writer.is_large() { + install_body_receive_provider(&mut request) + } else { + None + }; + if let Some(receiver) = &native_receiver { + receiver.enable_owner_handoff(); + } + let mut body = request.into_body(); + let written = if let Some(receiver) = native_receiver.as_deref() { + write_native_body_with_checksums_metered( + &mut body, + &mut writer, + receiver, + content_md5.as_deref(), + payload_sha256.as_deref(), + self.metrics.as_deref(), + ) + .await + } else { + let receive_bytes = usize::try_from(length) + .unwrap_or(1024 * 1024) + .clamp(1, 1024 * 1024); + write_body_with_checksums_buffered( + &mut body, + &mut writer, + content_md5.as_deref(), + payload_sha256.as_deref(), + receive_bytes, + self.metrics.as_deref(), + ) + .await + }; + let (etag, _) = match written { + Ok(value) => value, + Err(outcome) => { + let _ = writer.on_error().await; + return Err(map_put_outcome(&outcome)); + } + }; + let locations = writer + .on_finish() + .await + .map_err(|_| S3ErrorCode::ServiceUnavailable)?; + let actual: u64 = locations.iter().map(|location| location.logical_length).sum(); + if actual != length { + return Err(S3ErrorCode::InvalidRequest); + } + let part = MultipartPartRecord { + bucket_id: session.bucket_id, + upload_id: session.upload_id, + number, + revision: 1, + modified_ms: unix_millis(), + length, + raw_md5: parse_md5(&etag)?, + locations, + }; + let _saved = self + .multipart() + .put_stream_part(&session, &part, part.modified_ms) + .await + .map_err(|error| map_multipart_error(&error))? + .ok_or(S3ErrorCode::SlowDown)?; + Response::builder() + .status(StatusCode::OK) + .header(ETAG, format!("\"{etag}\"")) + .body(full_body(Vec::new().into())) + .map_err(|_| S3ErrorCode::InternalError) + } + + pub(super) async fn list_parts( + &self, + route: S3Route, + request: &Request, + ) -> Result, S3ErrorCode> { + let session = self.multipart_identity(&route).await?; + let query = Query::new(request.uri().query()); + let marker = parse_number(query.text("part-number-marker"), 0, 0, 10_000)?; + let limit = parse_number(query.text("max-parts"), 1_000, 1, 1_000)?; + let page = self + .multipart() + .list_parts(&session, marker, usize::from(limit)) + .await + .map_err(|error| map_multipart_error(&error))?; + xml_response(wire::list_multipart_parts( + required_bucket(&route)?, + required_key(&route)?, + &session.upload_id, + marker, + usize::from(limit), + &page, + )) + } + + pub(super) async fn list_multipart_uploads( + &self, + route: S3Route, + request: &Request, + ) -> Result, S3ErrorCode> { + let bucket_name = required_bucket(&route)?; + let bucket = self.resolve_bucket(bucket_name).await?; + let query = Query::new(request.uri().query()); + let prefix = query.bytes("prefix").unwrap_or_default(); + let key_marker = query.bytes("key-marker"); + let upload_marker = query + .text("upload-id-marker") + .as_deref() + .map(parse_upload_id) + .transpose()?; + let limit = parse_number(query.text("max-uploads"), 1_000, 1, 1_000)?; + let page = self + .multipart() + .list_uploads( + bucket, + &prefix, + key_marker.as_deref(), + upload_marker.as_ref(), + usize::from(limit), + unix_millis(), + ) + .await + .map_err(|error| map_multipart_error(&error))?; + xml_response(wire::list_multipart_uploads( + bucket_name, + &prefix, + key_marker.as_deref(), + upload_marker.as_ref(), + usize::from(limit), + &String::from_utf8_lossy(self.config.tenant.as_bytes()), + &page, + )) + } + + pub(super) async fn abort_multipart_upload( + &self, + route: S3Route, + ) -> Result, S3ErrorCode> { + let session = self.multipart_identity(&route).await?; + self.multipart() + .abort(&session) + .await + .map_err(|error| map_multipart_error(&error))?; + response(StatusCode::NO_CONTENT, Vec::new()) + } + + pub(super) async fn complete_multipart_upload( + &self, + route: S3Route, + request: Request, + ) -> Result, S3ErrorCode> { + let session = self.multipart_identity(&route).await?; + let location = request.uri().path().to_owned(); + let content_md5 = + strict_header(&request, "content-md5", S3ErrorCode::InvalidDigest)?.map(str::to_owned); + let payload_sha256 = strict_header(&request, "x-amz-content-sha256", S3ErrorCode::InvalidRequest)? + .filter(|value| *value != "UNSIGNED-PAYLOAD") + .map(str::to_owned); + let bytes = read_completion(request.into_body()).await?; + let mut integrity = SinglePartIntegrity::new(payload_sha256.is_some()); + integrity.update(&Bytes::copy_from_slice(&bytes)); + integrity + .finish_validated_checksums(content_md5.as_deref(), payload_sha256.as_deref()) + .map_err(map_integrity_error)?; + let selection = CompleteSelection::parse(&bytes).map_err(|error| match error { + CompleteRequestError::InvalidRequest => S3ErrorCode::InvalidRequest, + CompleteRequestError::InvalidPartOrder => S3ErrorCode::InvalidPartOrder, + })?; + let requested: Vec = selection + .parts() + .iter() + .map(|part| CompletionPart { + number: part.number, + etag: part.etag.clone(), + }) + .collect(); + let frozen = self + .multipart() + .freeze_completion(&session, &requested, unix_millis()) + .await + .map_err(|error| map_multipart_error(&error))? + .ok_or(S3ErrorCode::ServiceUnavailable)?; + let published = self + .multipart() + .publish_completion(&frozen) + .await + .map_err(|error| map_multipart_error(&error))?; + xml_response(wire::complete_multipart_upload( + &location, + required_bucket(&route)?, + required_key(&route)?, + published.etag.as_deref().ok_or(S3ErrorCode::InternalError)?, + )) + } +} + +fn map_multipart_error(error: &MultipartRepositoryError) -> S3ErrorCode { + match error { + MultipartRepositoryError::Key(_) | MultipartRepositoryError::Record(_) => S3ErrorCode::InvalidRequest, + MultipartRepositoryError::Store(_) => S3ErrorCode::ServiceUnavailable, + MultipartRepositoryError::Conflict => S3ErrorCode::NoSuchUpload, + MultipartRepositoryError::Busy | MultipartRepositoryError::ScanBudgetExhausted => { + S3ErrorCode::SlowDown + } + MultipartRepositoryError::InvalidPart => S3ErrorCode::InvalidPart, + MultipartRepositoryError::EntityTooSmall => S3ErrorCode::EntityTooSmall, + } +} + +fn map_integrity_error(error: IntegrityError) -> S3ErrorCode { + match error { + IntegrityError::InvalidDigest => S3ErrorCode::InvalidDigest, + IntegrityError::Mismatch => S3ErrorCode::BadDigest, + IntegrityError::InvalidPayloadDigest => S3ErrorCode::InvalidRequest, + IntegrityError::PayloadMismatch => S3ErrorCode::XAmzContentSHA256Mismatch, + } +} + +fn parse_number(value: Option, default: u16, minimum: u16, maximum: u16) -> Result { + value + .map_or(Ok(default), |value| value.parse::()) + .map_err(|_| S3ErrorCode::InvalidRequest) + .and_then(|number| { + (number >= minimum && number <= maximum) + .then_some(number) + .ok_or(S3ErrorCode::InvalidRequest) + }) +} + +fn parse_upload_id(value: &str) -> Result<[u8; 16], S3ErrorCode> { + if value.len() != 32 { + return Err(S3ErrorCode::InvalidRequest); + } + let mut id = [0; 16]; + for (byte, pair) in id.iter_mut().zip(value.as_bytes().chunks_exact(2)) { + *byte = u8::from_str_radix( + std::str::from_utf8(pair).map_err(|_| S3ErrorCode::InvalidRequest)?, + 16, + ) + .map_err(|_| S3ErrorCode::InvalidRequest)?; + } + (id != [0; 16]).then_some(id).ok_or(S3ErrorCode::InvalidRequest) +} + +fn parse_md5(value: &str) -> Result<[u8; 16], S3ErrorCode> { + if value.len() != 32 { + return Err(S3ErrorCode::InternalError); + } + let mut md5 = [0; 16]; + for (byte, pair) in md5.iter_mut().zip(value.as_bytes().chunks_exact(2)) { + *byte = u8::from_str_radix( + std::str::from_utf8(pair).map_err(|_| S3ErrorCode::InternalError)?, + 16, + ) + .map_err(|_| S3ErrorCode::InternalError)?; + } + Ok(md5) +} + +async fn read_completion(mut body: Incoming) -> Result, S3ErrorCode> { + let mut bytes = Vec::new(); + while let Some(frame) = body.frame().await { + let frame = frame.map_err(|_| S3ErrorCode::InvalidRequest)?; + let data = frame.into_data().map_err(|_| S3ErrorCode::InvalidRequest)?; + if bytes.len().saturating_add(data.len()) > MAX_COMPLETE_BODY { + return Err(S3ErrorCode::InvalidRequest); + } + bytes.extend_from_slice(&data); + } + Ok(bytes) +} diff --git a/app/crowdb-access-server/tests/common/iceberg_stack.rs b/app/crowdb-access-server/tests/common/iceberg_stack.rs index 3671dc9ce..96b98e292 100644 --- a/app/crowdb-access-server/tests/common/iceberg_stack.rs +++ b/app/crowdb-access-server/tests/common/iceberg_stack.rs @@ -166,6 +166,7 @@ async fn seed(cluster: &KvCluster) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![10], + ..Default::default() }, ) .await @@ -180,6 +181,7 @@ async fn seed(cluster: &KvCluster) { disk_group_ids: vec![100], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await @@ -193,6 +195,7 @@ async fn seed(cluster: &KvCluster) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: vec![disk], + name: String::new(), }, ) .await diff --git a/app/crowdb-access-server/tests/s3_e2e/basic.py b/app/crowdb-access-server/tests/s3_e2e/basic.py index 0b22c586c..2f5028f72 100644 --- a/app/crowdb-access-server/tests/s3_e2e/basic.py +++ b/app/crowdb-access-server/tests/s3_e2e/basic.py @@ -236,6 +236,70 @@ def test_independent_frontends_share_one_namespace(self): self.assertEqual(absent.exception.response["ResponseMetadata"]["HTTPStatusCode"], 404) second.delete_bucket(Bucket=bucket) + def test_multipart_replaces_parts_and_publishes_selected_bytes(self): + bucket = f"{self.bucket}-multipart" + key = "parts/object.bin" + first = b"a" * (5 * 1024 * 1024) + replacement = b"b" * len(first) + tail = b"final-part" + self.client.create_bucket(Bucket=bucket) + try: + upload_id = self.client.create_multipart_upload(Bucket=bucket, Key=key)["UploadId"] + self.assertIn(upload_id, [item["UploadId"] for item in + self.client.list_multipart_uploads(Bucket=bucket)["Uploads"]]) + tail_etag = self.client.upload_part( + Bucket=bucket, Key=key, UploadId=upload_id, PartNumber=2, Body=tail, + )["ETag"] + self.assertEqual(self.client.upload_part( + Bucket=bucket, Key=key, UploadId=upload_id, PartNumber=2, Body=tail, + )["ETag"], tail_etag) + self.client.upload_part(Bucket=bucket, Key=key, UploadId=upload_id, + PartNumber=1, Body=first) + first_etag = self.client.upload_part( + Bucket=bucket, Key=key, UploadId=upload_id, PartNumber=1, Body=replacement, + )["ETag"] + listed = self.client.list_parts(Bucket=bucket, Key=key, UploadId=upload_id) + self.assertEqual([part["PartNumber"] for part in listed["Parts"]], [1, 2]) + self.assertEqual(listed["Parts"][0]["ETag"], first_etag) + completed = self.client.complete_multipart_upload( + Bucket=bucket, Key=key, UploadId=upload_id, + MultipartUpload={"Parts": [ + {"PartNumber": 1, "ETag": first_etag}, + {"PartNumber": 2, "ETag": tail_etag}, + ]}, + ) + expected_etag = md5(md5(replacement).digest() + md5(tail).digest()).hexdigest() + "-2" + self.assertEqual(completed["ETag"], f'"{expected_etag}"') + replayed = self.client.complete_multipart_upload( + Bucket=bucket, Key=key, UploadId=upload_id, + MultipartUpload={"Parts": [ + {"PartNumber": 1, "ETag": first_etag}, + {"PartNumber": 2, "ETag": tail_etag}, + ]}, + ) + self.assertEqual(replayed["ETag"], completed["ETag"]) + self.assertEqual(self.client.get_object(Bucket=bucket, Key=key)["Body"].read(), replacement + tail) + self.client.delete_object(Bucket=bucket, Key=key) + + aborted = self.client.create_multipart_upload(Bucket=bucket, Key=key)["UploadId"] + self.client.upload_part(Bucket=bucket, Key=key, UploadId=aborted, PartNumber=1, Body=tail) + self.client.abort_multipart_upload(Bucket=bucket, Key=key, UploadId=aborted) + self.client.abort_multipart_upload(Bucket=bucket, Key=key, UploadId=aborted) + self.assertNotIn(aborted, [item["UploadId"] for item in + self.client.list_multipart_uploads(Bucket=bucket).get("Uploads", [])]) + + invalid = self.client.create_multipart_upload(Bucket=bucket, Key=key)["UploadId"] + self.client.upload_part(Bucket=bucket, Key=key, UploadId=invalid, PartNumber=1, Body=tail) + with self.assertRaises(ClientError) as mismatch: + self.client.complete_multipart_upload( + Bucket=bucket, Key=key, UploadId=invalid, + MultipartUpload={"Parts": [{"PartNumber": 1, "ETag": '"' + "00" * 16 + '"'}]}, + ) + self.assertEqual(mismatch.exception.response["Error"]["Code"], "InvalidPart") + self.client.abort_multipart_upload(Bucket=bucket, Key=key, UploadId=invalid) + finally: + self.client.delete_bucket(Bucket=bucket) + def test_slow_signed_upload_releases_native_buffers(self): bucket = f"{self.bucket}-slow" path = f"/{bucket}/slow.bin" diff --git a/app/crowdb-access-server/tests/s3_e2e/lost_reply.py b/app/crowdb-access-server/tests/s3_e2e/lost_reply.py index 1b5dd794b..e49684ff5 100644 --- a/app/crowdb-access-server/tests/s3_e2e/lost_reply.py +++ b/app/crowdb-access-server/tests/s3_e2e/lost_reply.py @@ -1,7 +1,7 @@ # Copyright 2026-present Gian # Licensed under the Apache License, Version 2.0. -"""Drop a completed PUT response at a loopback proxy, then retry the PUT.""" +"""Drop committed PUT and multipart responses at a loopback proxy, then retry.""" import os import socket @@ -17,7 +17,7 @@ from botocore.credentials import Credentials -def swallow_put_reply(listener, endpoint): +def swallow_reply(listener, endpoint, expected_status): parsed = urlsplit(endpoint) with listener: client, _ = listener.accept() @@ -30,11 +30,11 @@ def swallow_put_reply(listener, endpoint): assert received, "client closed before signed PUT headers" request.extend(received) headers, body = bytes(request).split(b"\r\n\r\n", 1) - content_length = next( + content_length = next(( int(line.split(b":", 1)[1].strip()) for line in headers.split(b"\r\n") if line.lower().startswith(b"content-length:") - ) + ), 0) backend.sendall(headers + b"\r\n\r\n" + body) remaining = content_length - len(body) while remaining: @@ -45,32 +45,15 @@ def swallow_put_reply(listener, endpoint): response = bytearray() while chunk := backend.recv(8192): response.extend(chunk) - assert response.startswith(b"HTTP/1.1 200 "), response[:256] + assert response.startswith(f"HTTP/1.1 {expected_status} ".encode()), response[:256] return bytes(response) -def main(): - endpoint = os.environ["CROWDB_S3_E2E_ENDPOINT"] +def drop_reply(endpoint, credentials, method, path, payload, expected_status): parsed = urlsplit(endpoint) - credentials = Credentials( - os.environ["CROWDB_S3_E2E_ACCESS_KEY"], os.environ["CROWDB_S3_E2E_SECRET_KEY"] - ) - client = boto3.client( - "s3", - endpoint_url=endpoint, - region_name=os.environ.get("CROWDB_S3_E2E_REGION", "us-east-1"), - aws_access_key_id=credentials.access_key, - aws_secret_access_key=credentials.secret_key, - config=Config(s3={"addressing_style": "path"}), - ) - bucket = "crowdb-e2e-lost-reply" - key = "retry/same-payload.bin" - payload = bytes(range(256)) * 257 - etag = f'"{md5(payload).hexdigest()}"' - client.create_bucket(Bucket=bucket) request = AWSRequest( - method="PUT", - url=f"{endpoint}/{bucket}/{key}", + method=method, + url=f"{endpoint}{path}", data=payload, headers={ "Host": parsed.netloc, @@ -84,26 +67,78 @@ def main(): with socket.socket() as listener, ThreadPoolExecutor(max_workers=1) as workers: listener.bind(("127.0.0.1", 0)) listener.listen(1) - forwarded = workers.submit(swallow_put_reply, listener, endpoint) + forwarded = workers.submit(swallow_reply, listener, endpoint, expected_status) proxy = HTTPConnection("127.0.0.1", listener.getsockname()[1], timeout=15) try: - proxy.request("PUT", f"/{bucket}/{key}", body=payload, headers=dict(request.headers.items())) + proxy.request(method, path, body=payload, headers=dict(request.headers.items())) try: proxy.getresponse() except RemoteDisconnected: pass else: - raise AssertionError("proxy unexpectedly returned the completed PUT response") + raise AssertionError(f"proxy unexpectedly returned the completed {method} response") finally: proxy.close() - response = forwarded.result(timeout=20) - assert f"\r\netag: {etag}\r\n".lower().encode() in response.lower(), response[:512] + return forwarded.result(timeout=20) + + +def main(): + endpoint = os.environ["CROWDB_S3_E2E_ENDPOINT"] + credentials = Credentials( + os.environ["CROWDB_S3_E2E_ACCESS_KEY"], os.environ["CROWDB_S3_E2E_SECRET_KEY"] + ) + client = boto3.client( + "s3", + endpoint_url=endpoint, + region_name=os.environ.get("CROWDB_S3_E2E_REGION", "us-east-1"), + aws_access_key_id=credentials.access_key, + aws_secret_access_key=credentials.secret_key, + config=Config(s3={"addressing_style": "path"}), + ) + bucket = "crowdb-e2e-lost-reply" + key = "retry/same-payload.bin" + payload = bytes(range(256)) * 257 + etag = f'"{md5(payload).hexdigest()}"' + client.create_bucket(Bucket=bucket) + response = drop_reply(endpoint, credentials, "PUT", f"/{bucket}/{key}", payload, 200) + assert f"\r\netag: {etag}\r\n".lower().encode() in response.lower(), response[:512] assert client.put_object(Bucket=bucket, Key=key, Body=payload)["ETag"] == etag assert client.get_object(Bucket=bucket, Key=key)["Body"].read() == payload listed = client.list_objects_v2(Bucket=bucket, Prefix="retry/") assert [item["Key"] for item in listed["Contents"]] == [key] client.delete_object(Bucket=bucket, Key=key) + + multipart_key = "retry/multipart.bin" + part = b"multipart-response-loss" * 512 + part_etag = f'"{md5(part).hexdigest()}"' + upload_id = client.create_multipart_upload(Bucket=bucket, Key=multipart_key)["UploadId"] + query = f"?partNumber=1&uploadId={upload_id}" + response = drop_reply(endpoint, credentials, "PUT", f"/{bucket}/{multipart_key}{query}", part, 200) + assert f"\r\netag: {part_etag}\r\n".lower().encode() in response.lower(), response[:512] + assert client.upload_part(Bucket=bucket, Key=multipart_key, UploadId=upload_id, + PartNumber=1, Body=part)["ETag"] == part_etag + assert len(client.list_parts(Bucket=bucket, Key=multipart_key, UploadId=upload_id)["Parts"]) == 1 + + complete = ( + f"1{part_etag}" + "" + ).encode() + path = f"/{bucket}/{multipart_key}?uploadId={upload_id}" + drop_reply(endpoint, credentials, "POST", path, complete, 200) + published = client.complete_multipart_upload( + Bucket=bucket, Key=multipart_key, UploadId=upload_id, + MultipartUpload={"Parts": [{"PartNumber": 1, "ETag": part_etag}]}, + ) + assert published["ETag"] == f'"{md5(md5(part).digest()).hexdigest()}-1"' + assert client.get_object(Bucket=bucket, Key=multipart_key)["Body"].read() == part + client.delete_object(Bucket=bucket, Key=multipart_key) + + aborted = client.create_multipart_upload(Bucket=bucket, Key=multipart_key)["UploadId"] + drop_reply(endpoint, credentials, "DELETE", f"/{bucket}/{multipart_key}?uploadId={aborted}", b"", 204) + client.abort_multipart_upload(Bucket=bucket, Key=multipart_key, UploadId=aborted) + assert all(item["UploadId"] != aborted for item in + client.list_multipart_uploads(Bucket=bucket).get("Uploads", [])) client.delete_bucket(Bucket=bucket) diff --git a/app/crowdb-access-server/tests/s3_e2e/restart.py b/app/crowdb-access-server/tests/s3_e2e/restart.py index 9eef09caf..ec5c51dc9 100644 --- a/app/crowdb-access-server/tests/s3_e2e/restart.py +++ b/app/crowdb-access-server/tests/s3_e2e/restart.py @@ -31,6 +31,9 @@ def main(): new_payload = b"new-generation" * 8192 new_etag = f'"{md5(new_payload).hexdigest()}"' deleted_key = "persisted/deleted-before-restart.bin" + multipart_key = "persisted/incomplete-before-restart.bin" + multipart_payload = b"durable-part-after-restart" * 512 + multipart_etag = f'"{md5(multipart_payload).hexdigest()}"' if phase == "prepare": client.create_bucket(Bucket=bucket) @@ -39,6 +42,9 @@ def main(): assert client.put_object(Bucket=bucket, Key=overwritten_key, Body=new_payload)["ETag"] == new_etag client.put_object(Bucket=bucket, Key=deleted_key, Body=old_payload) client.delete_object(Bucket=bucket, Key=deleted_key) + upload_id = client.create_multipart_upload(Bucket=bucket, Key=multipart_key)["UploadId"] + assert client.upload_part(Bucket=bucket, Key=multipart_key, UploadId=upload_id, + PartNumber=1, Body=multipart_payload)["ETag"] == multipart_etag elif phase in ( "verify", "verify-after-group0-restart", @@ -56,12 +62,28 @@ def main(): assert client.get_object(Bucket=bucket, Key=overwritten_key)["Body"].read() == new_payload keys = {entry["Key"] for entry in client.list_objects_v2(Bucket=bucket).get("Contents", [])} assert keys == {key, overwritten_key}, keys + uploads = [item for item in client.list_multipart_uploads(Bucket=bucket).get("Uploads", []) + if item["Key"] == multipart_key] + assert len(uploads) == 1, uploads + parts = client.list_parts(Bucket=bucket, Key=multipart_key, + UploadId=uploads[0]["UploadId"])["Parts"] + assert [(part["PartNumber"], part["ETag"]) for part in parts] == [(1, multipart_etag)] break except (BotoCoreError, ClientError): if time.monotonic() >= deadline: raise time.sleep(0.5) elif phase == "cleanup": + uploads = [item for item in client.list_multipart_uploads(Bucket=bucket).get("Uploads", []) + if item["Key"] == multipart_key] + assert len(uploads) == 1, uploads + completed = client.complete_multipart_upload( + Bucket=bucket, Key=multipart_key, UploadId=uploads[0]["UploadId"], + MultipartUpload={"Parts": [{"PartNumber": 1, "ETag": multipart_etag}]}, + ) + assert completed["ETag"] == f'"{md5(md5(multipart_payload).digest()).hexdigest()}-1"' + assert client.get_object(Bucket=bucket, Key=multipart_key)["Body"].read() == multipart_payload + client.delete_object(Bucket=bucket, Key=multipart_key) client.delete_object(Bucket=bucket, Key=key) client.delete_object(Bucket=bucket, Key=overwritten_key) client.delete_bucket(Bucket=bucket) diff --git a/app/crowdb-access-server/tests/s3_full_stack_test.rs b/app/crowdb-access-server/tests/s3_full_stack_test.rs index 77fdbf564..7c7f2176a 100644 --- a/app/crowdb-access-server/tests/s3_full_stack_test.rs +++ b/app/crowdb-access-server/tests/s3_full_stack_test.rs @@ -32,9 +32,10 @@ use hyper::body::Bytes; use serde_json::json; const MASTER_KEY: &str = "1111111111111111111111111111111111111111111111111111111111111111"; -const TEST_COUNT: usize = 18; +const TEST_COUNT: usize = 19; const BOTO3_CASES: &[&str] = &[ "test_signed_raw_http_wire_contract", + "test_multipart_replaces_parts_and_publishes_selected_bytes", "test_independent_frontends_share_one_namespace", "test_slow_signed_upload_releases_native_buffers", "test_truncated_signed_upload_does_not_publish_and_releases_credit", @@ -152,7 +153,7 @@ impl FullStackSetup { run_boto3_case(method, &context); case.pass(); } - let case = TestCase::start("boto3::lost_put_reply_is_idempotent"); + let case = TestCase::start("boto3::lost_put_and_multipart_replies_are_idempotent"); run_restart_phase("lost-reply", &self.listen, &self.access_key, &self.secret_key); assert_native_write_metrics(&self.listen); case.pass(); @@ -759,6 +760,7 @@ async fn seed_compact_hardware(hardware: &HardwareClient) -> Vec Vec Vec +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_server::s3::{CompleteRequestError, CompleteSelection}; + +#[test] +fn s3_and_iceberg_use_the_same_bounded_completion_parser() { + let digest = "ab".repeat(16); + let body = format!( + "1\"{digest}\"" + ); + let s3 = CompleteSelection::parse(body.as_bytes()).unwrap(); + let iceberg = crowdb_access_server::iceberg::CompleteSelection::parse(body.as_bytes()).unwrap(); + assert_eq!(s3.parts(), iceberg.parts()); + assert_eq!(s3.parts()[0].etag, digest); +} + +#[test] +fn completion_parser_distinguishes_part_order_from_malformed_requests() { + let digest = "ab".repeat(16); + let part = |number| format!("{number}\"{digest}\""); + for numbers in [[2, 1], [1, 1]] { + let body = format!( + "{}{}", + part(numbers[0]), + part(numbers[1]) + ); + assert_eq!( + CompleteSelection::parse(body.as_bytes()), + Err(CompleteRequestError::InvalidPartOrder) + ); + } + assert_eq!( + CompleteSelection::parse(b""), + Err(CompleteRequestError::InvalidRequest) + ); +} diff --git a/app/crowdb-chunkdb/tests/common/cluster.rs b/app/crowdb-chunkdb/tests/common/cluster.rs index 6956388fe..76c80a2a2 100644 --- a/app/crowdb-chunkdb/tests/common/cluster.rs +++ b/app/crowdb-chunkdb/tests/common/cluster.rs @@ -466,6 +466,7 @@ pub async fn seed_hardware_layout_from_disk_group( &RackValue { status: HwStatus::Up as i32, node_ids: node_ids.clone(), + ..Default::default() }, ) .await @@ -482,6 +483,7 @@ pub async fn seed_hardware_layout_from_disk_group( disk_group_ids: vec![dg_id], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await @@ -499,6 +501,7 @@ pub async fn seed_hardware_layout_from_disk_group( &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.clone(), + name: String::new(), }, ) .await diff --git a/app/crowdb-chunkdb/tests/e2e_test.rs b/app/crowdb-chunkdb/tests/e2e_test.rs index 78679855a..dbccbb441 100644 --- a/app/crowdb-chunkdb/tests/e2e_test.rs +++ b/app/crowdb-chunkdb/tests/e2e_test.rs @@ -41,6 +41,7 @@ fn build_test_topology() -> TopologyCache { value: DiskGroupValue { status: HwStatus::Up as i32, disk_ids: vec![], + name: String::new(), }, }); } diff --git a/app/crowdb-chunkdb/tests/selector_test.rs b/app/crowdb-chunkdb/tests/selector_test.rs index 90b0bfd20..19c517091 100644 --- a/app/crowdb-chunkdb/tests/selector_test.rs +++ b/app/crowdb-chunkdb/tests/selector_test.rs @@ -22,6 +22,7 @@ fn make_dg_entry(dg_id: u64, rack: u64, node: u64, status: HwStatus) -> DiskGrou value: DiskGroupValue { status: status as i32, disk_ids: vec![], + name: String::new(), }, } } diff --git a/app/crowdb-chunkdb/tests/topology_test.rs b/app/crowdb-chunkdb/tests/topology_test.rs index 3c23aef1f..b7fa5d9fc 100644 --- a/app/crowdb-chunkdb/tests/topology_test.rs +++ b/app/crowdb-chunkdb/tests/topology_test.rs @@ -19,6 +19,7 @@ fn make_dg_entry(dg_id: u64, rack: u64, node: u64, status: HwStatus) -> DiskGrou value: DiskGroupValue { status: status as i32, disk_ids: vec![], + name: String::new(), }, } } diff --git a/app/crowdb-cli/src/commands.rs b/app/crowdb-cli/src/commands.rs index c9129ff19..b33f611ab 100644 --- a/app/crowdb-cli/src/commands.rs +++ b/app/crowdb-cli/src/commands.rs @@ -28,102 +28,33 @@ pub(crate) use s3::{run_s3_verb, S3Verb}; use std::process::ExitCode; use crowdb_console_shared::ops::OpContext; -use crowdb_console_shared::ConsoleConfigEngine; use crate::Cli; -/// Build an [`OpContext`] from the CLI global flags. The system endpoint -/// is `http://{system_ip}:{system_port}` and the -/// CLI state is loaded from the fixed runtime location. -/// -/// When the config has server entries, their mgmt URLs are added as -/// additional seeds so the client can discover the group-0 leader even -/// if `--system-port` doesn't point at a running server (e.g. after -/// `local-deploy` which allocates dynamic ports). -pub(crate) fn op_context(cli: &Cli) -> Result { - let config = load_config(cli)?; - let mgmt_url = format!("http://{}:{}", cli.system_ip, cli.system_port); - let group0_endpoint = format!("{}:{}", cli.system_ip, cli.system_port); - - // Collect mgmt seeds: the explicit --system-port endpoint plus all - // server URLs from the config (so local-deploy'd servers are found). - let mut seeds = vec![mgmt_url]; - for server in &config.servers { - if !seeds.contains(&server.url) { - seeds.push(server.url.clone()); - } - } - - // Use the first config server's RPC URL as the group0 endpoint hint - // if available (more accurate than the default port). Strip the - // `http://` prefix since the crowdb-rpc endpoint format is `ip:port`. - let effective_g0 = - config - .servers - .first() - .and_then(|s| s.rpc_url.as_ref()) - .map_or(group0_endpoint, |url| { - url.strip_prefix("http://") - .or_else(|| url.strip_prefix("https://")) - .unwrap_or(url) - .to_string() - }); - - Ok(OpContext::new(effective_g0, seeds, config)) -} - -/// Load the CLI's internal persisted state from its fixed runtime location. -pub(crate) fn load_config(cli: &Cli) -> Result { +/// Build a Group 0 context for hardware operations without reading the old +/// local console state. The CLI endpoint is a discovery seed, while the +/// optional launch registry is validated only as local process policy. +pub(crate) async fn authority_context(cli: &Cli) -> Result { if let Some(path) = &cli.registry { crowdb_console_shared::config::web::LaunchRegistry::load(path).map_err(|error| { eprintln!("error: load launch registry: {error}"); ExitCode::from(2) })?; - return Ok(crowdb_console_shared::ConsoleConfig::default()); - } - let path = config_path(); - if !path.exists() { - return Ok(crowdb_console_shared::ConsoleConfig::default()); } - let engine = crowdb_console_shared::TomlFileEngine::new(path); - engine.load().map_err(|e| { - eprintln!("error: load config: {e}"); - ExitCode::from(2) - }) -} - -/// Resolve the private CLI state file. The environment override is reserved -/// for isolated test and benchmark harnesses and is intentionally not a CLI -/// option. -fn config_path() -> std::path::PathBuf { - std::env::var_os("CROWDB_CLI_STATE").map_or_else( - || { - crowdb_protocol::port::namespace::runtime_root() - .join("persistent") - .join("console") - .join("crowdb-kv.db.toml") - }, - std::path::PathBuf::from, - ) -} - -/// Persist the config from an [`OpContext`] back to the config file. -pub(crate) fn commit_config(cli: &Cli, ctx: &OpContext) -> Result<(), ExitCode> { - if cli.registry.is_some() { - eprintln!("error: launch registry mode cannot persist local cluster topology"); - return Err(ExitCode::from(2)); - } - let path = config_path(); - if let Some(parent) = path.parent() { - std::fs::create_dir_all(parent).map_err(|e| { - eprintln!("error: create config dir {}: {e}", parent.display()); - ExitCode::from(2) - })?; + let mgmt_url = format!("http://{}:{}", cli.system_ip, cli.system_port); + let group0_endpoint = match crowdb_console_shared::clients::http::ServerClient::new(&mgmt_url) { + Ok(client) => client + .topology() + .await + .ok() + .and_then(|stores| stores.into_iter().find(|store| store.store_id == 0)) + .and_then(|store| store.listen_addr), + Err(_) => None, } - let engine = crowdb_console_shared::TomlFileEngine::new(path.clone()); - let cfg = ctx.config().clone(); - engine.save(&cfg).map_err(|e| { - eprintln!("error: save config {}: {e}", path.display()); - ExitCode::from(2) - }) + .unwrap_or_else(|| format!("{}:{}", cli.system_ip, cli.system_port)); + Ok(OpContext::new( + group0_endpoint, + vec![mgmt_url], + crowdb_console_shared::ConsoleConfig::default(), + )) } diff --git a/app/crowdb-cli/src/commands/bench/chunk.rs b/app/crowdb-cli/src/commands/bench/chunk.rs index 1f0746353..2dd7445c8 100644 --- a/app/crowdb-cli/src/commands/bench/chunk.rs +++ b/app/crowdb-cli/src/commands/bench/chunk.rs @@ -40,7 +40,7 @@ pub async fn run(cli: &Cli, verb: ChunkdbBenchVerb) -> ExitCode { if !valid_args(&args) { return ExitCode::from(2); } - let kv = match build_kv_client(cli, ReadEndpointPolicy::Leader, &KvClientTunables::default()) { + let kv = match build_kv_client(cli, ReadEndpointPolicy::Leader, &KvClientTunables::default()).await { Ok(client) => Arc::new(client), Err(code) => return code, }; diff --git a/app/crowdb-cli/src/commands/bench/disk/db.rs b/app/crowdb-cli/src/commands/bench/disk/db.rs index 63875acac..21b359468 100644 --- a/app/crowdb-cli/src/commands/bench/disk/db.rs +++ b/app/crowdb-cli/src/commands/bench/disk/db.rs @@ -63,7 +63,7 @@ pub async fn run(cli: &Cli, verb: DiskdbBenchVerb) -> ExitCode { if !valid_args(&args) { return ExitCode::from(2); } - let kv = match build_kv_client(cli, ReadEndpointPolicy::Leader, &KvClientTunables::default()) { + let kv = match build_kv_client(cli, ReadEndpointPolicy::Leader, &KvClientTunables::default()).await { Ok(kv) => Arc::new(kv), Err(code) => return code, }; diff --git a/app/crowdb-cli/src/commands/bench/io.rs b/app/crowdb-cli/src/commands/bench/io.rs index 57b20f1fc..2a8fd4a08 100644 --- a/app/crowdb-cli/src/commands/bench/io.rs +++ b/app/crowdb-cli/src/commands/bench/io.rs @@ -35,17 +35,7 @@ async fn connect( diskio_connections_per_endpoint: usize, diskio_rpc_workers: u32, ) -> Result { - let config = crate::commands::load_config(cli)?; - let mut seeds = vec![format!("http://{}:{}", cli.system_ip, cli.system_port)]; - for server in config - .servers - .iter() - .filter(|server| server.service_type == crowdb_console_shared::config::ServiceType::Kv) - { - if !seeds.contains(&server.url) { - seeds.push(server.url.clone()); - } - } + let seeds = vec![format!("http://{}:{}", cli.system_ip, cli.system_port)]; ChunkIoClient::connect(ChunkIoClientConfig { management_seeds: seeds, diskio_connections_per_endpoint, diff --git a/app/crowdb-cli/src/commands/bench/kv/client.rs b/app/crowdb-cli/src/commands/bench/kv/client.rs index 542225045..ce6fe8dfc 100644 --- a/app/crowdb-cli/src/commands/bench/kv/client.rs +++ b/app/crowdb-cli/src/commands/bench/kv/client.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Shared helper for building a [`CrowdbKvClient`] from the CLI's config +//! Shared helper for building a [`CrowdbKvClient`] from the CLI endpoint //! with a bench-specific [`ReadEndpointPolicy`]. The standard `op_context` //! helper always uses the default `Leader` policy; bench read/scan //! commands need `AnyReplica` for distributed `MinSlot` reads. @@ -38,39 +38,32 @@ impl Default for KvClientTunables { } } -/// Build a `CrowdbKvClient` from private CLI state plus `--system-*` -/// with the given `read_endpoint_policy`. Seeds the system-group hint -/// from the first config server's RPC URL (same logic as `op_context`). +/// Build a `CrowdbKvClient` from `--system-*` with the given +/// `read_endpoint_policy`. /// /// # Errors -/// Returns `ExitCode::from(2)` if the config cannot be loaded. -pub(crate) fn build_kv_client( +/// Returns `ExitCode::from(2)` if the management endpoint is unavailable. +pub(crate) async fn build_kv_client( cli: &Cli, read_endpoint_policy: ReadEndpointPolicy, tunables: &KvClientTunables, ) -> Result { - let config = crate::commands::load_config(cli)?; - let mgmt_url = format!("http://{}:{}", cli.system_ip, cli.system_port); - let group0_endpoint = format!("{}:{}", cli.system_ip, cli.system_port); - - let mut seeds = vec![mgmt_url]; - for server in &config.servers { - if !seeds.contains(&server.url) { - seeds.push(server.url.clone()); - } - } - - let effective_g0 = - config - .servers - .first() - .and_then(|s| s.rpc_url.as_ref()) - .map_or(group0_endpoint, |url| { - url.strip_prefix("http://") - .or_else(|| url.strip_prefix("https://")) - .unwrap_or(url) - .to_string() - }); + let seed = format!("http://{}:{}", cli.system_ip, cli.system_port); + let server = crowdb_console_shared::clients::http::ServerClient::new(&seed).map_err(|error| { + eprintln!("bench kv client: {error}"); + ExitCode::from(2) + })?; + let effective_g0 = server + .topology() + .await + .ok() + .and_then(|stores| stores.into_iter().find(|store| store.store_id == 0)) + .and_then(|store| store.listen_addr) + .ok_or_else(|| { + eprintln!("bench kv client: Group 0 endpoint unavailable at {seed}"); + ExitCode::from(2) + })?; + let seeds = vec![seed]; let mut client_config = ClientConfig::new(seeds); client_config.read_endpoint_policy = read_endpoint_policy; diff --git a/app/crowdb-cli/src/commands/bench/kv/prepare.rs b/app/crowdb-cli/src/commands/bench/kv/prepare.rs index 464322559..f7fbbe4cf 100644 --- a/app/crowdb-cli/src/commands/bench/kv/prepare.rs +++ b/app/crowdb-cli/src/commands/bench/kv/prepare.rs @@ -23,7 +23,9 @@ pub async fn run(cli: &Cli, args: PrepareArgs) -> ExitCode { cli, crowdb_kv_client::ReadEndpointPolicy::Leader, &KvClientTunables::default(), - ) { + ) + .await + { Ok(c) => Arc::new(c), Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/bench/kv/read.rs b/app/crowdb-cli/src/commands/bench/kv/read.rs index 3f668c7ce..4d7073300 100644 --- a/app/crowdb-cli/src/commands/bench/kv/read.rs +++ b/app/crowdb-cli/src/commands/bench/kv/read.rs @@ -42,7 +42,9 @@ pub async fn run(cli: &Cli, args: ReadArgs) -> ExitCode { pool_size: args.connections, ..Default::default() }, - ) { + ) + .await + { Ok(c) => Arc::new(c), Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/bench/kv/scan.rs b/app/crowdb-cli/src/commands/bench/kv/scan.rs index 0f5945d96..cd84dc086 100644 --- a/app/crowdb-cli/src/commands/bench/kv/scan.rs +++ b/app/crowdb-cli/src/commands/bench/kv/scan.rs @@ -43,7 +43,9 @@ pub async fn run(cli: &Cli, args: ScanArgs) -> ExitCode { pool_size: args.connections, ..Default::default() }, - ) { + ) + .await + { Ok(c) => Arc::new(c), Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/bench/kv/write.rs b/app/crowdb-cli/src/commands/bench/kv/write.rs index 49f1a29f8..0c38e51ed 100644 --- a/app/crowdb-cli/src/commands/bench/kv/write.rs +++ b/app/crowdb-cli/src/commands/bench/kv/write.rs @@ -18,13 +18,13 @@ use rand::rngs::SmallRng; use rand::{Rng, SeedableRng}; use super::client::{build_kv_client, KvClientTunables}; +use crate::commands::authority_context; use crate::commands::bench::loader::{run_workload, BenchRecorder}; use crate::commands::bench::metrics::BenchMetrics; use crate::commands::bench::result::{ BenchOps, BenchResult, ReplicaStats, ServerMetrics, ServerRpcLatency, SnapshotStats, TransportStats, }; use crate::commands::bench::verb::WriteArgs; -use crate::commands::load_config; use crate::Cli; #[allow(clippy::too_many_lines)] @@ -45,7 +45,9 @@ pub async fn run(cli: &Cli, args: WriteArgs) -> ExitCode { pool_size: args.connections, ..Default::default() }, - ) { + ) + .await + { Ok(c) => Arc::new(c), Err(c) => return c, }; @@ -173,16 +175,24 @@ fn build_value(id: u64, size: usize) -> Vec { .collect() } -/// Fetch `/metrics` from every server in the config and aggregate into +/// Fetch `/metrics` from confirmed replica hosts and aggregate into /// `ServerMetrics`. Metrics are summed across nodes except for averages /// (which are averaged across nodes that report them). #[allow(clippy::too_many_lines)] async fn fetch_server_metrics(cli: &Cli, store_id: u64, group_id: u64) -> Option { - let config = load_config(cli).ok()?; + let ctx = authority_context(cli).await.ok()?; + let replicas = ctx + .sysmd() + .list_replicas_in_group(store_id, group_id) + .await + .ok()?; let group_prefix = format!("s.{store_id}.g.{group_id}."); let store_prefix = format!("s.{store_id}.rpc."); - let mut mgmt_urls: Vec = config.servers.iter().map(|s| s.url.clone()).collect(); + let mut mgmt_urls = Vec::with_capacity(replicas.len()); + for replica in replicas { + mgmt_urls.push(ctx.live_node_mgmt_url(replica.node_id).await.ok()?); + } mgmt_urls.sort(); mgmt_urls.dedup(); if mgmt_urls.is_empty() { diff --git a/app/crowdb-cli/src/commands/chunk/diskdb.rs b/app/crowdb-cli/src/commands/chunk/diskdb.rs index fbea07602..737697fe4 100644 --- a/app/crowdb-cli/src/commands/chunk/diskdb.rs +++ b/app/crowdb-cli/src/commands/chunk/diskdb.rs @@ -9,8 +9,8 @@ use clap::Subcommand; use crowdb_console_shared::ops::chunk; +use crate::commands::authority_context; use crate::commands::launch::{self, LaunchVerb}; -use crate::commands::op_context; use crate::Cli; #[derive(Subcommand, Debug)] @@ -93,7 +93,7 @@ pub async fn run_chunk_diskdb_verb(cli: &Cli, verb: ChunkDiskdbVerb) -> ExitCode } async fn run_list(cli: &Cli, explicit_endpoint: Option<&str>) -> ExitCode { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(ctx) => ctx, Err(code) => return code, }; diff --git a/app/crowdb-cli/src/commands/chunk/stub.rs b/app/crowdb-cli/src/commands/chunk/stub.rs index 825dde45b..a7900561b 100644 --- a/app/crowdb-cli/src/commands/chunk/stub.rs +++ b/app/crowdb-cli/src/commands/chunk/stub.rs @@ -70,7 +70,7 @@ async fn run_service(cli: &Cli, service: &str, verb: ServiceVerb) -> ExitCode { } async fn run_list(cli: &Cli, service: &str) -> ExitCode { - let ctx = match crate::commands::op_context(cli) { + let ctx = match crate::commands::authority_context(cli).await { Ok(ctx) => ctx, Err(code) => return code, }; diff --git a/app/crowdb-cli/src/commands/cluster.rs b/app/crowdb-cli/src/commands/cluster.rs index 356c0bf0a..c14caa9d7 100644 --- a/app/crowdb-cli/src/commands/cluster.rs +++ b/app/crowdb-cli/src/commands/cluster.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! `cluster` domain — cluster-level ops: init, reset, clean, status, +//! `cluster` domain — cluster-level ops: init, destroy, clean, status, //! topology, plus hardware subcommands (rack/node/disk-group/disk). pub mod hardware; @@ -15,15 +15,63 @@ use std::process::ExitCode; use clap::Subcommand; -use crate::commands::{commit_config, op_context}; +use crate::commands::authority_context; use crate::Cli; +async fn local_deploy_context( + cli: &Cli, + existing_cluster: bool, +) -> Result { + if !existing_cluster { + let mgmt = format!("http://{}:{}", cli.system_ip, cli.system_port); + let rpc_hint = format!("{}:{}", cli.system_ip, cli.system_port); + return Ok(crowdb_console_shared::ops::OpContext::new( + rpc_hint, + vec![mgmt], + crowdb_console_shared::ConsoleConfig::default(), + )); + } + let ctx = authority_context(cli).await?; + { + let racks = crowdb_console_shared::ops::hardware::list_racks_from_group0(&ctx) + .await + .map_err(|error| { + eprintln!("error: read Group 0 racks: {error}"); + ExitCode::from(2) + })?; + let nodes = crowdb_console_shared::ops::hardware::list_nodes_from_group0(&ctx, None) + .await + .map_err(|error| { + eprintln!("error: read Group 0 nodes: {error}"); + ExitCode::from(2) + })?; + let mut servers = Vec::with_capacity(nodes.len()); + for node in &nodes { + let url = ctx.live_node_mgmt_url(node.id).await.map_err(|error| { + eprintln!("error: resolve live KV node {}: {error}", node.id); + ExitCode::from(2) + })?; + let mut server = crowdb_console_shared::config::ServerEntry::new(node.id.to_string(), url); + server.node_id = Some(node.id); + servers.push(server); + } + let mut config = ctx.config_mut(); + config.racks = racks; + config.nodes = nodes; + config.servers = servers; + } + Ok(ctx) +} + #[derive(Subcommand, Debug)] pub enum ClusterVerb { /// Initialize the cluster by bootstrapping group 0 on the listed nodes. Init { #[arg(short = 'n', long, value_delimiter = ',')] nodes: Vec, + /// Versioned bootstrap topology input for the first registry-mode init. + #[arg(long, value_name = "PATH")] + bootstrap_file: Option, }, /// Deploy a local N-node KV cluster on 127.0.0.1 (forks /// `crowdb-kv-server` on each node, bootstraps group 0). @@ -131,8 +179,6 @@ pub enum ClusterVerb { }, /// Tear down the entire cluster (all groups, stores, servers, sysdata). Destroy, - /// Remove orphaned sysdata entries without stopping running servers. - Reset, /// Wipe user data on every node + wait for re-election. Preserves /// group-0 sysdata + topology — servers stay running. Use --store/--group /// to target a non-system group (recommended for benchmarks). @@ -180,8 +226,15 @@ pub enum ClusterVerb { #[allow(clippy::too_many_lines)] pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { match verb { - ClusterVerb::Init { nodes } => { - let ctx = match op_context(cli) { + ClusterVerb::Init { + nodes, + bootstrap_file, + } => { + let Some(registry) = &cli.registry else { + eprintln!("error: cluster init requires --registry and a versioned bootstrap file"); + return ExitCode::from(2); + }; + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -192,11 +245,31 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { return ExitCode::from(1); } }; - match crowdb_console_shared::ops::cluster::init(&ctx, &node_ids).await { - Ok(summary) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; + let sealed_path = registry.with_extension("bootstrap-intent.toml"); + if let Some(source) = bootstrap_file { + let intent = match crowdb_console_shared::bootstrap_intent::BootstrapIntent::load(&source) { + Ok(intent) => intent, + Err(error) => { + eprintln!("error: load bootstrap file: {error}"); + return ExitCode::from(2); } + }; + if intent.members() != node_ids.as_slice() { + eprintln!("error: bootstrap file members differ from --nodes"); + return ExitCode::from(1); + } + if let Err(error) = intent.seal(&sealed_path) { + eprintln!("error: seal bootstrap intent: {error}"); + return ExitCode::from(2); + } + } else if !sealed_path.exists() { + eprintln!("error: --bootstrap-file is required for the first init"); + return ExitCode::from(1); + } + let result = + crowdb_console_shared::ops::cluster::init_with_intent(&ctx, &node_ids, &sealed_path).await; + match result { + Ok(summary) => { println!( "cluster initialized: store {}, group {}, {} nodes", summary.store_id, @@ -246,7 +319,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { allow_unsafe_ec, } => match service_type.as_str() { "combined" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, false).await { Ok(context) => context, Err(code) => return code, }; @@ -299,9 +372,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(code) = commit_config(cli, &ctx) { - return code; - } println!( "local-deploy combined: {} KV nodes, {} racks, {} DiskDB, {} ChunkDB, {} DiskIO", summary.kv_nodes, @@ -310,19 +380,19 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { summary.chunkdb_instances, summary.diskio_instances ); + if let Some(seed) = ctx.config().servers.first() { + println!("Group 0 management seed: {}", seed.url); + } ExitCode::SUCCESS } Err(error) => { - if let Err(code) = commit_config(cli, &ctx) { - return code; - } eprintln!("error: local-deploy combined: {error}"); ExitCode::from(2) } } } "kv" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, false).await { Ok(c) => c, Err(c) => return c, }; @@ -356,9 +426,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!( "local-deploy complete: {} nodes (rack {}, nodes [{}]), group 0 bootstrapped", summary.node_count, @@ -370,6 +437,9 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .collect::>() .join(", ") ); + if let Some(seed) = ctx.config().servers.first() { + println!("Group 0 management seed: {}", seed.url); + } ExitCode::SUCCESS } Err(e) => { @@ -379,7 +449,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } "rpc" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, false).await { Ok(c) => c, Err(c) => return c, }; @@ -399,9 +469,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!( "local-deploy rpc: port={}, pid={}, io_engines={}, io_workers={}, nagle={}", summary.port, summary.pid, summary.io_engines, summary.io_workers, summary.nagle @@ -415,7 +482,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } "diskdb" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, true).await { Ok(c) => c, Err(c) => return c, }; @@ -438,9 +505,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(code) = commit_config(cli, &ctx) { - return code; - } println!( "local-deploy diskdb: {} instances, {} disk-groups, {} disks, data-groups {:?}", summary.instance_count, @@ -457,7 +521,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } "chunkdb" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, true).await { Ok(context) => context, Err(code) => return code, }; @@ -478,9 +542,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(code) = commit_config(cli, &ctx) { - return code; - } println!("local-deploy chunkdb: {} instances", summary.instance_count); ExitCode::SUCCESS } @@ -498,14 +559,38 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } }, ClusterVerb::Destroy => { - let ctx = match op_context(cli) { + let Some(path) = &cli.registry else { + eprintln!("error: cluster destroy requires --registry to stop local processes"); + return ExitCode::from(2); + }; + let registry = match crowdb_console_shared::config::web::LaunchRegistry::load(path) { + Ok(registry) => registry, + Err(error) => { + eprintln!("error: load launch registry: {error}"); + return ExitCode::from(2); + } + }; + let runtime = match crowdb_console_shared::launch::LaunchRuntime::for_registry(path) { + Ok(runtime) => runtime, + Err(error) => { + eprintln!("error: launch runtime: {error}"); + return ExitCode::from(2); + } + }; + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; match crowdb_console_shared::ops::cluster::destroy(&ctx).await { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; + for launch in ®istry.launches { + if let Err(error) = runtime.stop(launch).await { + eprintln!( + "error: stop {} on node {}: {error}", + launch.service_id, launch.node_id + ); + return ExitCode::from(2); + } } println!("cluster destroy complete"); ExitCode::SUCCESS @@ -516,45 +601,54 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } } - ClusterVerb::Reset => { - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - match crowdb_console_shared::ops::cluster::reset(&ctx).await { - Ok(()) => { - println!("cluster reset complete"); - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: cluster reset: {e}"); - ExitCode::from(2) - } - } - } ClusterVerb::Clean { store, group, restart_services, } => { - let ctx = match op_context(cli) { + let restart = if restart_services { + let Some(path) = &cli.registry else { + eprintln!("error: --restart-services requires --registry"); + return ExitCode::from(2); + }; + let registry = match crowdb_console_shared::config::web::LaunchRegistry::load(path) { + Ok(registry) => registry, + Err(error) => { + eprintln!("error: load launch registry: {error}"); + return ExitCode::from(2); + } + }; + let runtime = match crowdb_console_shared::launch::LaunchRuntime::for_registry(path) { + Ok(runtime) => runtime, + Err(error) => { + eprintln!("error: launch runtime: {error}"); + return ExitCode::from(2); + } + }; + Some((registry, runtime)) + } else { + None + }; + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; match crowdb_console_shared::ops::cluster::clean(&ctx, store, group).await { Ok(mut result) => { - if restart_services { - match crowdb_console_shared::ops::cluster::restart_storage_services(&ctx).await { - Ok(count) => result.restarted_services = count, - Err(error) => { - let _ = commit_config(cli, &ctx); - eprintln!("error: cluster clean service restart: {error}"); - return ExitCode::from(2); + if let Some((registry, runtime)) = restart { + for kind in ["diskio", "diskdb", "chunkdb"] { + for launch in registry + .launches + .iter() + .filter(|launch| launch.service_id == kind) + { + if let Err(error) = runtime.restart(launch).await { + eprintln!("error: restart {kind} on node {}: {error}", launch.node_id); + return ExitCode::from(2); + } + result.restarted_services += 1; } } - if let Err(code) = commit_config(cli, &ctx) { - return code; - } } println!( "cluster clean: wiped {} nodes, restarted {} services, leader = {}", @@ -569,7 +663,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } ClusterVerb::Status => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -601,7 +695,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } ClusterVerb::Topology { node } => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/cluster/hardware.rs b/app/crowdb-cli/src/commands/cluster/hardware.rs index f92e9d83c..e9aba57a4 100644 --- a/app/crowdb-cli/src/commands/cluster/hardware.rs +++ b/app/crowdb-cli/src/commands/cluster/hardware.rs @@ -10,7 +10,7 @@ use clap::Subcommand; use crowdb_console_shared::config::NodeEntry; use crowdb_protocol::{NodeId, RackId}; -use crate::commands::{commit_config, op_context}; +use crate::commands::authority_context; use crate::Cli; // ── rack ───────────────────────────────────────────────────────── @@ -40,15 +40,13 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; - match crowdb_console_shared::ops::hardware::add_rack(&ctx, rack_id, &name).await { + let result = crowdb_console_shared::ops::hardware::add_rack_to_group0(&ctx, rack_id, &name).await; + match result { Ok(entry) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!("added rack {}", entry.id); ExitCode::SUCCESS } @@ -66,15 +64,13 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; - match crowdb_console_shared::ops::hardware::remove_rack(&ctx, rack_id).await { + let result = crowdb_console_shared::ops::hardware::remove_rack_from_group0(&ctx, rack_id).await; + match result { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!("removed rack {id}"); ExitCode::SUCCESS } @@ -85,11 +81,17 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { } } RackVerb::List => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; - let racks = crowdb_console_shared::ops::hardware::list_racks(&ctx); + let racks = match crowdb_console_shared::ops::hardware::list_racks_from_group0(&ctx).await { + Ok(racks) => racks, + Err(error) => { + eprintln!("error: list racks: {error}"); + return ExitCode::from(2); + } + }; if racks.is_empty() { println!("(no racks)"); return ExitCode::SUCCESS; @@ -120,6 +122,8 @@ pub enum NodeVerb { ssh_user: String, #[arg(short = 'k', long)] ssh_key: Option, + #[arg(long)] + ssh_credential_ref: Option, }, Remove { #[arg(short = 'I', long)] @@ -143,7 +147,12 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { ssh_port, ssh_user, ssh_key, + ssh_credential_ref, } => { + if ssh_key.is_some() { + eprintln!("error: --ssh-key is local secret material; use --ssh-credential-ref"); + return ExitCode::from(1); + } let node_id: NodeId = match id.parse() { Ok(n) => n, Err(e) => { @@ -166,16 +175,15 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { ssh_user, ssh_key, ssh_password: None, + ssh_credential_ref, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; - match crowdb_console_shared::ops::hardware::add_node(&ctx, entry.clone()).await { + let result = crowdb_console_shared::ops::hardware::add_node_to_group0(&ctx, entry.clone()).await; + match result { Ok(e) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!("added node {} (rack {})", e.id, e.rack_id); ExitCode::SUCCESS } @@ -193,15 +201,13 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; - match crowdb_console_shared::ops::hardware::remove_node(&ctx, node_id).await { + let result = crowdb_console_shared::ops::hardware::remove_node_from_group0(&ctx, node_id).await; + match result { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!("removed node {id}"); ExitCode::SUCCESS } @@ -212,11 +218,17 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { } } NodeVerb::List => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; - let nodes = crowdb_console_shared::ops::hardware::list_nodes(&ctx, None); + let nodes = match crowdb_console_shared::ops::hardware::list_nodes_from_group0(&ctx, None).await { + Ok(nodes) => nodes, + Err(error) => { + eprintln!("error: list nodes: {error}"); + return ExitCode::from(2); + } + }; print_node_table(&nodes) } NodeVerb::ListRack { rack } => { @@ -227,11 +239,19 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; - let nodes = crowdb_console_shared::ops::hardware::list_nodes(&ctx, Some(rack_id)); + let nodes = + match crowdb_console_shared::ops::hardware::list_nodes_from_group0(&ctx, Some(rack_id)).await + { + Ok(nodes) => nodes, + Err(error) => { + eprintln!("error: list nodes: {error}"); + return ExitCode::from(2); + } + }; print_node_table(&nodes) } } @@ -274,9 +294,83 @@ pub enum DiskGroupVerb { } pub async fn run_disk_group_verb(cli: &Cli, verb: DiskGroupVerb) -> ExitCode { - let _ = (cli, verb); - eprintln!("disk-group commands not yet wired to ops (Phase 3)"); - ExitCode::from(1) + use crowdb_console_shared::ops::hardware; + let ctx = match authority_context(cli).await { + Ok(ctx) => ctx, + Err(code) => return code, + }; + let result = match verb { + DiskGroupVerb::Add { id, rack, node, name } => { + let (Ok(id), Ok(rack), Ok(node)) = (id.parse::(), rack.parse::(), node.parse::()) + else { + eprintln!("error: disk-group, rack and node IDs must be integers"); + return ExitCode::from(1); + }; + match hardware::list_nodes_from_group0(&ctx, Some(rack)).await { + Ok(nodes) if nodes.iter().any(|entry| entry.id == node) => {} + Ok(_) => { + eprintln!("error: node {node} is not in rack {rack}"); + return ExitCode::from(2); + } + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(2); + } + } + hardware::add_disk_group_to_group0(&ctx, node, id, &name) + .await + .map(|entry| { + println!("added disk group {} on node {}", entry.id, node); + }) + } + DiskGroupVerb::Remove { id } => { + let id = match id.parse::() { + Ok(id) => id, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(1); + } + }; + let groups = match ctx.sysmd().list_disk_groups().await { + Ok(groups) => groups, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(2); + } + }; + let matches: Vec<_> = groups.into_iter().filter(|group| group.dg_id == id).collect(); + if matches.len() != 1 { + eprintln!( + "error: disk group {id} has {} matches; specify a unique ID", + matches.len() + ); + return ExitCode::from(2); + } + hardware::remove_disk_group_from_group0(&ctx, matches[0].node_id, id) + .await + .map(|()| println!("removed disk group {id}")) + } + DiskGroupVerb::List => match ctx.sysmd().list_disk_groups().await { + Ok(mut groups) => { + groups.sort_unstable_by_key(|group| (group.rack_id, group.node_id, group.dg_id)); + for group in groups { + println!( + "{}\t{}\t{}\t{}", + group.rack_id, group.node_id, group.dg_id, group.value.name + ); + } + Ok(()) + } + Err(error) => Err(error.into()), + }, + }; + match result { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("error: {error}"); + ExitCode::from(2) + } + } } // ── disk ───────────────────────────────────────────────────────── @@ -310,8 +404,117 @@ pub enum DiskVerb { List, } +#[allow(clippy::too_many_lines)] pub async fn run_disk_verb(cli: &Cli, verb: DiskVerb) -> ExitCode { - let _ = (cli, verb); - eprintln!("disk commands not yet wired to ops (Phase 3)"); - ExitCode::from(1) + use crowdb_console_shared::ops::hardware::{self, AddDiskInput}; + use crowdb_protocol::DiskIdExt; + let ctx = match authority_context(cli).await { + Ok(ctx) => ctx, + Err(code) => return code, + }; + let result = match verb { + DiskVerb::Add { + id, + rack, + node, + group, + disk_type, + capacity, + zone_size, + unit_size, + device_path, + } => { + let (Ok(rack), Ok(node), Ok(group), Ok(capacity_bytes), Ok(zone_size_bytes), Ok(unit_size_bytes)) = ( + rack.parse::(), + node.parse::(), + group.parse::(), + capacity.parse::(), + zone_size.parse::(), + unit_size.parse::(), + ) else { + eprintln!("error: rack, node, group and size arguments must be integers"); + return ExitCode::from(1); + }; + let nodes = match hardware::list_nodes_from_group0(&ctx, Some(rack)).await { + Ok(nodes) => nodes, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(2); + } + }; + if !nodes.iter().any(|entry| entry.id == node) { + eprintln!("error: node {node} is not in rack {rack}"); + return ExitCode::from(2); + } + let input = AddDiskInput { + disk_id: id, + disk_type, + capacity_bytes, + zone_size_bytes, + unit_size_bytes, + device_path, + }; + hardware::add_disk_to_group0(&ctx, node, group, &input) + .await + .map(|entry| println!("added disk {}", entry.disk_id)) + } + DiskVerb::Remove { id } => { + let disk_id = match crowdb_protocol::common::DiskId::from_display_string(&id) { + Ok(id) => id, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(1); + } + }; + let disks = match ctx.sysmd().list_all_disks().await { + Ok(disks) => disks, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(2); + } + }; + let matches: Vec<_> = disks.into_iter().filter(|disk| disk.disk_id == disk_id).collect(); + if matches.len() != 1 { + eprintln!( + "error: disk {id} has {} matches; specify a unique ID", + matches.len() + ); + return ExitCode::from(2); + } + hardware::remove_disk_from_group0(&ctx, matches[0].node_id, matches[0].disk_group_id, &id) + .await + .map(|_| println!("removed disk {id}")) + } + DiskVerb::List => match ctx.sysmd().list_all_disks().await { + Ok(mut disks) => { + disks.sort_unstable_by_key(|disk| { + ( + disk.rack_id, + disk.node_id, + disk.disk_group_id, + disk.disk_id.high, + disk.disk_id.low, + ) + }); + for disk in disks { + println!( + "{}\t{}\t{}\t{}", + disk.rack_id, + disk.node_id, + disk.disk_group_id, + disk.disk_id.to_display_string() + ); + } + Ok(()) + } + Err(error) => Err(error.into()), + }, + }; + match result { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("error: {error}"); + ExitCode::from(2) + } + } } diff --git a/app/crowdb-cli/src/commands/kv/data.rs b/app/crowdb-cli/src/commands/kv/data.rs index 876b74bc5..4e22f9b1e 100644 --- a/app/crowdb-cli/src/commands/kv/data.rs +++ b/app/crowdb-cli/src/commands/kv/data.rs @@ -8,7 +8,7 @@ use std::process::ExitCode; use clap::Subcommand; use crowdb_kv_client::GetOutcome; -use crate::commands::op_context; +use crate::commands::authority_context; use crate::Cli; #[derive(Subcommand, Debug)] @@ -90,7 +90,7 @@ pub async fn run_kv_data_verb(cli: &Cli, verb: KvDataVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -119,7 +119,7 @@ pub async fn run_kv_data_verb(cli: &Cli, verb: KvDataVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -147,7 +147,7 @@ pub async fn run_kv_data_verb(cli: &Cli, verb: KvDataVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -174,7 +174,7 @@ pub async fn run_kv_data_verb(cli: &Cli, verb: KvDataVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -211,7 +211,7 @@ async fn run_snapshot_verb(cli: &Cli, verb: SnapshotVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -231,7 +231,7 @@ async fn run_snapshot_verb(cli: &Cli, verb: SnapshotVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -260,7 +260,7 @@ async fn run_snapshot_verb(cli: &Cli, verb: SnapshotVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/kv/logical.rs b/app/crowdb-cli/src/commands/kv/logical.rs index e5a83a4c7..98b60a429 100644 --- a/app/crowdb-cli/src/commands/kv/logical.rs +++ b/app/crowdb-cli/src/commands/kv/logical.rs @@ -7,7 +7,7 @@ use std::process::ExitCode; use clap::Subcommand; -use crate::commands::op_context; +use crate::commands::authority_context; use crate::Cli; // ── store ──────────────────────────────────────────────────────── @@ -45,7 +45,7 @@ pub async fn run_store_verb(cli: &Cli, verb: StoreVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -75,7 +75,7 @@ pub async fn run_store_verb(cli: &Cli, verb: StoreVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -91,7 +91,7 @@ pub async fn run_store_verb(cli: &Cli, verb: StoreVerb) -> ExitCode { } } StoreVerb::List => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -187,7 +187,7 @@ pub async fn run_group_verb(cli: &Cli, verb: GroupVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -221,7 +221,7 @@ pub async fn run_group_verb(cli: &Cli, verb: GroupVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -244,7 +244,7 @@ pub async fn run_group_verb(cli: &Cli, verb: GroupVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -333,7 +333,7 @@ pub async fn run_replica_verb(cli: &Cli, verb: ReplicaVerb) -> ExitCode { }, None => None, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -376,7 +376,7 @@ pub async fn run_replica_verb(cli: &Cli, verb: ReplicaVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/kv/server.rs b/app/crowdb-cli/src/commands/kv/server.rs index 260fff1e3..652528c13 100644 --- a/app/crowdb-cli/src/commands/kv/server.rs +++ b/app/crowdb-cli/src/commands/kv/server.rs @@ -1,16 +1,12 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! `kv server` command handlers — deploy/restart/stop/delete/list. +//! `kv server` process controls backed by a launch-only registry. -use std::path::PathBuf; use std::process::ExitCode; use clap::Subcommand; -use crowdb_console_shared::lifecycle::DeployRequest; -use crowdb_protocol::NodeId; -use crate::commands::{commit_config, op_context}; use crate::Cli; mod registry; @@ -20,12 +16,6 @@ pub enum KvServerVerb { Deploy { #[arg(short = 'n', long)] node: String, - #[arg(short = 'r', long)] - rest_port: Option, - #[arg(short = 'R', long)] - rpc_port: Option, - #[arg(short = 'b', long)] - binary: Option, }, Start { #[arg(short = 'n', long)] @@ -46,163 +36,10 @@ pub enum KvServerVerb { List, } -#[allow(clippy::too_many_lines)] pub async fn run_kv_server_verb(cli: &Cli, verb: KvServerVerb) -> ExitCode { - if let Some(path) = &cli.registry { - return registry::run(cli, path, verb).await; - } - match verb { - KvServerVerb::Deploy { - node, - rest_port, - rpc_port, - binary, - } => { - let (Some(rest_port), Some(rpc_port)) = (rest_port, rpc_port) else { - eprintln!("error: deploy requires management and RPC ports without --registry"); - return ExitCode::from(2); - }; - let node_id: NodeId = match node.parse() { - Ok(n) => n, - Err(e) => { - eprintln!("error: invalid node id: {e}"); - return ExitCode::from(1); - } - }; - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - let req = DeployRequest { - server_id: node_id.to_string(), - rest_port, - rpc_port, - binary: binary.map(PathBuf::from), - ..Default::default() - }; - match crowdb_console_shared::ops::kv_server::deploy(&ctx, &req, None).await { - Ok(d) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - println!( - "deployed server on node {} -> {} (pid {}, rpc {})", - node_id, d.mgmt_url, d.pid, d.rpc_url - ); - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: deploy on node {node_id}: {e}"); - ExitCode::from(2) - } - } - } - KvServerVerb::Restart { node } | KvServerVerb::Start { node } => { - let node_id: NodeId = match node.parse() { - Ok(n) => n, - Err(e) => { - eprintln!("error: invalid node id: {e}"); - return ExitCode::from(1); - } - }; - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - match crowdb_console_shared::ops::kv_server::restart(&ctx, node_id, None, None, &[]).await { - Ok(d) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - println!( - "restarted server on node {} -> {} (pid {}, rpc {})", - node_id, d.mgmt_url, d.pid, d.rpc_url - ); - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: restart on node {node_id}: {e}"); - ExitCode::from(2) - } - } - } - KvServerVerb::Stop { node } => { - let node_id: NodeId = match node.parse() { - Ok(n) => n, - Err(e) => { - eprintln!("error: invalid node id: {e}"); - return ExitCode::from(1); - } - }; - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - match crowdb_console_shared::ops::kv_server::stop(&ctx, node_id, None).await { - Ok(sent) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - if sent { - println!("sent SIGTERM to server on node {node_id}"); - } else { - println!("server on node {node_id} was already gone"); - } - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: stop on node {node_id}: {e}"); - ExitCode::from(2) - } - } - } - KvServerVerb::Delete { node } => { - let node_id: NodeId = match node.parse() { - Ok(n) => n, - Err(e) => { - eprintln!("error: invalid node id: {e}"); - return ExitCode::from(1); - } - }; - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - match crowdb_console_shared::ops::kv_server::delete(&ctx, node_id).await { - Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - println!("deleted server on node {node_id}"); - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: delete on node {node_id}: {e}"); - ExitCode::from(2) - } - } - } - KvServerVerb::List => { - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - let servers = crowdb_console_shared::ops::kv_server::list(&ctx); - if servers.is_empty() { - println!("(no servers deployed)"); - return ExitCode::SUCCESS; - } - println!("{:<12} {:<26} {:<26} {:<8}", "NODE", "MGMT", "RPC", "PID"); - for s in &servers { - println!( - "{:<12} {:<26} {:<26} {:<8}", - s.node_id.map_or_else(|| "-".into(), |n| n.to_string()), - s.url, - s.rpc_url.as_deref().unwrap_or("-"), - s.pid.map_or_else(|| "-".into(), |p| p.to_string()), - ); - } - ExitCode::SUCCESS - } - } + let Some(path) = &cli.registry else { + eprintln!("error: kv server controls require --registry"); + return ExitCode::from(2); + }; + registry::run(cli, path, verb).await } diff --git a/app/crowdb-cli/src/commands/kv/server/registry.rs b/app/crowdb-cli/src/commands/kv/server/registry.rs index 16e6ed993..30b993a03 100644 --- a/app/crowdb-cli/src/commands/kv/server/registry.rs +++ b/app/crowdb-cli/src/commands/kv/server/registry.rs @@ -64,22 +64,7 @@ async fn execute(cli: &Cli, path: &Path, verb: KvServerVerb) -> Result<()> { id: node.to_string(), })?; match verb { - KvServerVerb::Deploy { - rest_port, - rpc_port, - binary, - .. - } => { - if rest_port.is_some() || rpc_port.is_some() || binary.is_some() { - return Err(Error::Validation { - field: "registry".into(), - message: "binary and listener arguments must come from the launch registry".into(), - }); - } - let identity = runtime.start(&launch).await?; - println!("started kv on node {node} (pid {})", identity.pid); - } - KvServerVerb::Start { .. } => { + KvServerVerb::Deploy { .. } | KvServerVerb::Start { .. } => { let identity = runtime.start(&launch).await?; println!("started kv on node {node} (pid {})", identity.pid); } @@ -92,7 +77,8 @@ async fn execute(cli: &Cli, path: &Path, verb: KvServerVerb) -> Result<()> { println!("stopped kv on node {node}"); } KvServerVerb::Delete { .. } => { - let ctx = crate::commands::op_context(cli) + let ctx = crate::commands::authority_context(cli) + .await .map_err(|_| Error::Config("cannot initialize authority client".into()))?; if ctx .sysmd() diff --git a/app/crowdb-cli/tests/cluster_cli_test.rs b/app/crowdb-cli/tests/cluster_cli_test.rs index 23efa74ce..10e5f4118 100644 --- a/app/crowdb-cli/tests/cluster_cli_test.rs +++ b/app/crowdb-cli/tests/cluster_cli_test.rs @@ -71,21 +71,21 @@ async fn cluster_status_topology_via_direct_group0() { return; } - // cluster init — writes store/group/replica topology into group-0 - // sysdata (idempotent: group 0 already exists from spawn_group0, - // init handles the 409 conflict and still writes topology). + // Bootstrap requires an independent versioned intent and launch registry. let (code, _, stderr) = run( &cli, g0.mgmt_port, &g0.config_path, &["cluster", "init", "-n", "1"], ); - assert_eq!(code, 0, "cluster init stderr={stderr}"); + assert_eq!(code, 2, "cluster init stderr={stderr}"); + assert!(stderr.contains("--registry"), "stderr={stderr}"); + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); // status — lists stores from group-0 sysdata. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "status"]); assert_eq!(code, 0, "status stderr={stderr}"); - assert!(stdout.contains('0'), "stdout={stdout}"); + assert!(stdout.contains("(no stores)"), "stdout={stdout}"); // topology — from a node's /topology endpoint. let (code, stdout, stderr) = run( @@ -97,10 +97,10 @@ async fn cluster_status_topology_via_direct_group0() { assert_eq!(code, 0, "topology stderr={stderr}"); assert!(stdout.contains("store"), "stdout={stdout}"); - // Status always uses the human-readable console table. + // Status reflects the uninitialized authority without a local fallback. let (code, stdout, _) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "status"]); assert_eq!(code, 0); - assert!(stdout.contains("STORE"), "stdout={stdout}"); + assert!(stdout.contains("(no stores)"), "stdout={stdout}"); tokio::time::sleep(Duration::from_millis(50)).await; } diff --git a/app/crowdb-cli/tests/common/console.rs b/app/crowdb-cli/tests/common/console.rs index 58d1790a8..e8a1265c8 100644 --- a/app/crowdb-cli/tests/common/console.rs +++ b/app/crowdb-cli/tests/common/console.rs @@ -136,6 +136,7 @@ pub fn local_node(id: u64, rack: u64) -> NodeEntry { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, } } diff --git a/app/crowdb-cli/tests/common/direct.rs b/app/crowdb-cli/tests/common/direct.rs index 3b54c4612..0516de9a8 100644 --- a/app/crowdb-cli/tests/common/direct.rs +++ b/app/crowdb-cli/tests/common/direct.rs @@ -17,7 +17,7 @@ use std::time::Duration; use crowdb_console_shared::clients::http::ServerClient; use crowdb_console_shared::config::{NodeEntry, RackEntry, ServerEntry, ServiceType}; use crowdb_console_shared::lifecycle::{self, crowdb_kv_server_bin, DeployRequest}; -use crowdb_console_shared::{ConsoleConfig, ConsoleConfigEngine}; +use crowdb_console_shared::ConsoleConfig; use crowdb_test_harness::test_dirs; /// Allocate a free mgmt port for a kv-server. @@ -61,6 +61,7 @@ pub struct Group0 { pub mgmt_port: u16, pub rpc_port: u16, pub config_path: PathBuf, + pub bootstrap_config: ConsoleConfig, workspace: std::path::PathBuf, } @@ -85,11 +86,12 @@ pub fn local_node(id: u64, rack: u64) -> NodeEntry { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, } } /// Fork a real `crowdb-kv-server` for node 1, initialize group 0 on it, -/// and write a console config with the rack/node/server entries. Returns +/// and prepare in-memory bootstrap entries. Returns /// `None` when the server binary has not been built. pub async fn spawn_group0() -> Option { let bin = crowdb_kv_server_bin()?; @@ -122,7 +124,7 @@ pub async fn spawn_group0() -> Option { .await .expect("deploy_local_in_dir"); - // Write console config with rack/node/server entries. + // Prepare bootstrap input without persisting a topology copy. let mut cfg = ConsoleConfig::default(); cfg.racks.push(RackEntry { id: 1, @@ -147,8 +149,6 @@ pub async fn spawn_group0() -> Option { .unwrap(); let config_path = workspace.join("console.toml"); - let engine = crowdb_console_shared::TomlFileEngine::new(config_path.clone()); - engine.save(&cfg).expect("save config"); // Initialize group 0 on the server (single-node, self-elect). let client = ServerClient::new(deployed.mgmt_url.clone()).unwrap(); @@ -165,7 +165,7 @@ pub async fn spawn_group0() -> Option { let context = crowdb_console_shared::ops::OpContext::new( deployed.rpc_url.trim_start_matches("http://").to_string(), vec![deployed.mgmt_url.clone()], - cfg, + cfg.clone(), ); let deadline = std::time::Instant::now() + Duration::from_secs(5); loop { @@ -178,6 +178,12 @@ pub async fn spawn_group0() -> Option { ); tokio::time::sleep(Duration::from_millis(100)).await; } + crowdb_console_shared::ops::hardware::add_rack_to_group0(&context, 1, "rack-1") + .await + .expect("publish rack to Group 0"); + crowdb_console_shared::ops::hardware::add_node_to_group0(&context, local_node(1, 1)) + .await + .expect("publish node to Group 0"); Some(Group0 { pid: deployed.pid, @@ -186,6 +192,7 @@ pub async fn spawn_group0() -> Option { mgmt_port: rest_port, rpc_port, config_path, + bootstrap_config: cfg, workspace, }) } diff --git a/app/crowdb-cli/tests/direct_cli_test.rs b/app/crowdb-cli/tests/direct_cli_test.rs index 443942cf2..5a33bce5b 100644 --- a/app/crowdb-cli/tests/direct_cli_test.rs +++ b/app/crowdb-cli/tests/direct_cli_test.rs @@ -10,6 +10,16 @@ mod common; use common::direct::{crowdb_cli_bin, run, spawn_group0}; +#[test] +fn clean_rejects_service_restart_without_launch_registry() { + let output = std::process::Command::new(crowdb_cli_bin()) + .args(["cluster", "clean", "--restart-services"]) + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(2)); + assert!(String::from_utf8_lossy(&output.stderr).contains("--registry")); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn cluster_status_via_direct_group0() { let Some(g0) = spawn_group0().await else { @@ -33,7 +43,7 @@ async fn cluster_status_via_direct_group0() { } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn cluster_rack_list_via_direct_config() { +async fn cluster_rack_list_ignores_legacy_local_state() { let Some(g0) = spawn_group0().await else { eprintln!("skipping: crowdb-kv-server binary not built"); return; @@ -44,14 +54,15 @@ async fn cluster_rack_list_via_direct_config() { return; } - // `cluster rack list` should list rack 1 (from the config). + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); + // `cluster rack list` reads rack 1 from Group 0 despite the old file. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "rack", "list"]); assert_eq!(code, 0, "cluster rack list stderr={stderr}"); assert!(stdout.contains('1'), "cluster rack list stdout={stdout}"); } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn cluster_node_list_via_direct_config() { +async fn cluster_node_list_ignores_legacy_local_state() { let Some(g0) = spawn_group0().await else { eprintln!("skipping: crowdb-kv-server binary not built"); return; @@ -62,14 +73,15 @@ async fn cluster_node_list_via_direct_config() { return; } - // `cluster node list` should list node 1 (from the config). + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); + // `cluster node list` reads node 1 from Group 0 despite the old file. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "node", "list"]); assert_eq!(code, 0, "cluster node list stderr={stderr}"); assert!(stdout.contains('1'), "cluster node list stdout={stdout}"); } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn kv_server_list_via_direct_config() { +async fn kv_server_controls_require_launch_registry() { let Some(g0) = spawn_group0().await else { eprintln!("skipping: crowdb-kv-server binary not built"); return; @@ -80,8 +92,71 @@ async fn kv_server_list_via_direct_config() { return; } - // `kv server list` should list the server on node 1. + // A legacy topology file cannot supply process launch policy. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["kv", "server", "list"]); - assert_eq!(code, 0, "kv server list stderr={stderr}"); - assert!(stdout.contains('1'), "kv server list stdout={stdout}"); + assert_eq!(code, 2, "stdout={stdout} stderr={stderr}"); + assert!(stderr.contains("--registry"), "stderr={stderr}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn incremental_local_deploy_reads_group_zero_instead_of_legacy_file() { + let Some(g0) = spawn_group0().await else { + return; + }; + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); + let (code, _, stderr) = run( + &crowdb_cli_bin(), + g0.mgmt_port, + &g0.config_path, + &["cluster", "local-deploy", "-t", "diskdb", "--data-groups", "99"], + ); + assert_eq!(code, 2, "stderr={stderr}"); + assert!(stderr.contains("99"), "stderr={stderr}"); + assert!(!stderr.contains("load config"), "stderr={stderr}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn destroy_reads_group_zero_and_requires_launch_registry() { + let Some(g0) = spawn_group0().await else { + return; + }; + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); + let (code, _, stderr) = run( + &crowdb_cli_bin(), + g0.mgmt_port, + &g0.config_path, + &["cluster", "destroy"], + ); + assert_eq!(code, 2, "stderr={stderr}"); + assert!(stderr.contains("--registry"), "stderr={stderr}"); + + let context = crowdb_console_shared::ops::OpContext::new( + g0.rpc_url.trim_start_matches("http://").to_string(), + vec![g0.mgmt_url.clone()], + g0.bootstrap_config.clone(), + ); + crowdb_console_shared::ops::cluster::init(&context, &[1]) + .await + .expect("publish confirmed bootstrap metadata"); + + let registry_path = g0.config_path.with_file_name("launches.toml"); + crowdb_console_shared::config::web::LaunchRegistry { + version: 1, + launches: Vec::new(), + } + .save(®istry_path) + .unwrap(); + let output = std::process::Command::new(crowdb_cli_bin()) + .args(["--registry", registry_path.to_str().unwrap(), "--system-port"]) + .arg(g0.mgmt_port.to_string()) + .args(["cluster", "destroy"]) + .env("CROWDB_CLI_STATE", &g0.config_path) + .output() + .unwrap(); + assert!( + output.status.success(), + "stdout={} stderr={}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); } diff --git a/app/crowdb-cli/tests/kv_cli_test.rs b/app/crowdb-cli/tests/kv_cli_test.rs index 23730a2e5..84ea4bd43 100644 --- a/app/crowdb-cli/tests/kv_cli_test.rs +++ b/app/crowdb-cli/tests/kv_cli_test.rs @@ -22,6 +22,7 @@ async fn kv_put_get_delete_round_trip() { eprintln!("skipping: crowdb-cli binary not built ({})", cli.display()); return; } + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); // put on group 0 (system store). let (code, stdout, stderr) = run( diff --git a/app/crowdb-cli/tests/launch_registry_cli_test.rs b/app/crowdb-cli/tests/launch_registry_cli_test.rs index c3640d115..821f24525 100644 --- a/app/crowdb-cli/tests/launch_registry_cli_test.rs +++ b/app/crowdb-cli/tests/launch_registry_cli_test.rs @@ -10,6 +10,7 @@ use std::path::Path; use std::process::Command; use std::time::Duration; +use crowdb_console_shared::bootstrap_intent::BootstrapIntent; use crowdb_console_shared::config::web::{LaunchRecord, LaunchRegistry}; use crowdb_console_shared::launch::LaunchRuntime; use crowdb_console_shared::lifecycle; @@ -169,3 +170,97 @@ async fn chunk_commands_and_generic_launch_controls_share_process_identity() { } assert_eq!(LaunchRegistry::load(&path).unwrap().launches, records); } + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn registry_hardware_uses_group_zero_across_cli_invocations() { + let g0 = common::direct::spawn_group0() + .await + .expect("KV server binary must be built"); + let dir = tempdir_in_test_data("cli-registry-hardware"); + let path = dir.path().join("launches.toml"); + LaunchRegistry { + version: 1, + launches: Vec::new(), + } + .save(&path) + .unwrap(); + + run_command( + &path, + g0.mgmt_port, + &["cluster", "rack", "add", "--id", "2", "--name", "rack-two"], + ); + assert!(run_command(&path, g0.mgmt_port, &["cluster", "rack", "list"]).contains("rack-two")); + run_command( + &path, + g0.mgmt_port, + &[ + "cluster", + "node", + "add", + "--id", + "2", + "--rack", + "2", + "--host", + "10.0.0.2", + "--ssh-credential-ref", + "ops-key", + ], + ); + assert!(run_command(&path, g0.mgmt_port, &["cluster", "node", "list"]).contains("10.0.0.2")); + run_command( + &path, + g0.mgmt_port, + &["cluster", "rack", "add", "--id", "3", "--name", "empty"], + ); + run_command(&path, g0.mgmt_port, &["cluster", "rack", "remove", "--id", "3"]); + assert!(!run_command(&path, g0.mgmt_port, &["cluster", "rack", "list"]).contains("empty")); + run_command(&path, g0.mgmt_port, &["cluster", "node", "remove", "--id", "2"]); + assert!(!run_command(&path, g0.mgmt_port, &["cluster", "node", "list"]).contains("10.0.0.2")); + run_command(&path, g0.mgmt_port, &["cluster", "rack", "remove", "--id", "2"]); + assert!(!dir.path().join("invalid-legacy.toml").exists()); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn registry_bootstrap_uses_sealed_intent_without_legacy_topology_file() { + let g0 = common::direct::spawn_group0() + .await + .expect("KV server binary must be built"); + let dir = tempdir_in_test_data("cli-registry-bootstrap"); + let path = dir.path().join("launches.toml"); + LaunchRegistry { + version: 1, + launches: Vec::new(), + } + .save(&path) + .unwrap(); + let source = dir.path().join("bootstrap-source.toml"); + let intent = BootstrapIntent::capture(&g0.bootstrap_config, &[1]).unwrap(); + intent.seal(&source).unwrap(); + std::fs::write(dir.path().join("invalid-legacy.toml"), "invalid legacy config").unwrap(); + + run_command( + &path, + g0.mgmt_port, + &[ + "cluster", + "init", + "--nodes", + "1", + "--bootstrap-file", + source.to_str().unwrap(), + ], + ); + assert!(!path.with_extension("bootstrap-intent.toml").exists()); + intent + .seal(&path.with_extension("bootstrap-intent.toml")) + .unwrap(); + run_command(&path, g0.mgmt_port, &["cluster", "init", "--nodes", "1"]); + assert!(!path.with_extension("bootstrap-intent.toml").exists()); + assert_eq!( + std::fs::read_to_string(dir.path().join("invalid-legacy.toml")).unwrap(), + "invalid legacy config" + ); + assert!(run_command(&path, g0.mgmt_port, &["cluster", "rack", "list"]).contains("rack-1")); +} diff --git a/app/crowdb-cli/tests/lifecycle_cli_test.rs b/app/crowdb-cli/tests/lifecycle_cli_test.rs index 1fdd75599..04b11dd40 100644 --- a/app/crowdb-cli/tests/lifecycle_cli_test.rs +++ b/app/crowdb-cli/tests/lifecycle_cli_test.rs @@ -1,19 +1,16 @@ // Copyright 2026-present Gian -//! CLI e2e for the physical lifecycle verbs: `cluster rack/node` and -//! `kv server` round-trips through a system-group endpoint against a real +//! CLI e2e for `cluster rack/node` round-trips through a system-group endpoint against a real //! `crowdb-kv-server` with the system group //! initialized — no `crowdb-web` intermediary. mod common; -use std::time::Duration; - use common::direct::{crowdb_cli_bin, run, spawn_group0}; #[tokio::test(flavor = "multi_thread", worker_threads = 2)] #[allow(clippy::too_many_lines)] -async fn rack_node_server_lifecycle() { +async fn rack_node_lifecycle() { let Some(g0) = spawn_group0().await else { eprintln!("skipping: crowdb-kv-server binary not built"); return; @@ -67,48 +64,4 @@ async fn rack_node_server_lifecycle() { !stdout.contains("2 2"), "node 2 should be gone: stdout={stdout}" ); - - // kv server list — server on node 1 already exists. - let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["kv", "server", "list"]); - assert_eq!(code, 0, "server list stderr={stderr}"); - assert!(stdout.contains('1'), "stdout={stdout}"); - - // kv server restart — recover node 1 after an out-of-band process exit. - let pid = g0.pid; - tokio::task::spawn_blocking(move || { - let _ = crowdb_console_shared::lifecycle::stop_pid_with_timeout(pid, Duration::from_millis(100)); - }) - .await - .unwrap(); - let (code, stdout, stderr) = run( - &cli, - g0.mgmt_port, - &g0.config_path, - &["kv", "server", "restart", "--node", "1"], - ); - assert_eq!(code, 0, "server restart stdout={stdout}\nstderr={stderr}"); - - let restarted = crowdb_console_shared::ConsoleConfig::load(&g0.config_path).unwrap(); - let restarted_pid = restarted.server_for_node(1).unwrap().pid.unwrap(); - tokio::task::spawn_blocking(move || { - let _ = crowdb_console_shared::lifecycle::stop_pid_with_timeout( - restarted_pid, - Duration::from_millis(100), - ); - }) - .await - .unwrap(); - - // kv server stop — clear the deployment state after an out-of-band exit. - let (code, _, stderr) = run( - &cli, - g0.mgmt_port, - &g0.config_path, - &["kv", "server", "stop", "--node", "1"], - ); - assert_eq!(code, 0, "server stop stderr={stderr}"); - let stopped = crowdb_console_shared::ConsoleConfig::load(&g0.config_path).unwrap(); - assert!(stopped.server_for_node(1).unwrap().pid.is_none()); - - tokio::time::sleep(Duration::from_millis(100)).await; } diff --git a/app/crowdb-cli/tests/mgmt_cli_test.rs b/app/crowdb-cli/tests/mgmt_cli_test.rs index e4f3eeec1..b75a15f7c 100644 --- a/app/crowdb-cli/tests/mgmt_cli_test.rs +++ b/app/crowdb-cli/tests/mgmt_cli_test.rs @@ -22,6 +22,7 @@ async fn store_group_replica_round_trip() { eprintln!("skipping: crowdb-cli binary not built ({})", cli.display()); return; } + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); let store_id = "9"; let group_id = "90"; diff --git a/app/crowdb-cli/tests/s3_cli_test.rs b/app/crowdb-cli/tests/s3_cli_test.rs index c46a200bd..1ee9da093 100644 --- a/app/crowdb-cli/tests/s3_cli_test.rs +++ b/app/crowdb-cli/tests/s3_cli_test.rs @@ -1,6 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. +use crowdb_test_harness::test_dirs::TestDir; use std::io::{Read, Write}; use std::net::TcpListener; use std::path::{Path, PathBuf}; @@ -11,6 +12,90 @@ fn cli() -> Command { Command::new(env!("CARGO_BIN_EXE_crowdb-cli")) } +#[test] +#[ignore = "starts the complete local storage stack twice"] +fn interrupted_s3_launch_resumes_confirmed_group_zero_without_topology_file() { + let directory = TestDir::new("s3-bootstrap-replay-cli").expect("test directory"); + let root = directory.path(); + let workspace = crowdb_test_harness::test_dirs::workspace_root(); + let relative_root = root + .strip_prefix(&workspace) + .expect("test root is below workspace"); + let failed = cli() + .current_dir(&workspace) + .args(["s3", "cluster", "start", "--root"]) + .arg(relative_root) + .env("CROWDB_CHUNK_KV_SERVER_BIN", "/bin/false") + .output() + .expect("run interrupted launch"); + assert!(!failed.status.success(), "failure injection must stop launch"); + assert!(root.join("s3-mini-cluster.initializing.json").exists()); + assert!(root.join("s3-local-state.toml").exists()); + assert!(!root.join("console.toml").exists()); + let restarted = cli() + .args(["s3", "cluster", "start", "--root"]) + .arg(root) + .env_remove("CROWDB_CHUNK_KV_SERVER_BIN") + .output() + .expect("resume interrupted launch"); + assert!( + restarted.status.success(), + "{}", + String::from_utf8_lossy(&restarted.stderr) + ); + assert!(!root.join("bootstrap-intent.toml").exists()); + assert!(!root.join("s3-mini-cluster.initializing.json").exists()); + let deleted = cli() + .args(["s3", "cluster", "delete", "--root"]) + .arg(root) + .output() + .expect("delete test cluster"); + assert!( + deleted.status.success(), + "{}", + String::from_utf8_lossy(&deleted.stderr) + ); +} + +#[test] +#[ignore = "starts the complete local storage stack twice"] +fn interrupted_s3_storage_launch_replays_without_local_topology() { + let directory = TestDir::new("s3-storage-replay-cli").expect("test directory"); + let root = directory.path(); + let failed = cli() + .args(["s3", "cluster", "start", "--root"]) + .arg(root) + .env("CROWDB_DISKIO_BIN", "/bin/false") + .output() + .expect("run interrupted storage launch"); + assert!(!failed.status.success(), "failure injection must stop launch"); + let local = std::fs::read_to_string(root.join("s3-local-state.toml")).unwrap(); + assert!(local.contains("diskdb-1"), "{local}"); + assert!(!local.contains("[[rack]]")); + let restarted = cli() + .args(["s3", "cluster", "start", "--root"]) + .arg(root) + .env_remove("CROWDB_DISKIO_BIN") + .output() + .expect("resume interrupted storage launch"); + assert!( + restarted.status.success(), + "{}", + String::from_utf8_lossy(&restarted.stderr) + ); + assert!(!root.join("s3-mini-cluster.initializing.json").exists()); + let deleted = cli() + .args(["s3", "cluster", "delete", "--root"]) + .arg(root) + .output() + .expect("delete test cluster"); + assert!( + deleted.status.success(), + "{}", + String::from_utf8_lossy(&deleted.stderr) + ); +} + fn tempdir(tag: &str) -> PathBuf { let nonce = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) @@ -34,6 +119,11 @@ fn write_cluster_record(root: &Path, endpoint: &str) { serde_json::to_vec_pretty(&record).expect("record json"), ) .expect("write record"); + std::fs::write( + root.join("s3-local-state.toml"), + "version = 1\ngroup0_seeds = ['http://127.0.0.1:10000']\n[[service]]\nid = 'kv-1'\nurl = 'http://127.0.0.1:10000'\nnode_id = 1\n", + ) + .expect("write launch-only state"); } fn mock_http_once(response_content_type: &str, response_body: &[u8]) -> (String, mpsc::Receiver>) { diff --git a/app/crowdb-diskdb/tests/diskdb_e2e_test.rs b/app/crowdb-diskdb/tests/diskdb_e2e_test.rs index 99f3deb54..bd0761a4a 100644 --- a/app/crowdb-diskdb/tests/diskdb_e2e_test.rs +++ b/app/crowdb-diskdb/tests/diskdb_e2e_test.rs @@ -76,6 +76,7 @@ async fn seed_hardware(hw: &HardwareClient) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![NODE_ID], + ..Default::default() }, ) .await @@ -91,6 +92,7 @@ async fn seed_hardware(hw: &HardwareClient) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await @@ -105,6 +107,7 @@ async fn seed_hardware(hw: &HardwareClient) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.clone(), + name: String::new(), }, ) .await diff --git a/app/crowdb-diskdb/tests/recovery_test.rs b/app/crowdb-diskdb/tests/recovery_test.rs index 2f0b01419..934a2679b 100644 --- a/app/crowdb-diskdb/tests/recovery_test.rs +++ b/app/crowdb-diskdb/tests/recovery_test.rs @@ -57,6 +57,7 @@ async fn seed_hardware(hw: &HardwareClient) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![NODE_ID], + ..Default::default() }, ) .await @@ -70,6 +71,7 @@ async fn seed_hardware(hw: &HardwareClient) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await @@ -82,6 +84,7 @@ async fn seed_hardware(hw: &HardwareClient) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.clone(), + name: String::new(), }, ) .await diff --git a/app/crowdb-diskdb/tests/relocation_journal_test.rs b/app/crowdb-diskdb/tests/relocation_journal_test.rs index 48ea2f75b..f11ace9f7 100644 --- a/app/crowdb-diskdb/tests/relocation_journal_test.rs +++ b/app/crowdb-diskdb/tests/relocation_journal_test.rs @@ -105,6 +105,7 @@ async fn seed_hardware(hw: &HardwareClient) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![10], + ..Default::default() }, ) .await @@ -118,6 +119,7 @@ async fn seed_hardware(hw: &HardwareClient) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await @@ -134,6 +136,7 @@ async fn seed_hardware(hw: &HardwareClient) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.clone(), + name: String::new(), }, ) .await diff --git a/app/crowdb-diskdb/tests/scanner_test.rs b/app/crowdb-diskdb/tests/scanner_test.rs index 6d4e1e728..ef96f859e 100644 --- a/app/crowdb-diskdb/tests/scanner_test.rs +++ b/app/crowdb-diskdb/tests/scanner_test.rs @@ -67,6 +67,7 @@ async fn seed_hardware(hw: &HardwareClient) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![NODE_ID], + ..Default::default() }, ) .await @@ -80,6 +81,7 @@ async fn seed_hardware(hw: &HardwareClient) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await @@ -92,6 +94,7 @@ async fn seed_hardware(hw: &HardwareClient) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.clone(), + name: String::new(), }, ) .await diff --git a/app/crowdb-diskio/CMakeLists.txt b/app/crowdb-diskio/CMakeLists.txt index b9898ef09..8b0676305 100644 --- a/app/crowdb-diskio/CMakeLists.txt +++ b/app/crowdb-diskio/CMakeLists.txt @@ -25,7 +25,7 @@ add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/../../lib/crowdb-rpc crowdb-rpc-bui # when built with the `ffi` feature. set(CROWDB_KV_CLIENT_DIR ${CMAKE_CURRENT_SOURCE_DIR}/../../lib/crowdb-kv-client) set(CROWDB_ROOT_DIR ${CMAKE_CURRENT_SOURCE_DIR}/../..) -if(CMAKE_BUILD_TYPE STREQUAL "Release") +if(CMAKE_BUILD_TYPE STREQUAL "Release" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") set(CARGO_PROFILE release) else() set(CARGO_PROFILE debug) diff --git a/app/crowdb-web/src/lib.rs b/app/crowdb-web/src/lib.rs index 9a8986b93..3891e4284 100644 --- a/app/crowdb-web/src/lib.rs +++ b/app/crowdb-web/src/lib.rs @@ -19,6 +19,7 @@ pub mod kv; mod launch; pub mod lifecycle; mod managed; +mod managed_hardware; mod managed_logical; pub mod mgmt; pub mod owner_assignment; @@ -75,7 +76,59 @@ pub fn router(state: AppState) -> axum::Router { .route("/api/*path", any(health::managed_api_unavailable)) .fallback(spa::spa_fallback); let managed = if state.web_mode == Some(crowdb_console_shared::config::web::WebMode::BareMetal) { - managed.merge(launch::routes().route_layer(authorization)) + let hardware = axum::Router::new() + .route( + "/api/cluster/init", + post(mgmt::http_cluster_init).route_layer(authorization.clone()), + ) + .route( + "/api/racks", + get(managed_hardware::list_racks) + .merge(post(managed_hardware::add_rack).route_layer(authorization.clone())), + ) + .route( + "/api/racks/:rack_id", + get(managed_hardware::get_rack) + .merge(delete(managed_hardware::remove_rack).route_layer(authorization.clone())), + ) + .route( + "/api/racks/:rack_id/nodes", + get(managed_hardware::list_rack_nodes), + ) + .route( + "/api/nodes", + get(managed_hardware::list_nodes) + .merge(post(managed_hardware::add_node).route_layer(authorization.clone())), + ) + .route( + "/api/nodes/:id", + get(managed_hardware::get_node) + .merge(delete(managed_hardware::remove_node).route_layer(authorization.clone())), + ) + .route( + "/api/nodes/:id/disk-groups", + get(managed_hardware::list_disk_groups) + .merge(post(managed_hardware::add_disk_group).route_layer(authorization.clone())), + ) + .route( + "/api/nodes/:id/disk-groups/:dg_id", + get(managed_hardware::get_disk_group).merge( + delete(managed_hardware::remove_disk_group).route_layer(authorization.clone()), + ), + ) + .route( + "/api/nodes/:id/disk-groups/:dg_id/disks", + get(managed_hardware::list_disks) + .merge(post(managed_hardware::add_disk).route_layer(authorization.clone())), + ) + .route( + "/api/nodes/:id/disk-groups/:dg_id/disks/:disk_id", + get(managed_hardware::get_disk) + .merge(delete(managed_hardware::remove_disk).route_layer(authorization.clone())), + ); + managed + .merge(hardware) + .merge(launch::routes().route_layer(authorization)) } else { managed }; @@ -254,7 +307,6 @@ pub fn router(state: AppState) -> axum::Router { // ── Cluster init (R2): system group bootstrap ──────────────── .route("/api/cluster/init", post(mgmt::http_cluster_init)) .route("/api/cluster/destroy", post(lifecycle::http_internal_reset)) - .route("/api/cluster/reset", post(lifecycle::http_cluster_reset)) .route("/api/cluster/clean", post(lifecycle::http_cluster_clean)) // ── Internal: E2E test reset (alias for destroy) ───────────── .route("/internal/reset", post(lifecycle::http_internal_reset)) diff --git a/app/crowdb-web/src/lifecycle.rs b/app/crowdb-web/src/lifecycle.rs index 90c736684..6eeb1bbd3 100644 --- a/app/crowdb-web/src/lifecycle.rs +++ b/app/crowdb-web/src/lifecycle.rs @@ -113,6 +113,7 @@ pub async fn http_add_rack( let value = crowdb_protocol::common::RackValue { status: crowdb_protocol::common::HwStatus::Up as i32, node_ids: Vec::new(), + name: entry.name.clone(), }; let _ = ctx.sysmd().add_rack(body.id, &value).await; } @@ -193,6 +194,10 @@ pub async fn http_add_node( disk_group_ids: Vec::new(), status_changed_at_ms: 0, temp_failure_since_ms: None, + management_host: entry.host.clone(), + ssh_port: entry.ssh_port, + ssh_user: entry.ssh_user.clone(), + ssh_credential_ref: entry.ssh_credential_ref.clone(), }; let _ = ctx.sysmd().add_node(entry.rack_id, entry.id, &value).await; } @@ -468,6 +473,10 @@ pub async fn http_add_rack_node( disk_group_ids: Vec::new(), status_changed_at_ms: 0, temp_failure_since_ms: None, + management_host: entry.host.clone(), + ssh_port: entry.ssh_port, + ssh_user: entry.ssh_user.clone(), + ssh_credential_ref: entry.ssh_credential_ref.clone(), }; let _ = ctx.sysmd().add_node(entry.rack_id, entry.id, &value).await; } @@ -1038,22 +1047,6 @@ pub async fn http_cluster_clean( .map_err(|e| err_502(format!("{e}"))) } -/// `POST /api/cluster/reset`. Remove orphaned sysdata entries -/// (stores/groups/replicas that have no corresponding running server). -/// Does not stop any running servers. -/// -/// # Errors -/// Returns `502` if the sysdata scan fails. -pub async fn http_cluster_reset( - State(state): State, -) -> Result)> { - let ctx = state.op_context().await.map_err(|e| err_502(format!("{e}")))?; - ops::cluster::reset(&ctx) - .await - .map_err(|e| err_502(format!("{e}")))?; - Ok(StatusCode::NO_CONTENT) -} - /// `POST /api/cluster/destroy` (alias: `/internal/reset`). Tear down /// the entire cluster in dependency order: groups → stores → server /// processes → nodes → racks, then clear workspace dirs and caches. diff --git a/app/crowdb-web/src/main.rs b/app/crowdb-web/src/main.rs index 567809290..880f3617a 100644 --- a/app/crowdb-web/src/main.rs +++ b/app/crowdb-web/src/main.rs @@ -68,6 +68,9 @@ struct Args { #[tokio::main] async fn main() -> Result<(), Box> { let args = Args::parse(); + if args.config.is_none() && !args.test_mode { + return Err("crowdb-web requires a versioned --config outside test mode".into()); + } let process_config = args.config.as_deref().map(WebProcessConfig::load).transpose()?; if args.registry.is_some() && process_config @@ -89,22 +92,7 @@ async fn main() -> Result<(), Box> { let addr: SocketAddr = format!("{bind}:{port}").parse()?; info!(%addr, "crowdb-web starting"); - // Load the persisted registry; absence yields an empty default. - // Mutating handlers (rack/node/server CRUD) write back to this path. - let path = if args.test_mode || process_config.is_some() { - None - } else { - crowdb_console_shared::TomlFileEngine::default_path() - }; - let cfg = match path.as_ref() { - Some(p) => { - let engine = crowdb_console_shared::TomlFileEngine::new(p.clone()); - crowdb_console_shared::ConsoleConfig::load_with_engine(&engine)? - } - None => crowdb_console_shared::ConsoleConfig::default(), - }; - let server_count = cfg.servers.len(); - let mut state = crowdb_web::AppState::with_config(cfg, path).with_test_mode(args.test_mode); + let mut state = crowdb_web::AppState::default().with_test_mode(args.test_mode); if let Some(config) = process_config { state = state.with_process_config(&config); state = state.with_management_token(std::env::var("CROWDB_ICEBERG_MANAGE_TOKEN")?)?; @@ -115,7 +103,7 @@ async fn main() -> Result<(), Box> { info!(started, "reconciled configured service launches"); } tracing::info!( - servers = server_count, + servers = 0, launches = launch_registry .as_ref() .map_or(0, |registry| registry.launches.len()), diff --git a/app/crowdb-web/src/managed.rs b/app/crowdb-web/src/managed.rs index c13f5f04e..2f65be289 100644 --- a/app/crowdb-web/src/managed.rs +++ b/app/crowdb-web/src/managed.rs @@ -164,8 +164,8 @@ async fn load_snapshot(state: &AppState) -> Result +// Licensed under the Apache License, Version 2.0. + +//! Bare-metal hardware routes backed by confirmed Group 0 state. + +use axum::extract::{Path, Query, State}; +use axum::http::StatusCode; +use axum::Json; +use crowdb_console_shared::config::{DiskEntry, DiskGroupEntry, NodeEntry, RackEntry}; +use crowdb_console_shared::ops::hardware; +use serde::Deserialize; + +use crate::error::ErrorBody; +use crate::managed_logical::api_error; +use crate::state::AppState; +use crowdb_console_shared::error::Error as ConsoleError; + +type ApiError = (StatusCode, Json); + +#[derive(Deserialize)] +pub(crate) struct CreateRack { + id: u64, + #[serde(default)] + name: String, +} + +#[derive(Deserialize)] +pub(crate) struct NodeFilter { + rack_id: Option, +} + +#[derive(Deserialize)] +pub(crate) struct CreateDiskGroup { + id: u64, + #[serde(default)] + name: String, +} + +pub(crate) async fn list_racks(State(state): State) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_racks_from_group0(&ctx) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn get_rack( + State(state): State, + Path(id): Path, +) -> Result, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_racks_from_group0(&ctx) + .await + .map_err(api_error)? + .into_iter() + .find(|rack| rack.id == id) + .map(Json) + .ok_or_else(|| { + api_error(ConsoleError::NotFound { + kind: "rack".into(), + id: id.to_string(), + }) + }) +} + +pub(crate) async fn add_rack( + State(state): State, + Json(body): Json, +) -> Result<(StatusCode, Json), ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + let rack = hardware::add_rack_to_group0(&ctx, body.id, &body.name) + .await + .map_err(api_error)?; + Ok((StatusCode::CREATED, Json(rack))) +} + +pub(crate) async fn remove_rack( + State(state): State, + Path(id): Path, +) -> Result { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::remove_rack_from_group0(&ctx, id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} + +pub(crate) async fn list_nodes( + State(state): State, + Query(filter): Query, +) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_nodes_from_group0(&ctx, filter.rack_id) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn list_rack_nodes( + State(state): State, + Path(rack_id): Path, +) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_nodes_from_group0(&ctx, Some(rack_id)) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn get_node( + State(state): State, + Path(id): Path, +) -> Result, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_nodes_from_group0(&ctx, None) + .await + .map_err(api_error)? + .into_iter() + .find(|node| node.id == id) + .map(Json) + .ok_or_else(|| { + api_error(ConsoleError::NotFound { + kind: "node".into(), + id: id.to_string(), + }) + }) +} + +pub(crate) async fn add_node( + State(state): State, + Json(node): Json, +) -> Result<(StatusCode, Json), ApiError> { + if node.ssh_key.is_some() || node.ssh_password.is_some() { + return Err(( + StatusCode::BAD_REQUEST, + Json(ErrorBody { + error: "inline SSH secrets are not accepted; use ssh_credential_ref".into(), + }), + )); + } + let ctx = state.op_context().await.map_err(api_error)?; + let node = hardware::add_node_to_group0(&ctx, node) + .await + .map_err(api_error)?; + Ok((StatusCode::CREATED, Json(node))) +} + +pub(crate) async fn remove_node( + State(state): State, + Path(id): Path, +) -> Result { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::remove_node_from_group0(&ctx, id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} + +pub(crate) async fn list_disk_groups( + State(state): State, + Path(node_id): Path, +) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_disk_groups_from_group0(&ctx, node_id) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn get_disk_group( + State(state): State, + Path((node_id, dg_id)): Path<(u64, u64)>, +) -> Result, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_disk_groups_from_group0(&ctx, node_id) + .await + .map_err(api_error)? + .into_iter() + .find(|group| group.id == dg_id) + .map(Json) + .ok_or_else(|| { + api_error(ConsoleError::NotFound { + kind: "disk_group".into(), + id: dg_id.to_string(), + }) + }) +} + +pub(crate) async fn add_disk_group( + State(state): State, + Path(node_id): Path, + Json(body): Json, +) -> Result<(StatusCode, Json), ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + let group = hardware::add_disk_group_to_group0(&ctx, node_id, body.id, &body.name) + .await + .map_err(api_error)?; + Ok((StatusCode::CREATED, Json(group))) +} + +pub(crate) async fn remove_disk_group( + State(state): State, + Path((node_id, dg_id)): Path<(u64, u64)>, +) -> Result { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::remove_disk_group_from_group0(&ctx, node_id, dg_id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} + +pub(crate) async fn list_disks( + State(state): State, + Path((node_id, dg_id)): Path<(u64, u64)>, +) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_disks_from_group0(&ctx, node_id, dg_id) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn get_disk( + State(state): State, + Path((node_id, dg_id, disk_id)): Path<(u64, u64, String)>, +) -> Result, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_disks_from_group0(&ctx, node_id, dg_id) + .await + .map_err(api_error)? + .into_iter() + .find(|disk| disk.disk_id == disk_id) + .map(Json) + .ok_or_else(|| { + api_error(ConsoleError::NotFound { + kind: "disk".into(), + id: disk_id, + }) + }) +} + +pub(crate) async fn add_disk( + State(state): State, + Path((node_id, dg_id)): Path<(u64, u64)>, + Json(body): Json, +) -> Result<(StatusCode, Json), ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + let disk = hardware::add_disk_to_group0(&ctx, node_id, dg_id, &body) + .await + .map_err(api_error)?; + Ok((StatusCode::CREATED, Json(disk))) +} + +pub(crate) async fn remove_disk( + State(state): State, + Path((node_id, dg_id, disk_id)): Path<(u64, u64, String)>, +) -> Result { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::remove_disk_from_group0(&ctx, node_id, dg_id, &disk_id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} diff --git a/app/crowdb-web/src/managed_logical.rs b/app/crowdb-web/src/managed_logical.rs index 648d9bd37..f412dc7b6 100644 --- a/app/crowdb-web/src/managed_logical.rs +++ b/app/crowdb-web/src/managed_logical.rs @@ -13,7 +13,7 @@ use crate::state::AppState; type ApiError = (StatusCode, Json); #[allow(clippy::needless_pass_by_value)] -fn api_error(error: Error) -> ApiError { +pub(crate) fn api_error(error: Error) -> ApiError { let status = match error { Error::NotFound { .. } => StatusCode::NOT_FOUND, Error::Conflict { .. } => StatusCode::CONFLICT, diff --git a/app/crowdb-web/src/mgmt/cluster_init.rs b/app/crowdb-web/src/mgmt/cluster_init.rs index 79f8274b1..353815f7b 100644 --- a/app/crowdb-web/src/mgmt/cluster_init.rs +++ b/app/crowdb-web/src/mgmt/cluster_init.rs @@ -19,6 +19,9 @@ pub(crate) struct ClusterInitBody { /// Must be non-empty. For a single node, group 0 self-elects. /// For multiple nodes, remotes are wired and election starts after. pub nodes: Vec, + /// Optional versioned bootstrap topology file for a first bare-metal init. + #[serde(default)] + pub bootstrap_file: Option, } /// `POST /api/cluster/init` — initialize the cluster by bootstrapping @@ -38,9 +41,31 @@ pub(crate) async fn http_cluster_init( Json(body): Json, ) -> Result<(StatusCode, Json), (StatusCode, Json)> { let ctx = state.op_context().await.map_err(|e| err_502(format!("{e}")))?; - let summary = ops::cluster::init(&ctx, &body.nodes) - .await - .map_err(map_config_err)?; + let summary = if state.web_mode.is_some() { + let path = state.runtime_root.join("bootstrap-intent.toml"); + if state.web_mode == Some(crowdb_console_shared::config::web::WebMode::BareMetal) { + if let Some(source) = &body.bootstrap_file { + let intent = crowdb_console_shared::bootstrap_intent::BootstrapIntent::load(source) + .map_err(map_config_err)?; + if intent.members() != body.nodes.as_slice() { + return Err(map_config_err(crowdb_console_shared::error::Error::Validation { + field: "nodes".into(), + message: "bootstrap file members differ from requested nodes".into(), + })); + } + intent.seal(&path).map_err(map_config_err)?; + } else if !path.exists() { + return Err(map_config_err(crowdb_console_shared::error::Error::Validation { + field: "bootstrap_file".into(), + message: "required for the first bare-metal cluster init".into(), + })); + } + } + ops::cluster::init_with_intent(&ctx, &body.nodes, &path).await + } else { + ops::cluster::init(&ctx, &body.nodes).await + } + .map_err(map_config_err)?; state.commit_op_context(&ctx).map_err(map_persist_err)?; // The cluster is now live — re-seed the shared kv_client with the diff --git a/app/crowdb-web/src/mgmt/group_ops.rs b/app/crowdb-web/src/mgmt/group_ops.rs index ee73c4f44..7ac47607c 100644 --- a/app/crowdb-web/src/mgmt/group_ops.rs +++ b/app/crowdb-web/src/mgmt/group_ops.rs @@ -243,7 +243,9 @@ pub(crate) async fn http_remove_group( return Err(( StatusCode::CONFLICT, Json(ErrorBody { - error: "group 0 in store 0 is the system group; use POST /api/cluster/reset to tear down the entire cluster".into(), + error: + "group 0 in store 0 is the system group; destroy and recreate the cluster to remove it" + .into(), }), )); } diff --git a/app/crowdb-web/src/mgmt/store_ops.rs b/app/crowdb-web/src/mgmt/store_ops.rs index 804574d95..0188f72a5 100644 --- a/app/crowdb-web/src/mgmt/store_ops.rs +++ b/app/crowdb-web/src/mgmt/store_ops.rs @@ -202,9 +202,7 @@ pub(crate) async fn http_remove_store( return Err(( StatusCode::CONFLICT, Json(ErrorBody { - error: - "store 0 is the system store; use POST /api/cluster/reset to tear down the entire cluster" - .into(), + error: "store 0 is the system store; destroy and recreate the cluster to remove it".into(), }), )); } diff --git a/app/crowdb-web/src/physical/view.rs b/app/crowdb-web/src/physical/view.rs index b8e78ddc7..00dc9f7aa 100644 --- a/app/crowdb-web/src/physical/view.rs +++ b/app/crowdb-web/src/physical/view.rs @@ -371,6 +371,7 @@ mod tests { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); @@ -505,6 +506,7 @@ mod tests { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); let snap = BTreeMap::new(); diff --git a/app/crowdb-web/src/state.rs b/app/crowdb-web/src/state.rs index 16c98abd0..e4ff7a437 100644 --- a/app/crowdb-web/src/state.rs +++ b/app/crowdb-web/src/state.rs @@ -10,23 +10,18 @@ use crowdb_console_shared::error::{Error, Result}; use crowdb_console_shared::launch::LaunchRuntime; use crowdb_console_shared::monitor::MonitorCache; use crowdb_console_shared::ops::OpContext; -use crowdb_console_shared::{ - config::{ConsoleConfigEngine, ServerEntry, TomlFileEngine}, - ConsoleConfig, -}; +use crowdb_console_shared::{config::ServerEntry, ConsoleConfig}; /// Shared, mutable console state. /// -/// `config` carries the full `ConsoleConfig` (racks, nodes, servers) -/// behind a `RwLock`; mutations are persisted via `ConsoleConfig::save` -/// to `config_path` when present. +/// `config` is an in-memory context for bootstrap and test-only routes. +/// Production authority reads use Group 0 and live registrations. /// /// `diskdb_client` is lazily initialized on the first `/api/diskdb/*` /// request (the service registry may not be ready at console startup). #[derive(Clone)] pub struct AppState { pub config: Arc>, - pub config_engine: Option>, pub runtime_root: Arc, pub monitor_cache: Arc, pub runtime_pids: Arc>>, @@ -79,32 +74,24 @@ impl AppState { Self::with_config(cfg, None) } - /// Build state from an already-loaded `ConsoleConfig`. `path` is the - /// on-disk location used by mutating handlers to persist changes; - /// pass `None` for in-memory-only state (tests). + /// Build test state from an in-memory `ConsoleConfig`. `path` contributes + /// only the runtime workspace directory; it is never a topology file. #[must_use] pub fn with_config(config: ConsoleConfig, path: Option) -> Self { let runtime_root = path - .as_ref() .and_then(|path| path.parent().map(std::path::Path::to_path_buf)) .unwrap_or_else(|| { crowdb_protocol::port::namespace::runtime_root() .join("persistent") .join("console") }); - let engine = path.map(|path| Arc::new(TomlFileEngine::new(path)) as Arc); - Self::with_config_engine(config, engine, runtime_root) + Self::with_runtime_root(config, runtime_root) } #[must_use] - pub fn with_config_engine( - config: ConsoleConfig, - engine: Option>, - runtime_root: PathBuf, - ) -> Self { + pub fn with_runtime_root(config: ConsoleConfig, runtime_root: PathBuf) -> Self { Self { config: Arc::new(RwLock::new(config)), - config_engine: engine, runtime_root: Arc::new(runtime_root), monitor_cache: Arc::new(MonitorCache::new()), runtime_pids: Arc::new(std::sync::Mutex::new(HashMap::new())), @@ -189,19 +176,12 @@ impl AppState { self } - /// Persist the current config to `config_path`, if one was provided. - /// No-op for in-memory state. - /// - /// # Panics - /// Panics if the `RwLock` is poisoned. + /// The legacy in-memory handlers share state only within this Web process. + /// No topology is written to a local file. /// /// # Errors - /// Returns an error if config saving fails. + /// Reserved for callers that propagate operation errors. pub fn persist(&self) -> crowdb_console_shared::error::Result<()> { - if let Some(engine) = self.config_engine.as_ref() { - let cfg = self.config.read().unwrap(); - cfg.save_with_engine(engine.as_ref())?; - } Ok(()) } @@ -590,12 +570,10 @@ impl AppState { /// /// The write-back is a short critical section with no `await` /// inside the lock — the `OpContext`'s config is cloned in, the - /// old config is replaced, and the lock is released before - /// persistence (which may do file I/O). + /// old config is replaced, and the lock is released. /// /// # Errors - /// Returns an error if the config lock is poisoned or persistence - /// fails. + /// Returns an error if the config lock is poisoned. pub fn commit_op_context(&self, ctx: &OpContext) -> Result<()> { let new_config = ctx.config().clone(); { @@ -680,8 +658,7 @@ mod tests { let root = tempdir("relative-runtime-root"); std::env::set_current_dir(&root).unwrap(); - let state = - AppState::with_config_engine(ConsoleConfig::default(), None, PathBuf::from("example-runtime")); + let state = AppState::with_runtime_root(ConsoleConfig::default(), PathBuf::from("example-runtime")); let workspace = state.prepare_node_workspace("n1").unwrap(); assert!(workspace.is_absolute()); diff --git a/app/crowdb-web/tests/bare_metal_authority_test.rs b/app/crowdb-web/tests/bare_metal_authority_test.rs index e7f7796a6..8f1d7b71d 100644 --- a/app/crowdb-web/tests/bare_metal_authority_test.rs +++ b/app/crowdb-web/tests/bare_metal_authority_test.rs @@ -56,6 +56,7 @@ async fn initialized_authority(cluster: &KvCluster) -> CrowdbSysmdClient { &RackValue { status: HwStatus::Up as i32, node_ids: vec![1], + name: "rack-a".into(), }, ) .await @@ -66,6 +67,10 @@ async fn initialized_authority(cluster: &KvCluster) -> CrowdbSysmdClient { 1, &NodeValue { status: HwStatus::Up as i32, + management_host: "node-a.example".into(), + ssh_port: 2222, + ssh_user: "operator".into(), + ssh_credential_ref: Some("node-a-key".into()), ..Default::default() }, ) @@ -90,7 +95,291 @@ fn application(cluster: &KvCluster) -> axum::Router { log_max_files: 5, request_timeout_ms: Some(500), }; - router(AppState::default().with_process_config(&config)) + router( + AppState::default() + .with_process_config(&config) + .with_management_token("bare-metal-test-token-123456789012345".into()) + .unwrap(), + ) +} + +async fn hardware_request( + app: &axum::Router, + method: axum::http::Method, + path: &str, + body: Option, + authenticated: bool, +) -> (StatusCode, serde_json::Value) { + let mut request = Request::builder().method(method).uri(path); + if authenticated { + request = request.header("authorization", "Bearer bare-metal-test-token-123456789012345"); + } + let request = if let Some(body) = body { + request + .header("content-type", "application/json") + .body(Body::from(body.to_string())) + .unwrap() + } else { + request.body(Body::empty()).unwrap() + }; + let response = app.clone().oneshot(request).await.unwrap(); + let status = response.status(); + let bytes = axum::body::to_bytes(response.into_body(), 1024 * 1024) + .await + .unwrap(); + let value = if bytes.is_empty() { + serde_json::Value::Null + } else { + serde_json::from_slice(&bytes).unwrap() + }; + (status, value) +} + +#[tokio::test] +async fn bare_metal_hardware_routes_share_confirmed_group_zero_state() { + let cluster = KvCluster::start().await; + initialized_authority(&cluster).await; + let first = application(&cluster); + let second = application(&cluster); + let rack = serde_json::json!({"id": 8, "name": "rack-eight"}); + assert_eq!( + hardware_request( + &first, + axum::http::Method::POST, + "/api/racks", + Some(rack.clone()), + false + ) + .await + .0, + StatusCode::UNAUTHORIZED + ); + assert_eq!( + hardware_request(&first, axum::http::Method::POST, "/api/racks", Some(rack), true) + .await + .0, + StatusCode::CREATED + ); + let (_, racks) = hardware_request(&second, axum::http::Method::GET, "/api/racks", None, false).await; + assert!(racks + .as_array() + .unwrap() + .iter() + .any(|rack| rack["id"] == 8 && rack["name"] == "rack-eight")); + let node = serde_json::json!({"id": 9, "rack_id": 8, "host": "node-nine.example", "ssh_port": 2222, "ssh_user": "operator", "ssh_credential_ref": "ops-key"}); + let mut secret_node = node.clone(); + secret_node["ssh_key"] = serde_json::json!("/tmp/private-key"); + assert_eq!( + hardware_request( + &first, + axum::http::Method::POST, + "/api/nodes", + Some(secret_node), + true + ) + .await + .0, + StatusCode::BAD_REQUEST + ); + assert_eq!( + hardware_request(&first, axum::http::Method::POST, "/api/nodes", Some(node), true) + .await + .0, + StatusCode::CREATED + ); + let (_, nodes) = hardware_request( + &second, + axum::http::Method::GET, + "/api/nodes?rack_id=8", + None, + false, + ) + .await; + assert_eq!(nodes[0]["host"], "node-nine.example"); + assert_eq!(nodes[0]["ssh_credential_ref"], "ops-key"); + assert!(nodes[0].get("ssh_key").is_none()); + assert_eq!( + hardware_request(&first, axum::http::Method::GET, "/api/racks/8", None, false) + .await + .1["name"], + "rack-eight" + ); + assert_eq!( + hardware_request(&first, axum::http::Method::GET, "/api/racks/8/nodes", None, false) + .await + .1[0]["id"], + 9 + ); + assert_eq!( + hardware_request(&first, axum::http::Method::GET, "/api/nodes/9", None, false) + .await + .1["host"], + "node-nine.example" + ); + assert_eq!( + hardware_request(&second, axum::http::Method::DELETE, "/api/racks/8", None, true) + .await + .0, + StatusCode::CONFLICT + ); + assert_eq!( + hardware_request( + &second, + axum::http::Method::POST, + "/api/racks", + Some(serde_json::json!({"id": 8, "name": "changed"})), + true + ) + .await + .0, + StatusCode::CONFLICT + ); + verify_storage_hardware(&first, &second).await; + verify_hardware_deletion(&first, &second).await; +} + +async fn verify_storage_hardware(first: &axum::Router, second: &axum::Router) { + let groups = "/api/nodes/9/disk-groups"; + let group = serde_json::json!({"id": 4, "name": "hot"}); + assert_eq!( + hardware_request( + first, + axum::http::Method::POST, + groups, + Some(group.clone()), + false + ) + .await + .0, + StatusCode::UNAUTHORIZED + ); + assert_eq!( + hardware_request(first, axum::http::Method::POST, groups, Some(group), true) + .await + .0, + StatusCode::CREATED + ); + assert_eq!( + hardware_request(second, axum::http::Method::GET, groups, None, false) + .await + .1[0]["name"], + "hot" + ); + let disks = "/api/nodes/9/disk-groups/4/disks"; + assert_eq!( + hardware_request( + second, + axum::http::Method::GET, + "/api/nodes/9/disk-groups/4", + None, + false + ) + .await + .1["name"], + "hot" + ); + let disk = serde_json::json!({ + "disk_id": "0000000000000000-0000000000000009", "disk_type": "Ssd", + "capacity_bytes": 4096, "zone_size_bytes": 4096, "unit_size_bytes": 4096, + "device_path": "/dev/test", + }); + assert_eq!( + hardware_request(second, axum::http::Method::POST, disks, Some(disk), true) + .await + .0, + StatusCode::CREATED + ); + assert_eq!( + hardware_request(first, axum::http::Method::GET, disks, None, false) + .await + .1 + .as_array() + .unwrap() + .len(), + 1 + ); + assert_eq!( + hardware_request( + first, + axum::http::Method::GET, + "/api/nodes/9/disk-groups/4/disks/0000000000000000-0000000000000009", + None, + false + ) + .await + .1["device_path"], + "/dev/test" + ); + verify_storage_removal(first, second).await; +} + +async fn verify_storage_removal(first: &axum::Router, second: &axum::Router) { + assert_eq!( + hardware_request( + first, + axum::http::Method::DELETE, + "/api/nodes/9/disk-groups/4", + None, + true + ) + .await + .0, + StatusCode::CONFLICT + ); + assert_eq!( + hardware_request( + second, + axum::http::Method::DELETE, + "/api/nodes/9/disk-groups/4/disks/0000000000000000-0000000000000009", + None, + true + ) + .await + .0, + StatusCode::NO_CONTENT + ); + assert_eq!( + hardware_request( + first, + axum::http::Method::DELETE, + "/api/nodes/9/disk-groups/4", + None, + true + ) + .await + .0, + StatusCode::NO_CONTENT + ); +} + +async fn verify_hardware_deletion(first: &axum::Router, second: &axum::Router) { + assert_eq!( + hardware_request(second, axum::http::Method::DELETE, "/api/nodes/9", None, false) + .await + .0, + StatusCode::UNAUTHORIZED + ); + assert_eq!( + hardware_request(second, axum::http::Method::DELETE, "/api/nodes/9", None, true) + .await + .0, + StatusCode::NO_CONTENT + ); + let (_, nodes) = hardware_request( + first, + axum::http::Method::GET, + "/api/nodes?rack_id=8", + None, + false, + ) + .await; + assert!(nodes.as_array().unwrap().is_empty()); + assert_eq!( + hardware_request(first, axum::http::Method::DELETE, "/api/racks/8", None, true) + .await + .0, + StatusCode::NO_CONTENT + ); } async fn unavailable(app: &axum::Router) { @@ -110,6 +399,12 @@ async fn bare_metal_snapshot_requires_live_authority_without_a_docker_monitor() assert_eq!(code, StatusCode::OK, "{view}"); assert_eq!(view["source"], "group0"); assert_eq!(view["nodes"][0]["id"], 1); + assert_eq!(view["racks"][0]["name"], "rack-a"); + assert_eq!(view["nodes"][0]["management_host"], "node-a.example"); + assert_eq!(view["nodes"][0]["ssh_port"], 2222); + assert_eq!(view["nodes"][0]["ssh_user"], "operator"); + assert_eq!(view["nodes"][0]["ssh_credential_ref"], "node-a-key"); + assert_eq!(snapshot(&application(&cluster)).await.1["nodes"], view["nodes"]); assert!(view["monitor"].is_null()); register(&sysmd, 1, 7002, &cluster.mgmt_endpoints[0]).await; @@ -143,6 +438,12 @@ async fn bare_metal_snapshot_requires_live_authority_without_a_docker_monitor() drop(cluster); unavailable(&app).await; + assert_eq!( + hardware_request(&app, axum::http::Method::GET, "/api/racks", None, false) + .await + .0, + StatusCode::SERVICE_UNAVAILABLE + ); } #[tokio::test] diff --git a/app/crowdb-web/tests/cluster_deployer_test.rs b/app/crowdb-web/tests/cluster_deployer_test.rs index 09cf806ff..bcec457ab 100644 --- a/app/crowdb-web/tests/cluster_deployer_test.rs +++ b/app/crowdb-web/tests/cluster_deployer_test.rs @@ -26,7 +26,7 @@ fn tempdir(tag: &str) -> PathBuf { async fn spawn_web(cfg_path: PathBuf) -> String { let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.expect("bind"); let addr = listener.local_addr().expect("local_addr"); - let cfg = ConsoleConfig::load(&cfg_path).unwrap_or_default(); + let cfg = ConsoleConfig::default(); let state = AppState::with_config(cfg, Some(cfg_path)); tokio::spawn(async move { let _ = axum::serve(listener, router(state)).await; diff --git a/app/crowdb-web/tests/cluster_restart_incremental_test.rs b/app/crowdb-web/tests/cluster_restart_incremental_test.rs index e16897b5b..8a1f11816 100644 --- a/app/crowdb-web/tests/cluster_restart_incremental_test.rs +++ b/app/crowdb-web/tests/cluster_restart_incremental_test.rs @@ -97,7 +97,7 @@ async fn spawn_web_with_path(path: PathBuf) -> SocketAddr { .await .expect("bind"); let addr = listener.local_addr().expect("local_addr"); - let cfg = ConsoleConfig::load(&path).unwrap_or_default(); + let cfg = ConsoleConfig::default(); let state = AppState::with_config(cfg, Some(path.clone())); tokio::spawn(async move { axum::serve(listener, router(state)).await.unwrap(); diff --git a/app/crowdb-web/tests/diskdb_auto_start_test.rs b/app/crowdb-web/tests/diskdb_auto_start_test.rs index 3d7db0c9f..c128066e7 100644 --- a/app/crowdb-web/tests/diskdb_auto_start_test.rs +++ b/app/crowdb-web/tests/diskdb_auto_start_test.rs @@ -23,6 +23,7 @@ async fn startup_does_not_replay_local_diskdb_launch_policy_without_group0() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); config diff --git a/app/crowdb-web/tests/diskdb_routes_test.rs b/app/crowdb-web/tests/diskdb_routes_test.rs index da495b8f6..7cad78b0c 100644 --- a/app/crowdb-web/tests/diskdb_routes_test.rs +++ b/app/crowdb-web/tests/diskdb_routes_test.rs @@ -38,6 +38,7 @@ async fn spawn_web_with_disk_group(rack_id: u64, node_id: u64, dg_id: u64) -> So ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); config diff --git a/app/crowdb-web/tests/kv_routes_test.rs b/app/crowdb-web/tests/kv_routes_test.rs index 3fb2c45a1..5a100954d 100644 --- a/app/crowdb-web/tests/kv_routes_test.rs +++ b/app/crowdb-web/tests/kv_routes_test.rs @@ -48,6 +48,7 @@ async fn spawn_upstream() -> Option { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "1".to_string(), @@ -83,6 +84,7 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: "n1".to_string(), @@ -226,7 +228,6 @@ async fn kv_put_get_delete_through_web_routes() { async fn kv_get_returns_502_when_leader_unreachable() { use crowdb_console_shared::cluster::{LocalReplicaInfo, NodeGroup, NodeStore, ReplicaRole, ReplicaState}; use std::collections::BTreeMap; - // Pick a free port, drop the listener: nothing accepts on it now. let dead = std::net::TcpListener::bind(("127.0.0.1", 0)).unwrap(); let dead_port = dead.local_addr().unwrap().port(); @@ -250,6 +251,7 @@ async fn kv_get_returns_502_when_leader_unreachable() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); // The node has a configured rpc_url, but the port is dead. cfg.add_server(ServerEntry { @@ -320,14 +322,12 @@ async fn kv_get_returns_502_when_leader_unreachable() { tokio::time::sleep(Duration::from_millis(50)).await; let http = reqwest::Client::new(); - let url = format!("http://{web}/api/stores/7/groups/70/kv/get?key=anything"); - let resp = http.get(&url).send().await.unwrap(); - assert_eq!( - resp.status(), - 502, - "expected 502 when leader crowdb-rpc port is dead, got {}", - resp.status() - ); + let resp = http + .get(format!("http://{web}/api/stores/7/groups/70/kv/get?key=anything")) + .send() + .await + .unwrap(); + assert_eq!(resp.status(), 502); let endpoint = http .get(format!("http://{web}/api/stores/7/groups/70/endpoint")) .send() diff --git a/app/crowdb-web/tests/lifecycle_routes_test.rs b/app/crowdb-web/tests/lifecycle_routes_test.rs index 60aa7266c..2e728c5ee 100644 --- a/app/crowdb-web/tests/lifecycle_routes_test.rs +++ b/app/crowdb-web/tests/lifecycle_routes_test.rs @@ -37,7 +37,7 @@ async fn spawn_web_with_path(path: std::path::PathBuf) -> SocketAddr { .await .expect("bind"); let addr = listener.local_addr().expect("local_addr"); - let cfg = ConsoleConfig::load(&path).unwrap_or_default(); + let cfg = ConsoleConfig::default(); let state = AppState::with_config(cfg, Some(path)).with_test_mode(true); tokio::spawn(async move { axum::serve(listener, router(state)).await.unwrap(); @@ -220,13 +220,7 @@ async fn rack_node_crud_through_web_routes() { let s = delete_status(&client, &format!("{base}/api/racks/1")).await; assert_eq!(s.as_u16(), 204); - // Persisted file reflects the empty state. - let on_disk = std::fs::read_to_string(&cfg_path).unwrap_or_default(); - assert!( - !on_disk.contains("[[rack]]"), - "rack should be gone from {cfg_path:?}: {on_disk}" - ); - assert!(!on_disk.contains("[[node]]")); + assert!(!cfg_path.exists(), "Web must not write local topology"); } #[tokio::test] diff --git a/app/crowdb-web/tests/managed_mode_test.rs b/app/crowdb-web/tests/managed_mode_test.rs index a70b7a68a..3395f0452 100644 --- a/app/crowdb-web/tests/managed_mode_test.rs +++ b/app/crowdb-web/tests/managed_mode_test.rs @@ -239,6 +239,7 @@ async fn managed_snapshot_uses_group0_and_monitor_without_local_fallback() { &RackValue { status: HwStatus::Up as i32, node_ids: vec![1], + ..Default::default() }, ) .await @@ -369,7 +370,16 @@ fn old_mixed_config_is_rejected_before_startup() { } #[test] -fn malformed_bare_metal_registry_fails_before_listener_bind() { +fn production_web_requires_versioned_process_config() { + let output = std::process::Command::new(env!("CARGO_BIN_EXE_crowdb-web")) + .output() + .unwrap(); + assert!(!output.status.success()); + assert!(String::from_utf8_lossy(&output.stderr).contains("requires a versioned --config")); +} + +#[test] +fn legacy_mixed_registry_is_not_loaded_without_versioned_config() { let root = std::env::temp_dir().join(format!( "crowdb-web-malformed-registry-{}-{}", std::process::id(), @@ -389,10 +399,7 @@ fn malformed_bare_metal_registry_fails_before_listener_bind() { std::fs::remove_dir_all(root).unwrap(); assert!(!output.status.success()); let stderr = String::from_utf8_lossy(&output.stderr); - assert!( - stderr.contains("Error: Config(") && stderr.contains("invalid table header"), - "{stderr}" - ); + assert!(stderr.contains("requires a versioned --config"), "{stderr}"); } #[test] diff --git a/app/crowdb-web/tests/metrics_proxy_test.rs b/app/crowdb-web/tests/metrics_proxy_test.rs index caec2564c..41eb3b8f8 100644 --- a/app/crowdb-web/tests/metrics_proxy_test.rs +++ b/app/crowdb-web/tests/metrics_proxy_test.rs @@ -50,6 +50,7 @@ async fn spawn_upstream() -> Option { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "1".into(), @@ -85,6 +86,7 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: "n1".into(), diff --git a/app/crowdb-web/tests/mgmt_routes_test.rs b/app/crowdb-web/tests/mgmt_routes_test.rs index f0b74a626..483d71c49 100644 --- a/app/crowdb-web/tests/mgmt_routes_test.rs +++ b/app/crowdb-web/tests/mgmt_routes_test.rs @@ -11,7 +11,9 @@ use std::net::SocketAddr; use std::path::PathBuf; use std::time::Duration; +use crowdb_console_shared::bootstrap_intent::BootstrapIntent; use crowdb_console_shared::cluster::{NodeHealth, NodeStore}; +use crowdb_console_shared::config::web::{WebMode, WebProcessConfig}; use crowdb_console_shared::config::{NodeEntry, RackEntry, ServerEntry, ServiceType}; use crowdb_console_shared::lifecycle::{self, crowdb_kv_server_bin, stop_pid_with_timeout, DeployRequest}; use crowdb_console_shared::monitor::NodeRecord; @@ -62,6 +64,7 @@ async fn spawn_upstream() -> Option { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "1".to_string(), @@ -83,10 +86,30 @@ async fn spawn_upstream() -> Option { } async fn spawn_web(upstream: &Upstream) -> SocketAddr { + spawn_web_with_config_path(upstream, None).await +} + +async fn spawn_web_with_config_path( + upstream: &Upstream, + config_path: Option, +) -> SocketAddr { let listener = tokio::net::TcpListener::bind(SocketAddr::from(([127, 0, 0, 1], 0))) .await .expect("bind"); let addr = listener.local_addr().expect("local_addr"); + let cfg = config_for_upstream(upstream); + let state = AppState::with_config(cfg, config_path); + // Register the upstream's pid so `refresh_node_cache` (which skips + // nodes with no tracked runtime pid) refreshes after mutations. + state.set_runtime_pid(1, upstream.pid); + tokio::spawn(async move { + axum::serve(listener, router(state)).await.unwrap(); + }); + tokio::time::sleep(Duration::from_millis(50)).await; + addr +} + +fn config_for_upstream(upstream: &Upstream) -> ConsoleConfig { let mut cfg = ConsoleConfig::default(); cfg.racks.push(RackEntry { id: 1, @@ -100,6 +123,7 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: "n1".to_string(), @@ -117,15 +141,73 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { no_fsync: false, }) .unwrap(); - let state = AppState::with_config(cfg, None); - // Register the upstream's pid so `refresh_node_cache` (which skips - // nodes with no tracked runtime pid) refreshes after mutations. - state.set_runtime_pid(1, upstream.pid); + cfg +} + +#[tokio::test] +async fn legacy_in_memory_web_bootstrap_does_not_persist_topology() { + let Some(upstream) = spawn_upstream().await else { + eprintln!("skipping: crowdb-kv-server binary not built"); + return; + }; + let config_path = upstream.workspace.join("console.toml"); + let web = spawn_web_with_config_path(&upstream, Some(config_path.clone())).await; + let response = reqwest::Client::new() + .post(format!("http://{web}/api/cluster/init")) + .json(&json!({"nodes": [1]})) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 201, "{}", response.text().await.unwrap()); + assert!(!config_path.exists()); + assert!(!upstream.workspace.join("bootstrap-intent.toml").exists()); +} + +#[tokio::test] +async fn bare_metal_web_bootstrap_consumes_sealed_topology_input() { + let Some(upstream) = spawn_upstream().await else { + eprintln!("skipping: crowdb-kv-server binary not built"); + return; + }; + let source = upstream.workspace.join("bootstrap-source.toml"); + BootstrapIntent::capture(&config_for_upstream(&upstream), &[1]) + .unwrap() + .seal(&source) + .unwrap(); + let config = WebProcessConfig { + version: 1, + mode: WebMode::BareMetal, + bind: "127.0.0.1".into(), + port: 14000, + group0_management_seeds: vec![upstream.mgmt_url.clone()], + ui_root: "/tmp".into(), + monitor_status: None, + log_dir: "/tmp".into(), + log_max_file_mb: 30, + log_max_files: 5, + request_timeout_ms: Some(500), + }; + let state = AppState::with_runtime_root(ConsoleConfig::default(), upstream.workspace.clone()) + .with_process_config(&config) + .with_management_token("bare-metal-bootstrap-test-token-12345".into()) + .unwrap(); + let listener = tokio::net::TcpListener::bind(SocketAddr::from(([127, 0, 0, 1], 0))) + .await + .unwrap(); + let addr = listener.local_addr().unwrap(); tokio::spawn(async move { axum::serve(listener, router(state)).await.unwrap(); }); - tokio::time::sleep(Duration::from_millis(50)).await; - addr + let response = reqwest::Client::new() + .post(format!("http://{addr}/api/cluster/init")) + .bearer_auth("bare-metal-bootstrap-test-token-12345") + .json(&json!({"nodes": [1], "bootstrap_file": source})) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 201, "{}", response.text().await.unwrap()); + assert!(!upstream.workspace.join("bootstrap-intent.toml").exists()); + assert!(!upstream.workspace.join("console.toml").exists()); } #[tokio::test] diff --git a/app/crowdb-web/tests/ops_migration_test.rs b/app/crowdb-web/tests/ops_migration_test.rs index 996d05673..fecc7faaf 100644 --- a/app/crowdb-web/tests/ops_migration_test.rs +++ b/app/crowdb-web/tests/ops_migration_test.rs @@ -52,6 +52,7 @@ async fn spawn_upstream() -> Option { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "1".to_string(), @@ -89,6 +90,7 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: "n1".to_string(), diff --git a/app/crowdb-web/tests/recursive_query_test.rs b/app/crowdb-web/tests/recursive_query_test.rs index 81d0150fa..cd0b23cc9 100644 --- a/app/crowdb-web/tests/recursive_query_test.rs +++ b/app/crowdb-web/tests/recursive_query_test.rs @@ -142,6 +142,7 @@ async fn spawn_web_with_seeded_physical_tree() -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); let state = AppState::with_config(cfg, None); diff --git a/app/crowdb-web/tests/replica_leader_removal_test.rs b/app/crowdb-web/tests/replica_leader_removal_test.rs index 666e788f1..e584c0c07 100644 --- a/app/crowdb-web/tests/replica_leader_removal_test.rs +++ b/app/crowdb-web/tests/replica_leader_removal_test.rs @@ -95,6 +95,7 @@ async fn spawn_upstream(node_id: u64, workspace: &std::path::Path) -> Option) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: u.node_id.to_string(), diff --git a/app/crowdb-web/tests/replica_routes_test.rs b/app/crowdb-web/tests/replica_routes_test.rs index 070077f86..55d1cc072 100644 --- a/app/crowdb-web/tests/replica_routes_test.rs +++ b/app/crowdb-web/tests/replica_routes_test.rs @@ -82,6 +82,7 @@ async fn spawn_upstream(node_id: u64, workspace: &std::path::Path) -> Option SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: u.node_id.to_string(), diff --git a/app/crowdb-web/tests/rolling_upgrade_test.rs b/app/crowdb-web/tests/rolling_upgrade_test.rs deleted file mode 100644 index 0c8b75285..000000000 --- a/app/crowdb-web/tests/rolling_upgrade_test.rs +++ /dev/null @@ -1,493 +0,0 @@ -// Copyright 2026-present Gian -// Licensed under the Apache License, Version 2.0. - -//! M5 rolling-upgrade version-compat test: exercise a 3-node cluster with -//! two different `crowdb-kv-server` binary builds and verify a KV workload -//! does not diverge. -//! -//! By default the test uses the current `crowdb-kv-server` binary for all -//! nodes, copying it to a distinct path so the harness treats the second -//! node as a separate "version" build. To test a real version boundary, -//! set `CROWDB_KV_SERVER_BIN_V2` to a different binary (e.g. an older build). -//! -//! Uses the same real-subprocess harness as `replica_leader_removal_test.rs`. - -use std::collections::BTreeMap; -use std::net::SocketAddr; -use std::path::{Path, PathBuf}; -use std::time::{Duration, Instant}; - -use crowdb_console_shared::clients::http::ServerClient; -use crowdb_console_shared::cluster::NodeHealth; -use crowdb_console_shared::config::{NodeEntry, RackEntry, ServerEntry, ServiceType}; -use crowdb_console_shared::lifecycle::{self, crowdb_kv_server_bin, process_is_alive, DeployRequest}; -use crowdb_console_shared::monitor::{legacy_topology_to_node_stores, NodeRecord}; -use crowdb_console_shared::ConsoleConfig; -use crowdb_web::{router, AppState}; -use serde_json::json; - -fn pick_free_port() -> u16 { - crowdb_protocol::port::alloc::alloc_test_port(crowdb_protocol::ServicePort::Web) -} - -struct Upstream { - node_id: u64, - pid: u32, - mgmt_url: String, - rpc_url: String, - rest_port: u16, - rpc_port: u16, - binary: PathBuf, -} - -struct Cluster { - nodes: BTreeMap, - web: SocketAddr, - workspace: PathBuf, -} - -impl Cluster { - const fn sid() -> u64 { - 3 - } - - const fn gid() -> u64 { - 3 - } - - fn base_url(&self) -> String { - format!("http://{}", self.web) - } - - fn stop(&mut self) { - for n in self.nodes.values() { - let _ = lifecycle::stop_pid_with_timeout(n.pid, Duration::from_secs(5)); - } - } - - /// Restart a previously-killed node with the same ports, binary, and - /// workspace directory so it recovers from WAL and rejoins the group. - async fn restart_node(&mut self, node_id: u64) { - let u = self.nodes.get(&node_id).expect("node exists"); - let node = NodeEntry { - id: node_id, - rack_id: 1, - host: "127.0.0.1".into(), - ssh_port: 22, - ssh_user: String::new(), - ssh_key: None, - ssh_password: None, - }; - let replica_id = node_id; - let extra_args = vec![ - "--stores".to_string(), - Cluster::sid().to_string(), - "--groups".to_string(), - Cluster::gid().to_string(), - "--replica".to_string(), - replica_id.to_string(), - ]; - let req = DeployRequest { - server_id: node_id.to_string(), - rest_port: u.rest_port, - rpc_port: u.rpc_port, - election_profile: Some("e2e".into()), - binary: Some(u.binary.clone()), - ..Default::default() - }; - let node_dir = self.workspace.join(node_id.to_string()); - let deployed = lifecycle::deploy_local_in_dir_with_extra_args(&req, &node, &node_dir, &extra_args) - .await - .expect("restart node"); - // Update the stored pid so stop() cleans up the new process. - self.nodes.get_mut(&node_id).unwrap().pid = deployed.pid; - } -} - -impl Drop for Cluster { - fn drop(&mut self) { - for n in self.nodes.values() { - let _ = lifecycle::stop_pid_with_timeout(n.pid, Duration::from_secs(5)); - } - } -} - -fn tempdir(tag: &str) -> PathBuf { - let base = crowdb_test_harness::test_dirs::ephemeral_root().join("web-e2e"); - let millis = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_millis(); - let dir = base.join(format!("{tag}-{millis}")); - std::fs::create_dir_all(&dir).unwrap(); - dir -} - -fn resolve_binary() -> Option { - let bin = crowdb_kv_server_bin()?; - if !bin.exists() { - return None; - } - Some(bin) -} - -fn resolve_second_binary() -> Option { - if let Ok(v2) = std::env::var("CROWDB_KV_SERVER_BIN_V2") { - let p = PathBuf::from(v2); - if p.exists() { - return Some(p); - } - } - None -} - -async fn spawn_upstream(node_id: u64, workspace: &std::path::Path, binary: &Path) -> Option { - let node = NodeEntry { - id: node_id, - rack_id: 1, - host: "127.0.0.1".into(), - ssh_port: 22, - ssh_user: String::new(), - ssh_key: None, - ssh_password: None, - }; - let req = DeployRequest { - server_id: node_id.to_string(), - rest_port: pick_free_port(), - rpc_port: pick_free_port(), - election_profile: Some("e2e".into()), - binary: Some(binary.to_path_buf()), - ..Default::default() - }; - let node_dir = workspace.join(node_id.to_string()); - std::fs::create_dir_all(node_dir.join("bin")).unwrap(); - std::fs::create_dir_all(node_dir.join("log")).unwrap(); - let deployed = lifecycle::deploy_local_in_dir(&req, &node, &node_dir) - .await - .expect("deploy_local_in_dir"); - Some(Upstream { - node_id, - pid: deployed.pid, - mgmt_url: deployed.mgmt_url, - rpc_url: deployed.rpc_url, - rest_port: req.rest_port, - rpc_port: req.rpc_port, - binary: binary.to_path_buf(), - }) -} - -async fn spawn_web(upstreams: &BTreeMap) -> SocketAddr { - let listener = tokio::net::TcpListener::bind(SocketAddr::from(([127, 0, 0, 1], 0))) - .await - .unwrap(); - let addr = listener.local_addr().unwrap(); - let mut cfg = ConsoleConfig::default(); - cfg.racks.push(RackEntry { - id: 1, - name: "r1".into(), - }); - for u in upstreams.values() { - cfg.nodes.push(NodeEntry { - id: u.node_id, - rack_id: 1, - host: "127.0.0.1".into(), - ssh_port: 22, - ssh_user: String::new(), - ssh_key: None, - ssh_password: None, - }); - cfg.add_server(ServerEntry { - id: u.node_id.to_string(), - url: u.mgmt_url.clone(), - node_id: Some(u.node_id), - rpc_url: Some(u.rpc_url.clone()), - rest_port: None, - rpc_port: None, - auto_start: true, - binary: None, - election_profile: None, - pid: Some(u.pid), - service_type: ServiceType::Kv, - rpc_workers: None, - no_fsync: false, - }) - .unwrap(); - } - let state = AppState::with_config(cfg, None); - // Register each upstream's pid so `refresh_node_cache` (which - // skips nodes with no tracked runtime pid) refreshes after - // mutations. - for u in upstreams.values() { - state.set_runtime_pid(u.node_id, u.pid); - } - - for u in upstreams.values() { - let client = ServerClient::new(u.mgmt_url.clone()).unwrap(); - if let Ok(stores) = client.topology().await { - let rec = NodeRecord { - health: NodeHealth::Up, - last_seen_ms: 1, - stores: legacy_topology_to_node_stores(u.node_id, &stores), - last_error: None, - recovering: false, - }; - state.monitor_cache.set_node_report(u.node_id, rec).await; - } - } - - tokio::spawn(async move { - axum::serve(listener, router(state)).await.unwrap(); - }); - tokio::time::sleep(Duration::from_millis(50)).await; - addr -} - -async fn spawn_mixed_cluster(workspace: &std::path::Path, v2_binary: &PathBuf) -> Option { - let current = resolve_binary()?; - let mut nodes: BTreeMap = BTreeMap::new(); - for (id, binary) in [(1u64, ¤t), (2u64, ¤t), (3u64, v2_binary)] { - let Some(node) = spawn_upstream(id, workspace, binary).await else { - for n in nodes.into_values() { - let _ = lifecycle::stop_pid_with_timeout(n.pid, Duration::from_secs(5)); - } - eprintln!("skipping: crowdb-kv-server binary not built"); - return None; - }; - nodes.insert(id, node); - } - let web = spawn_web(&nodes).await; - Some(Cluster { - nodes, - web, - workspace: workspace.to_path_buf(), - }) -} - -async fn create_three_node_group(cluster: &Cluster) { - let base = cluster.base_url(); - let http = reqwest::Client::new(); - let sid = Cluster::sid(); - let gid = Cluster::gid(); - - // Initialize the system group so non-zero stores can be created. - let resp = http - .post(format!("{base}/api/cluster/init")) - .json(&json!({"nodes": [1, 2, 3]})) - .send() - .await - .unwrap(); - assert_eq!(resp.status(), 201, "cluster init: {:?}", resp.text().await.ok()); - - let resp = http - .post(format!("{base}/api/stores")) - .json(&json!({"store_id": sid, "nodes": [1]})) - .send() - .await - .unwrap(); - assert_eq!(resp.status(), 201, "create store: {:?}", resp.text().await.ok()); - - let resp = http - .post(format!("{base}/api/stores/{sid}/groups")) - .json(&json!({"group_id": gid, "replica_id": 1, "nodes": [1]})) - .send() - .await - .unwrap(); - assert_eq!(resp.status(), 201, "create group: {:?}", resp.text().await.ok()); - - for node_id in 2u64..=3 { - let resp = http - .post(format!("{base}/api/stores/{sid}/groups/{gid}/replicas")) - .json(&json!({"node_id": node_id, "replica_id": node_id})) - .send() - .await - .unwrap(); - assert_eq!( - resp.status(), - 201, - "add replica {node_id}: {:?}", - resp.text().await.ok() - ); - } -} - -async fn wait_for_leader( - cluster: &Cluster, - timeout: Duration, - exclude_node: Option, -) -> Option<(u64, u64)> { - let base = cluster.base_url(); - let http = reqwest::Client::new(); - let sid = Cluster::sid(); - let gid = Cluster::gid(); - let deadline = Instant::now() + timeout; - while Instant::now() < deadline { - let group: serde_json::Value = http - .get(format!("{base}/api/stores/{sid}/groups/{gid}")) - .send() - .await - .ok()? - .json() - .await - .ok()?; - if let Some(leader) = group["replicas"].as_array().unwrap_or(&vec![]).iter().find(|r| { - r["role"] == "leader" && exclude_node.map_or(true, |ex| r["node_id"].as_u64() != Some(ex)) - }) { - let rid = leader["replica_id"].as_u64()?; - let node_id = leader["node_id"].as_u64()?; - return Some((rid, node_id)); - } - tokio::time::sleep(Duration::from_millis(100)).await; - } - None -} - -async fn put_and_get(base: &str, http: &reqwest::Client, sid: u64, gid: u64, key: &str, value: &str) { - let put_resp = http - .post(format!("{base}/api/stores/{sid}/groups/{gid}/kv/put")) - .json(&json!({"key": key, "value": value})) - .send() - .await - .unwrap(); - assert_eq!( - put_resp.status(), - 200, - "kv put {key}: {:?}", - put_resp.text().await.ok() - ); - let put_body: serde_json::Value = put_resp.json().await.unwrap(); - assert_eq!(put_body["ok"], true, "kv put {key}: {put_body}"); - - let get_resp = http - .get(format!("{base}/api/stores/{sid}/groups/{gid}/kv/get?key={key}")) - .send() - .await - .unwrap(); - assert_eq!( - get_resp.status(), - 200, - "kv get {key}: {:?}", - get_resp.text().await.ok() - ); - let get_body: serde_json::Value = get_resp.json().await.unwrap(); - assert_eq!(get_body["value_utf8"], value, "kv get {key}: {get_body}"); -} - -#[tokio::test] -#[allow(clippy::too_many_lines)] -async fn mixed_version_3_node_cluster_kv_no_divergence() { - let Some(current) = resolve_binary() else { - eprintln!("skipping: crowdb-kv-server binary not built"); - return; - }; - - // If a second binary is not explicitly provided, make a copy of the - // current binary so the test still validates the mixed-build harness. - // For a real version boundary, set CROWDB_KV_SERVER_BIN_V2 to an old build. - let v2_binary = resolve_second_binary().unwrap_or_else(|| { - let copy = current.parent().unwrap().join("crowdb-kv-server-v2"); - let _ = std::fs::remove_file(©); - std::fs::copy(¤t, ©).expect("copy current binary as v2"); - copy - }); - - let workspace = tempdir("rolling_upgrade"); - let Some(mut cluster) = spawn_mixed_cluster(&workspace, &v2_binary).await else { - return; - }; - - create_three_node_group(&cluster).await; - - let _leader = wait_for_leader(&cluster, Duration::from_secs(10), None) - .await - .expect("leader should be elected in 3-node group"); - - let base = cluster.base_url(); - let http = reqwest::Client::new(); - let sid = Cluster::sid(); - let gid = Cluster::gid(); - - // Serve a small KV workload and verify each key round-trips. - for i in 0..20 { - let key = format!("rk-{i:02}"); - let value = format!("rv-{i:02}"); - put_and_get(&base, &http, sid, gid, &key, &value).await; - } - - // Restart each node one at a time (rolling upgrade) and verify the - // workload continues after each restart. - for node_id in 1u64..=3 { - let node = &cluster.nodes[&node_id]; - let pid = node.pid; - let status = std::process::Command::new("kill") - .arg("-TERM") - .arg(pid.to_string()) - .status() - .expect("terminate node"); - assert!(status.success()); - - // Wait for graceful shutdown to complete. The server's per-layer - // shutdown timeout is 10s (ServerConfig::DEFAULT.shutdown_timeout_ms) - // and shutdown performs a real fsync of the WAL + engine snapshot, - // which can take several seconds under CI disk contention. The poll - // returns as soon as the process exits (normally ~200ms); the 15s - // cap is a safety net above the server's own 10s/layer budget. - let dead = Instant::now() + Duration::from_secs(15); - while process_is_alive(pid) && Instant::now() < dead { - tokio::time::sleep(Duration::from_millis(50)).await; - } - assert!(!process_is_alive(pid), "node {node_id} should be stopped"); - - // Wait for a new leader to be elected before attempting reads. - let _new_leader = wait_for_leader(&cluster, Duration::from_secs(10), Some(node_id)) - .await - .expect("survivors should elect a new leader after node stop"); - - // The remaining nodes should still serve reads. - for i in 0..5 { - let key = format!("rk-{i:02}"); - let deadline = Instant::now() + Duration::from_secs(5); - let get_body = loop { - let get_resp = http - .get(format!("{base}/api/stores/{sid}/groups/{gid}/kv/get?key={key}")) - .send() - .await - .unwrap(); - if get_resp.status().is_success() { - break get_resp.json::().await.unwrap(); - } - if Instant::now() >= deadline { - let status = get_resp.status(); - let body = get_resp.text().await.unwrap_or_default(); - panic!("read during rolling restart failed for {key}: status {status}, body: {body}"); - } - tokio::time::sleep(Duration::from_millis(100)).await; - }; - let expected = format!("rv-{i:02}"); - assert_eq!( - get_body["value_utf8"], expected, - "value diverged for {key}: {get_body}" - ); - } - - // Restart the killed node so it recovers from WAL and rejoins the - // group, restoring full quorum before the next rolling step. - cluster.restart_node(node_id).await; - - // Wait for the restarted node to come back up. - let ready = Instant::now() + Duration::from_secs(5); - while Instant::now() < ready { - let client = ServerClient::new(cluster.nodes[&node_id].mgmt_url.clone()).unwrap(); - if client.health().await.is_ok() { - break; - } - tokio::time::sleep(Duration::from_millis(100)).await; - } - - // Wait for leader to stabilize after the node rejoins. - let _leader_after_restart = wait_for_leader(&cluster, Duration::from_secs(10), None) - .await - .expect("leader should stabilize after node restart"); - } - - cluster.stop(); -} diff --git a/container/crowdb-monitor/src/bootstrap/hardware.rs b/container/crowdb-monitor/src/bootstrap/hardware.rs index 81535fdb3..77f7a3d28 100644 --- a/container/crowdb-monitor/src/bootstrap/hardware.rs +++ b/container/crowdb-monitor/src/bootstrap/hardware.rs @@ -403,6 +403,7 @@ fn expected(profile: &DeploymentProfile) -> Result Result +// Licensed under the Apache License, Version 2.0. + +//! Private retention for file-based Linux core dumps. + +use std::ffi::OsStr; +use std::fs; +use std::io; +use std::os::unix::fs::DirBuilderExt; +use std::os::unix::fs::PermissionsExt; +use std::path::{Path, PathBuf}; +use std::time::SystemTime; + +pub struct CrashRetention { + root: PathBuf, +} + +impl CrashRetention { + /// Opens a private core directory and removes all but its newest core. + /// + /// # Errors + /// Rejects symlinked roots and inaccessible directories or entries. + pub fn open(root: PathBuf) -> io::Result { + match fs::DirBuilder::new().mode(0o700).create(&root) { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {} + Err(error) => return Err(error), + } + if !fs::symlink_metadata(&root)?.file_type().is_dir() { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "core root is not a directory", + )); + } + fs::set_permissions(&root, fs::Permissions::from_mode(0o700))?; + let retention = Self { root }; + retention.prune()?; + Ok(retention) + } + + #[must_use] + pub fn root(&self) -> &Path { + &self.root + } + + /// Keeps the newest regular `core` or `core.*` file only. + /// + /// # Errors + /// Returns a directory or removal error without exposing dump contents. + pub fn prune(&self) -> io::Result<()> { + let mut cores = Vec::new(); + for entry in fs::read_dir(&self.root)? { + let entry = entry?; + if !core_name(&entry.file_name()) || !entry.file_type()?.is_file() { + continue; + } + let modified = entry.metadata()?.modified().unwrap_or(SystemTime::UNIX_EPOCH); + cores.push((modified, entry.file_name(), entry.path())); + } + cores.sort_by(|left, right| right.0.cmp(&left.0).then_with(|| right.1.cmp(&left.1))); + for (_, _, path) in cores.into_iter().skip(1) { + fs::remove_file(path)?; + } + Ok(()) + } +} + +fn core_name(name: &OsStr) -> bool { + name == "core" || name.as_encoded_bytes().starts_with(b"core.") +} diff --git a/container/crowdb-monitor/src/lib.rs b/container/crowdb-monitor/src/lib.rs index cdb72eba6..ba2b2db88 100644 --- a/container/crowdb-monitor/src/lib.rs +++ b/container/crowdb-monitor/src/lib.rs @@ -2,6 +2,7 @@ // Licensed under the Apache License, Version 2.0. mod bootstrap; +mod crash; mod credentials; mod layout; mod liveness; @@ -22,6 +23,7 @@ pub use bootstrap::{ KvBootstrap, KvBootstrapError, LogicalBootstrap, LogicalBootstrapError, S3Bootstrap, S3BootstrapError, StorageProbeError, }; +pub use crash::CrashRetention; pub use credentials::{show_client_credentials, ClientCredentials, CredentialError, ServerCredentials}; pub use liveness::{probe_liveness, LivenessError, LivenessServer}; pub use manifest::{BootstrapManifest, BootstrapSession, ManifestError, ManifestState}; diff --git a/container/crowdb-monitor/src/preview.rs b/container/crowdb-monitor/src/preview.rs index ff72738d0..7dc8b7d0b 100644 --- a/container/crowdb-monitor/src/preview.rs +++ b/container/crowdb-monitor/src/preview.rs @@ -1,4 +1,4 @@ -use std::collections::{BTreeMap, BTreeSet}; +use std::collections::BTreeMap; use std::fs; use std::future::Future; use std::path::Path; @@ -10,11 +10,11 @@ use tokio::time::{sleep, Instant}; use crate::{ disk_step_names, ensure_disk_files, hardware_step_names, iceberg_step_names, kv_step_names, logical_step_names, render_configs, s3_step_names, verify_chunk_services, verify_diskio_disks, - BootstrapSession, ChunkBootstrapError, CredentialError, DeploymentProfile, DiskBootstrapError, - HardwareBootstrap, HardwareBootstrapError, IcebergBootstrap, IcebergBootstrapError, KvBootstrap, - KvBootstrapError, LivenessError, LivenessServer, LogicalBootstrap, LogicalBootstrapError, ManifestError, - ManifestState, MonitorEvent, MonitorEventKind, MonitorLogError, ProfileError, RenderError, S3Bootstrap, - S3BootstrapError, ServerCredentials, StorageProbeError, Supervisor, SupervisorError, + BootstrapSession, ChunkBootstrapError, CrashRetention, CredentialError, DeploymentProfile, + DiskBootstrapError, HardwareBootstrap, HardwareBootstrapError, IcebergBootstrap, IcebergBootstrapError, + KvBootstrap, KvBootstrapError, LivenessError, LivenessServer, LogicalBootstrap, LogicalBootstrapError, + ManifestError, ManifestState, MonitorEvent, MonitorEventKind, MonitorLogError, ProfileError, RenderError, + S3Bootstrap, S3BootstrapError, ServerCredentials, StorageProbeError, Supervisor, SupervisorError, }; const PROFILE_NAME: &str = "single-node-container"; @@ -83,6 +83,10 @@ pub async fn run_preview(profile_path: &Path) -> Result<(), PreviewError> { &config_bytes, &step_refs, )?; + if let Some(root) = std::env::var_os("CROWDB_CORE_DIR") { + let crashes = CrashRetention::open(root.into())?; + std::env::set_current_dir(crashes.root())?; + } let credentials = if session.manifest().state() == ManifestState::Ready || session.manifest().step_complete("s3-user") == Some(true) { @@ -478,20 +482,23 @@ fn kv_root(profile: &DeploymentProfile) -> Result Result, PreviewError> { - let mut files = BTreeSet::new(); + let mut files = BTreeMap::new(); for service in &profile.services { if let Some(path) = &service.config_template { let name = path .file_name() .ok_or(PreviewError::Invalid("template has no name"))?; - if !files.insert(name.to_os_string()) { + if files + .insert(name.to_os_string(), path) + .is_some_and(|existing| existing != path) + { return Err(PreviewError::Invalid("template name is duplicated")); } } } let mut input = Vec::new(); - for name in files { - let path = profile.paths.template_root.join(&name); + for name in files.keys() { + let path = profile.paths.template_root.join(name); let metadata = fs::symlink_metadata(&path)?; if !metadata.file_type().is_file() || metadata.len() > MAX_TEMPLATE_BYTES { return Err(PreviewError::Invalid("template is not a bounded regular file")); diff --git a/container/crowdb-monitor/src/process.rs b/container/crowdb-monitor/src/process.rs index e8b92c1ad..9a6fd4310 100644 --- a/container/crowdb-monitor/src/process.rs +++ b/container/crowdb-monitor/src/process.rs @@ -13,7 +13,9 @@ use tokio::process::{Child, Command}; use tokio::task::JoinHandle; use tokio::time::{sleep, timeout}; -use crate::{LogProfile, MonitorEvent, MonitorEventKind, MonitorLog, MonitorLogError, ServiceProfile}; +use crate::{ + CrashRetention, LogProfile, MonitorEvent, MonitorEventKind, MonitorLog, MonitorLogError, ServiceProfile, +}; #[derive(Debug, Error)] pub enum ProcessError { @@ -37,12 +39,17 @@ pub struct ProcessManager { log_root: PathBuf, log_policy: LogProfile, events: MonitorLog, + crashes: Option, } impl ProcessManager { /// # Errors /// Rejects an unavailable monitor lifecycle log. pub async fn new(log_root: PathBuf, log_policy: LogProfile) -> Result { + let crashes = std::env::var_os("CROWDB_CORE_DIR") + .map(PathBuf::from) + .map(CrashRetention::open) + .transpose()?; let mut events = MonitorLog::open(&log_root, log_policy.clone()).await?; events .record(&MonitorEvent { @@ -57,6 +64,7 @@ impl ProcessManager { log_root, log_policy, events, + crashes, }) } @@ -74,6 +82,9 @@ impl ProcessManager { std::fs::create_dir_all(&log_directory)?; retention::prune(&log_directory, None, self.log_policy.max_files.saturating_sub(1)).await?; let mut command = Command::new(&service.program); + if let Some(crashes) = &self.crashes { + command.current_dir(crashes.root()); + } command .args(&service.args) .envs(&service.env) @@ -191,6 +202,9 @@ impl ProcessManager { }) .await?; retention::prune(&self.log_root.join(id), None, self.log_policy.max_files).await?; + if let Some(crashes) = &self.crashes { + crashes.prune()?; + } Ok(()) } diff --git a/container/crowdb-monitor/src/render.rs b/container/crowdb-monitor/src/render.rs index 5ccc05036..afd02716e 100644 --- a/container/crowdb-monitor/src/render.rs +++ b/container/crowdb-monitor/src/render.rs @@ -1,4 +1,4 @@ -use std::collections::{BTreeMap, BTreeSet}; +use std::collections::BTreeMap; use std::fs::{self, File, OpenOptions}; use std::io::Write; use std::os::unix::fs::OpenOptionsExt; @@ -45,7 +45,7 @@ pub fn render_configs( Err(error) => return Err(error.into()), } let mut outputs = Vec::new(); - let mut file_names = BTreeSet::new(); + let mut file_names = BTreeMap::new(); for service in &profile.services { let Some(template) = &service.config_template else { continue; @@ -53,9 +53,20 @@ pub fn render_configs( let name = template .file_name() .ok_or_else(|| RenderError::Invalid("template path has no file name".into()))?; - if !file_names.insert(name.to_os_string()) { + if file_names + .insert(name.to_os_string(), template) + .is_some_and(|existing| existing != template) + { return invalid("multiple services render to the same file name"); } + let path = destination.join(name); + if outputs.iter().any(|output: &RenderedConfig| output.path == path) { + outputs.push(RenderedConfig { + service_id: service.id.clone(), + path, + }); + continue; + } let source = template_root.join(name); let metadata = fs::symlink_metadata(&source)?; if !metadata.file_type().is_file() || metadata.len() > MAX_TEMPLATE_BYTES { @@ -69,7 +80,6 @@ pub fn render_configs( toml::from_str::(&rendered).map_err(|error| { RenderError::Invalid(format!("template for {} is not TOML: {error}", service.id)) })?; - let path = destination.join(name); atomic_write(&path, rendered.as_bytes())?; outputs.push(RenderedConfig { service_id: service.id.clone(), diff --git a/container/crowdb-monitor/tests/crash_retention_test.rs b/container/crowdb-monitor/tests/crash_retention_test.rs new file mode 100644 index 000000000..898746119 --- /dev/null +++ b/container/crowdb-monitor/tests/crash_retention_test.rs @@ -0,0 +1,70 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs::{self, File, FileTimes}; +use std::os::unix::fs::{symlink, PermissionsExt}; +use std::path::{Path, PathBuf}; +use std::time::{Duration, SystemTime}; + +use crowdb_monitor::CrashRetention; +use uuid::Uuid; + +struct TestRoot(PathBuf); + +impl TestRoot { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-crash-{}", Uuid::new_v4())); + fs::create_dir_all(&root).unwrap(); + Self(root) + } +} + +impl Drop for TestRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn core(root: &Path, name: &str, seconds: u64) { + let file = File::create(root.join(name)).unwrap(); + file.set_times(FileTimes::new().set_modified(SystemTime::UNIX_EPOCH + Duration::from_secs(seconds))) + .unwrap(); +} + +#[test] +fn latest_core_survives_startup_and_recovery_without_touching_other_files() { + let test = TestRoot::new(); + let root = test.0.join("crash"); + fs::create_dir(&root).unwrap(); + fs::set_permissions(&root, fs::Permissions::from_mode(0o755)).unwrap(); + core(&root, "core.10", 10); + core(&root, "core.11", 11); + core(&root, "notes", 12); + symlink(root.join("notes"), root.join("core.link")).unwrap(); + + let retention = CrashRetention::open(root.clone()).unwrap(); + assert_eq!(fs::metadata(&root).unwrap().permissions().mode() & 0o777, 0o700); + assert!(!root.join("core.10").exists()); + assert!(root.join("core.11").exists()); + assert!(root.join("notes").exists()); + assert!(root.join("core.link").exists()); + + core(&root, "core.12", 12); + retention.prune().unwrap(); + assert!(!root.join("core.11").exists()); + assert!(root.join("core.12").exists()); +} + +#[test] +fn symlinked_core_directory_is_rejected() { + let test = TestRoot::new(); + let real = test.0.join("real"); + fs::create_dir(&real).unwrap(); + let alias = test.0.join("alias"); + symlink(real, &alias).unwrap(); + assert!(CrashRetention::open(alias).is_err()); +} diff --git a/container/crowdb-monitor/tests/hardware_bootstrap_test.rs b/container/crowdb-monitor/tests/hardware_bootstrap_test.rs index 42529ebb7..171486275 100644 --- a/container/crowdb-monitor/tests/hardware_bootstrap_test.rs +++ b/container/crowdb-monitor/tests/hardware_bootstrap_test.rs @@ -121,6 +121,7 @@ async fn conflicting_group_zero_record_rejects_without_creating_hardware() { &RackValue { status: HwStatus::Up as i32, node_ids: vec![99], + ..Default::default() }, ) .await diff --git a/container/crowdb-monitor/tests/render_test.rs b/container/crowdb-monitor/tests/render_test.rs index 644a18679..1dcf470c5 100644 --- a/container/crowdb-monitor/tests/render_test.rs +++ b/container/crowdb-monitor/tests/render_test.rs @@ -52,7 +52,14 @@ fn profile() -> DeploymentProfile { fn renders_profile_paths_and_topology_without_secrets() { let dirs = TestDirs::new(); let outputs = render_configs(&profile(), &dirs.templates(), &dirs.run()).unwrap(); - assert_eq!(outputs.len(), 6); + assert_eq!(outputs.len(), 8); + assert_eq!( + outputs + .iter() + .filter(|output| output.path == dirs.run().join("config/access.toml")) + .count(), + 2 + ); let diskio = fs::read_to_string(dirs.run().join("config/diskio.toml")).unwrap(); assert!(diskio.contains("path = \"/opt/crowdb/data/disks/disk-0004.img\"")); assert!(diskio.contains("zone_capacity = 17179869184")); diff --git a/container/crowdb-monitor/tests/single_node_profile_test.rs b/container/crowdb-monitor/tests/single_node_profile_test.rs index 302d29319..e651e82f3 100644 --- a/container/crowdb-monitor/tests/single_node_profile_test.rs +++ b/container/crowdb-monitor/tests/single_node_profile_test.rs @@ -48,18 +48,27 @@ fn single_node_preview_has_exact_topology_and_endpoints() { iceberg.env.get("CROWDB_ICEBERG_PUBLIC_URI"), Some(&"http://localhost".to_owned()) ); + assert_eq!(iceberg.probe.target, "http://127.0.0.1:80/v1/config"); assert_eq!( - iceberg.env.get("CROWDB_ICEBERG_LISTEN"), - Some(&"0.0.0.0:80".to_owned()) + iceberg.env.get("CROWDB_MANAGEMENT_SEEDS"), + Some(&"http://127.0.0.1:10000".to_owned()) ); - assert_eq!(iceberg.probe.target, "http://127.0.0.1:80/v1/config"); let s3 = profile .services .iter() .find(|service| service.id == "s3") .unwrap(); - assert_eq!(s3.env.get("CROWDB_S3_LISTEN"), Some(&"0.0.0.0:81".to_owned())); assert_eq!(s3.probe.target, "http://127.0.0.1:81/_crowdb/health/ready"); + assert_eq!(s3.args[1], "/opt/crowdb/run/config/access.toml"); + assert_eq!(iceberg.args[2], s3.args[1]); + assert_eq!(iceberg.config_template, s3.config_template); + let access = std::fs::read_to_string( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/templates/access.toml"), + ) + .unwrap(); + let access: toml::Value = toml::from_str(&access).unwrap(); + assert_eq!(access["iceberg"]["listen"].as_str(), Some("0.0.0.0:80")); + assert_eq!(access["s3"]["listen"].as_str(), Some("0.0.0.0:81")); let web = profile .services .iter() diff --git a/container/crowdb-monitor/tests/storage_bootstrap_test.rs b/container/crowdb-monitor/tests/storage_bootstrap_test.rs index 3b57a9b0d..338806038 100644 --- a/container/crowdb-monitor/tests/storage_bootstrap_test.rs +++ b/container/crowdb-monitor/tests/storage_bootstrap_test.rs @@ -84,7 +84,11 @@ impl TestRoot { .to_string_lossy() .into_owned(), ], - "iceberg" => vec!["serve".into()], + "iceberg" => vec![ + "serve".into(), + "--config".into(), + format!("{}/run/config/access.toml", self.0.display()), + ], _ => unreachable!(), }; if service.id == "iceberg" { @@ -135,6 +139,7 @@ impl TestRoot { "diskio.toml", "chunkdb.toml", "chunk-kv.toml", + "access.toml", ] { let body = fs::read_to_string(source.join(name)).unwrap(); let body = body @@ -149,7 +154,8 @@ impl TestRoot { .replace("127.0.0.1:12100", &format!("127.0.0.1:{}", ports.chunkdb_http)) .replace("127.0.0.1:12200", &format!("127.0.0.1:{}", ports.chunkdb_rpc)) .replace("127.0.0.1:15100", &format!("127.0.0.1:{}", ports.chunk_kv_http)) - .replace("127.0.0.1:15200", &format!("127.0.0.1:{}", ports.chunk_kv_rpc)); + .replace("127.0.0.1:15200", &format!("127.0.0.1:{}", ports.chunk_kv_rpc)) + .replace("0.0.0.0:80", &format!("127.0.0.1:{}", ports.iceberg)); fs::write(self.0.join("templates").join(name), body).unwrap(); } } diff --git a/container/single-node-container/Dockerfile b/container/single-node-container/Dockerfile index db18caefb..c32d125b2 100644 --- a/container/single-node-container/Dockerfile +++ b/container/single-node-container/Dockerfile @@ -21,6 +21,7 @@ RUN chmod 0755 /opt/crowdb/bin/entrypoint \ ARG SOURCE_REVISION ARG PREVIEW_VERSION +ARG RUNTIME_SHA256 RUN --mount=type=bind,source=.,target=/staged,ro \ test -n "$SOURCE_REVISION" && test -n "$PREVIEW_VERSION" \ && test "$(cat /staged/SOURCE_REVISION)" = "$SOURCE_REVISION" \ @@ -32,7 +33,8 @@ LABEL org.opencontainers.image.title="CROWDB Iceberg" \ org.opencontainers.image.authors="Gian " \ org.opencontainers.image.licenses="Apache-2.0" \ org.opencontainers.image.revision="$SOURCE_REVISION" \ - org.opencontainers.image.version="$PREVIEW_VERSION" + org.opencontainers.image.version="$PREVIEW_VERSION" \ + org.crowdb.runtime.sha256="$RUNTIME_SHA256" ENV PATH="/opt/crowdb/bin:${PATH}" \ LD_LIBRARY_PATH="/opt/crowdb/lib" \ CROWDB_RUNTIME_ROOT="/opt/crowdb/run" diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index e8b5b3dbd..8fb2891cf 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -25,7 +25,109 @@ pixi run test-single-node-container `CROWDB_CONTAINER_IMAGE` to build and test a separate candidate tag. `pixi run stage-single-node-container` produces the runtime directory without -building a Docker image. The release workflow archives the verified directory -and packages those same files in its publish job, without recompiling them. -Docker Hub publication is manual; actual publication verification is deferred -until administrator preparation is complete. +building a Docker image. To prepare a release from a clean, current `main` +checkout, preview the patch bump and then run it explicitly: + +```sh +pixi run -- python tools/release.py --dry-run +pixi run -- python tools/release.py --execute +``` + +`--bump minor` and `--bump major` select larger version changes. The script +updates every version manifest, commits and pushes the candidate to `main`, +then dispatches the release workflow. It does not create a tag or GitHub +Release. You can also run the workflow manually on `main`; it derives the tag +from `VERSION`, so no tag input is needed. Add `--symbols` to either command +to include the large exact-build symbol archive; the default release skips it. +Execution requires authenticated `gh` and GitHub permission to push `main`; +the dry run changes no files or remote state. +The script passes its candidate commit SHA to the workflow so a later push to +`main` cannot silently change which commit gets published. + +The workflow checks that CI passed for the exact candidate commit, builds and +tests the container, then waits for DockerHub environment approval. Only after +verification does it create the Git tag and publish the signed Docker image +and GitHub Release. A failed verification leaves no tag or draft release. Fix +the candidate and run the workflow again; if code changes after a tag was +created, use the next patch version. A publication retry for the same commit +reuses an existing image only when both registry tags have the same digest and +the image labels match the release version, commit, and verified runtime archive. + +The workflow archives the verified runtime, then packages those same files in +its publish job without recompiling them. With `--symbols`, it also archives +exact-build symbols from that build and attaches +`crowdb-symbols--git--linux-amd64.tar.zst` to the GitHub +Release. The workflow publishes the GitHub Release after the Docker image and +signature succeed. If the optional symbol upload fails, the published release +remains available and the workflow reports a warning. + +## Crash collection boundary + +For host configuration, restoring its collector, and GDB commands for both +container and bare-metal cores, see the +[crash debugging guide](../../doc/dev/crash_debugging.md). + +The image does not configure the host's Linux core collector. Inspect +`/proc/sys/kernel/core_pattern` on the Docker host before expecting a dump in +the mounted data volume. A leading `|` sends a crash to a host-side collector; +relative `core` or `core.*` patterns write in the crashing process's working +directory. The container runs the monitor and managed children from the private +`/opt/crowdb/data/crash` directory. On monitor startup and after a child is +reaped, it removes older regular `core` files and keeps the newest one. +The directory has mode `0700`; symlinks named `core.*` are not followed or +deleted. This is one-core retention, not a promise that the host creates a +volume file. Other relative filename patterns are outside this retention rule. + +For a host with a relative `core` pattern, add a size bound to `docker run`: + +```sh +--ulimit core=1073741824:1073741824 +``` + +The example bounds each core to 1 GiB. A small bound may truncate a dump and +make some stack frames unavailable. Core collection can also be suppressed by +the host's dumpability policy, including for executables with file capabilities. +The container never changes `core_pattern` or the host's dumpability policy. + +- On a systemd-coredump host, use `coredumpctl list crowdb-kv-server` to find + the host report, then `coredumpctl --output=/private/core dump + crowdb-kv-server` as an authorized host user to export it. Check that the + result is readable and nonempty before symbolization. +- On an Ubuntu Apport host, find the matching report in `/var/crash` on the + Docker host. Create a private directory, then run `sudo apport-unpack + /var/crash/REPORT.crash /private/crowdb-core/unpacked`. The extracted + `CoreDump` is the file to pass to the symbolizer. Apport reports may be + readable only by the host administrator; preserve the private permissions + when granting the debugging user access. A pipe pattern does not create a + volume file. Apport may fail to resolve a CROWDB executable because its + `/opt/crowdb/bin` path exists only inside the container; it may also ignore + executables outside host distribution packages. Check the host's Apport log + when no report appears. + If the report or `CoreDump` is absent, collection is unavailable for that + crash; do not substitute a log or an unrelated dump. +- On Docker Desktop, inspect the Linux VM's collector. The desktop host's + native crash directory is not the container's core directory. + +Core dumps can contain credentials and user data. Keep exports in a private +directory and do not attach them to ordinary logs or issues. If the release +included the optional symbol archive, use the exact image and matching archive +to show source-line stacks: + +```sh +pixi run -- python tools/symbolize-container-core.py \ + --image 'docker.io/crowdb/crowdb-iceberg:' \ + --symbols '/private/path/crowdb-symbols--git--linux-amd64.tar.zst' \ + --binary crowdb-monitor --core /private/path/core +``` + +The tool copies binaries from a stopped container into a temporary private +directory, verifies source revision, version and SHA-256 hashes, then runs +`gdb` without printing frame arguments. Use the crashed child binary instead +of `crowdb-monitor` for a child core. The temporary binaries are removed after +the stack is shown; the core stays at the path supplied by the operator. +The operator will validate collection and source-line output when a real +crash is available. No host collector change is required by the image build. + +Collector behavior follows the [Linux core pattern documentation](https://docs.kernel.org/admin-guide/sysctl/kernel.html), +[systemd-coredump manual](https://www.freedesktop.org/software/systemd/man/250/systemd-coredump.socket.html), +and [Ubuntu Apport documentation](https://ubuntu.com/project/docs/contributors/debugging/apport/). diff --git a/container/single-node-container/build.sh b/container/single-node-container/build.sh index 663ad44cb..78d0bd5a4 100644 --- a/container/single-node-container/build.sh +++ b/container/single-node-container/build.sh @@ -10,13 +10,22 @@ cd "$(git rev-parse --show-toplevel)" for tool in patchelf strip ldd; do command -v "$tool" >/dev/null || { echo "Missing packaging tool: $tool" >&2; exit 1; } done +if [[ "${CROWDB_PACKAGE_SYMBOLS:-0}" == 1 ]]; then + for tool in objcopy readelf; do + command -v "$tool" >/dev/null || { echo "Missing packaging tool: $tool" >&2; exit 1; } + done + export CARGO_PROFILE_RELEASE_DEBUG=line-tables-only + cmake_build_type=RelWithDebInfo +else + cmake_build_type=Release +fi if [[ "$mode" == image ]]; then docker info >/dev/null fi # Build on the host, reusing the existing Cargo, CMake and npm artifacts. cargo build --locked --release -p crowdb-kv-client --features ffi -cmake -S app/crowdb-diskio -B app/crowdb-diskio/build -DCMAKE_BUILD_TYPE=Release +cmake -S app/crowdb-diskio -B app/crowdb-diskio/build -DCMAKE_BUILD_TYPE="$cmake_build_type" cmake --build app/crowdb-diskio/build -j 4 --target crowdb-diskio cargo build --locked --release \ -p crowdb-monitor -p crowdb-kv-server -p crowdb-diskdb \ @@ -33,6 +42,13 @@ cp -a container/single-node-container/templates "$staging/templates" cp container/single-node-container/{Dockerfile,profile.toml,entrypoint.sh} "$staging/" git rev-parse HEAD > "$staging/SOURCE_REVISION" cp VERSION "$staging/VERSION" +if [[ "${CROWDB_PACKAGE_SYMBOLS:-0}" == 1 ]]; then + cp VERSION "$staging/symbols/VERSION" + git rev-parse HEAD > "$staging/symbols/SOURCE_REVISION" + (cd "$staging" && sha256sum bin/* lib/libcrowdb*.so) > "$staging/symbols/RUNTIME_SHA256SUMS" + rm -rf target/container-symbols + mv "$staging/symbols" target/container-symbols +fi rm -rf target/container-runtime mv "$staging" target/container-runtime trap - EXIT diff --git a/container/single-node-container/collect-libs.sh b/container/single-node-container/collect-libs.sh index 4276ad68a..72b950d93 100644 --- a/container/single-node-container/collect-libs.sh +++ b/container/single-node-container/collect-libs.sh @@ -56,6 +56,19 @@ for library in "$output"/lib/*; do done rm "$output/dependencies.txt" +if [[ "${CROWDB_PACKAGE_SYMBOLS:-0}" == 1 ]]; then + for artifact in "$output"/bin/* "$output"/lib/libcrowdb*.so; do + [[ -f "$artifact" ]] || continue + if ! readelf -W -S "$artifact" | grep -E '[[:space:]]\.debug_line[[:space:]]' >/dev/null; then + echo "Missing source-line symbols: $artifact" >&2 + exit 1 + fi + symbol="$output/symbols/$(basename "$(dirname "$artifact")")/$(basename "$artifact").debug" + mkdir -p "$(dirname "$symbol")" + objcopy --only-keep-debug "$artifact" "$symbol" + objcopy --add-gnu-debuglink="$symbol" "$artifact" + done +fi for artifact in "$output"/bin/* "$output"/lib/*; do strip --strip-debug "$artifact" done diff --git a/container/single-node-container/entrypoint.sh b/container/single-node-container/entrypoint.sh index 220c5c483..f578f62d0 100644 --- a/container/single-node-container/entrypoint.sh +++ b/container/single-node-container/entrypoint.sh @@ -8,4 +8,21 @@ fi echo 'CROWDB preview data: Docker creates an anonymous volume when none is specified. For data you want to keep across container recreation, use --mount type=volume,source=crowdb-data,target=/opt/crowdb/data.' +CROWDB_CORE_DIR=/opt/crowdb/data/crash +export CROWDB_CORE_DIR + +case "$(cat /proc/sys/kernel/core_pattern)" in + core|core.*) + if [ "$(ulimit -c)" = 0 ]; then + echo 'CROWDB core files require a nonzero Docker --ulimit core setting.' >&2 + fi + ;; + '|'*) + echo 'CROWDB core dumps use the Docker host collector; inspect the host for dumps.' >&2 + ;; + *) + echo 'CROWDB core_pattern does not use a relative core filename; inspect the Docker host for dumps.' >&2 + ;; +esac + exec /opt/crowdb/bin/crowdb-monitor run --profile /opt/crowdb/etc/profile.toml diff --git a/container/single-node-container/profile.toml b/container/single-node-container/profile.toml index b3bf07dd4..de9b299aa 100644 --- a/container/single-node-container/profile.toml +++ b/container/single-node-container/profile.toml @@ -173,8 +173,8 @@ backoff_max_ms = 5000 [[services]] id = "s3" program = "/opt/crowdb/bin/crowdb-access-server" -args = ["--config", "/opt/crowdb/run/config/s3.toml"] -env = { CROWDB_S3_PUBLIC_URI = "http://localhost:81", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000" } +args = ["--config", "/opt/crowdb/run/config/access.toml"] +env = { CROWDB_S3_PUBLIC_URI = "http://localhost:81", CROWDB_S3_REGION = "us-east-1", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000" } dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] fence_listeners = ["127.0.0.1:81"] config_template = "/opt/crowdb/etc/templates/access.toml" @@ -191,8 +191,8 @@ backoff_max_ms = 5000 [[services]] id = "iceberg" program = "/opt/crowdb/bin/crowdb-iceberg" -args = ["serve", "--config", "/opt/crowdb/run/config/iceberg.toml"] -env = { CROWDB_ICEBERG_PUBLIC_URI = "http://localhost", CROWDB_ICEBERG_GC_ENABLED = "0" } +args = ["serve", "--config", "/opt/crowdb/run/config/access.toml"] +env = { CROWDB_ICEBERG_PUBLIC_URI = "http://localhost", CROWDB_ICEBERG_GC_ENABLED = "0", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000" } dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] fence_listeners = ["127.0.0.1:80"] config_template = "/opt/crowdb/etc/templates/access.toml" diff --git a/container/single-node-container/tests/container-e2e.sh b/container/single-node-container/tests/container-e2e.sh index daf21126f..476e00db9 100644 --- a/container/single-node-container/tests/container-e2e.sh +++ b/container/single-node-container/tests/container-e2e.sh @@ -284,6 +284,10 @@ client_env=$(docker exec "$name" crowdb-monitor credentials show --format env) [[ $(docker exec "$name" stat -c %a /opt/crowdb/data/secrets/server.env) == 600 ]] [[ $(docker exec "$name" stat -c %a /opt/crowdb/data/secrets/client.env) == 600 ]] docker exec "$name" cat /opt/crowdb/data/bootstrap/manifest.json | jq -e '.state == "ready"' >/dev/null +[[ $(docker exec "$name" stat -c %a /opt/crowdb/data/crash) == 700 ]] +[[ $(docker exec "$name" readlink /proc/1/cwd) == /opt/crowdb/data/crash ]] +kv_pid=$(docker exec "$name" cat /opt/crowdb/run/status/monitor.json | jq -er '.services.kv.pid') +[[ $(docker exec "$name" readlink "/proc/$kv_pid/cwd") == /opt/crowdb/data/crash ]] verify_public_services node container/single-node-container/tests/web-ui.cjs "http://127.0.0.1:$(port 8080)" "$name" echo "checking S3 and Iceberg client writes" diff --git a/container/single-node-container/tests/release-policy.sh b/container/single-node-container/tests/release-policy.sh index cc75258f3..8fc721df0 100644 --- a/container/single-node-container/tests/release-policy.sh +++ b/container/single-node-container/tests/release-policy.sh @@ -2,7 +2,7 @@ set -euo pipefail release=.github/workflows/release-container.yml -ci=.github/workflows/ci.yml +preview=.github/workflows/docker-preview.yml events=$(sed -n '/^on:/,/^concurrency:/p' "$release") [[ "$events" == *'workflow_dispatch:'* ]] @@ -11,13 +11,23 @@ events=$(sed -n '/^on:/,/^concurrency:/p' "$release") for required in \ 'environment: DockerHub' \ 'DOCKERHUB_TOKEN' \ - 'ref: ${{ inputs.tag }}' \ - 'git rev-parse --verify "refs/tags/$RELEASE_TAG^{commit}"' \ - '[[ "$revision" == "$(git rev-parse HEAD)" ]]' \ + 'ref: ${{ github.sha }}' \ + 'ref: ${{ needs.verify.outputs.revision }}' \ + 'git rev-parse "refs/tags/$RELEASE_TAG^{commit}"' \ + '[[ "$revision" == "$GITHUB_SHA" ]]' \ 'gh release view "$RELEASE_TAG"' \ - '[[ "$status" == 404 ]]' \ + 'gh release create "$RELEASE_TAG"' \ + 'git push origin "refs/tags/$RELEASE_TAG"' \ + 'actions: read' \ + 'head_sha=$REVISION&branch=main&event=push' \ + 'completed/success) exit 0' \ + 'reuse=false' \ + 'reuse=true' \ + 'runtime_sha256: ${{ steps.runtime_digest.outputs.sha256 }}' \ + 'RUNTIME_SHA256=${{ needs.verify.outputs.runtime_sha256 }}' \ + 'org.crowdb.runtime.sha256' \ 'needs: verify' \ - 'docker.io/crowdb/crowdb-iceberg:${{ inputs.tag }}' \ + 'docker.io/crowdb/crowdb-iceberg:${{ needs.verify.outputs.tag }}' \ 'docker.io/crowdb/crowdb-iceberg:git-${{ needs.verify.outputs.revision }}' \ 'provenance: mode=max' \ 'sbom: true' \ @@ -28,22 +38,46 @@ done [[ $(grep -c 'push: true' "$release") == 1 ]] [[ $(grep -c 'id-token: write' "$release") == 1 ]] [[ "$events" != *'schedule:'* ]] +[[ "$events" != *' tag:'* ]] verify_job=$(sed -n '/^ verify:/,/^ publish:/p' "$release") publish_job=$(sed -n '/^ publish:/,$p' "$release") [[ "$verify_job" == *'name: verified-container-runtime'* ]] [[ "$publish_job" == *'name: verified-container-runtime'* ]] +[[ "$verify_job" == *'name: verified-container-symbols'* ]] +[[ "$publish_job" == *'name: verified-container-symbols'* ]] +[[ "$events" == *'include_symbols:'* && "$events" == *'default: false'* ]] +[[ "$verify_job" == *"CROWDB_PACKAGE_SYMBOLS: \${{ inputs.include_symbols && '1' || '0' }}"* ]] +[[ "$verify_job" == *'pixi run -- python tools/ci-checks/check-container-symbols.py'* ]] +[[ "$verify_job" == *'if: inputs.include_symbols'* ]] +[[ "$publish_job" == *'if: inputs.include_symbols && steps.symbols_download.outcome'* ]] +[[ "$verify_job" == *'continue-on-error: true'* ]] +[[ "$publish_job" == *'continue-on-error: true'* ]] +[[ "$publish_job" == *'gh release upload "$RELEASE_TAG"'* ]] +[[ "$publish_job" == *'gh release edit "$RELEASE_TAG"'* ]] [[ "$publish_job" == *'context: target/container-runtime'* ]] -for gate in 'pixi run test-single-node-container' 'test-boto3-e2e' 'test-pyiceberg-e2e' \ - 'pixi run test-console' 'pixi run test-console-ui' 'pixi run rs-fmt-check && pixi run rs-lint'; do +for gate in 'pixi run test-single-node-container' 'Require CI success for the release commit'; do [[ "$verify_job" == *"$gate"* ]] done +[[ "$verify_job" != *'git push origin'* && "$verify_job" != *'gh release create'* ]] ! grep -Eq 'DOCKERHUB_|push: true|id-token: write' <<<"$verify_job" [[ "$publish_job" == *'needs: verify'* && "$publish_job" == *'environment: DockerHub'* ]] [[ "$publish_job" != *'RELEASE_ENABLED'* ]] [[ "$publish_job" == *'[[ "$(git rev-parse HEAD)" == "$REVISION" ]]'* ]] +[[ "$publish_job" == *'Create release tag after verification'* ]] +grep -Fq 'org.crowdb.runtime.sha256="$RUNTIME_SHA256"' container/single-node-container/Dockerfile -ci_job=$(sed -n '/^ DockerPreview:/,$p' "$ci") -[[ "$ci_job" == *'contents: read'* && "$ci_job" == *'pixi run test-single-node-container'* ]] -[[ "$ci_job" == *'Upload preview failure logs'* && "$ci_job" == *'CROWDB_PREVIEW_TEST_ARTIFACTS'* ]] -! grep -Eq 'secrets\.|docker/login-action|docker/build-push-action' <<<"$ci_job" +preview_events=$(sed -n '/^on:/,/^jobs:/p' "$preview") +[[ "$preview_events" == *'workflow_dispatch:'* ]] +! grep -Eq '^ (push|pull_request|release|create):' <<<"$preview_events" +preview_job=$(sed -n '/^ DockerPreview:/,$p' "$preview") +[[ "$preview_job" == *'contents: read'* && "$preview_job" == *'pixi run test-single-node-container'* ]] +[[ "$preview_job" == *'Upload preview failure logs'* && "$preview_job" == *'CROWDB_PREVIEW_TEST_ARTIFACTS'* ]] +! grep -Eq 'secrets\.|docker/login-action|docker/build-push-action' <<<"$preview_job" + +release_tool=tools/release.py +for required in '--dry-run' '--execute' '--symbols' '"push", "origin", "HEAD:refs/heads/main"' \ + '"workflow", "run"' '"--ref", "main"'; do + grep -Fq -- "$required" "$release_tool" +done +! grep -Eq '"release", "create"|"tag", "-a"' "$release_tool" diff --git a/doc/backlog/R148-chunk-stream-scale-out.md b/doc/backlog/R148-chunk-stream-scale-out.md index c33416348..b309fe779 100644 --- a/doc/backlog/R148-chunk-stream-scale-out.md +++ b/doc/backlog/R148-chunk-stream-scale-out.md @@ -140,3 +140,15 @@ Required gates: - `pixi run -- cargo test -p crowdb-chunk-kv --all-targets` - `pixi run -- cargo test -p crowdb-chunkdb --all-targets` - `pixi run clean-env && pixi run test-server` + +## Open Issues + +- The CI log for + `small_object_writer_e2e::eight_closed_mirror_strips_become_one_durable_ec_strip_without_reread` + failed at `location strip`, while the exact test, its 16-test suite and the + full `test-server` task pass locally at default test concurrency. CI run + `36449749925` completed its "Upload test logs on failure" step, but the run + artifact list contains only `docker-preview-1`; the `runtime-server` artifact + is absent. The test now prints the queried chunk and location on this failure. + Wait for a recurrence and use those diagnostics to identify the first + divergent state before changing the assertion or retry policy. diff --git a/doc/backlog/R167-s3-multipart-upload.md b/doc/backlog/R167-s3-multipart-upload.md deleted file mode 100644 index c7927bd96..000000000 --- a/doc/backlog/R167-s3-multipart-upload.md +++ /dev/null @@ -1,80 +0,0 @@ - - - -### R167: access server / S3 — Multipart upload - -## Status - -**Deferred until the R152–R166 basic S3 milestone is complete.** -It is unblocked when single-request streaming, publication recovery, ETag, -listing, deletion, and compatibility E2E behavior are stable. - -## Problem - -Multipart upload adds durable upload/part listing, independent part retries, -completion ordering, abort cleanup, and multipart ETag semantics. Adding it -before the basic publication and cleanup paths stabilize would duplicate -unsettled recovery rules and delay the deliberately limited first service. - -The scope boundary is -`doc/design/access-server/s3/design-crowdb-access-s3.md` §1. - -## Solution - -1. Add create, upload-part, list-parts, complete, abort, and required upload - listing operations through S3-owned API and namespace adapters over the - protocol-neutral multipart session, part, and completion core from R190. -2. Store immutable part identities and integrity records durably; a retried - part number replaces only that part's selected generation and schedules old - private data for cleanup. -3. Complete with one fenced metadata transaction that validates ordered part - identities, sizes, checksums, and expected upload state before publishing - one immutable object generation. Compose the selected parts' chunk-location - arrays with adjusted logical offsets. Complete does not read part data or - concatenate it through access-server memory. -4. Persist each uploaded part's raw 16-byte MD5. The multipart ETag is the - lowercase hexadecimal MD5 of the selected parts' raw MD5 bytes in order, - followed by `-`. Keep the basic single-part ETag rule unchanged. - Abort and expiry create bounded, idempotent cleanup records. -5. Preserve the basic admission bounds for parallel part traffic and wire - compatibility for all retry and conflict outcomes. - -## Dependencies - -- Depends on R152–R166. -- Reuses the basic milestone's publication/recovery, streaming input, logical - deletion, and integrity contracts. -- Reuses R190's protocol-neutral multipart core; S3 retains its own - authorization, namespace, wire errors, ETag response, and object publication. -- R170 owns any accelerated multipart transfer and additionally depends on this - requirement before enabling that operation. - -## Acceptance - -- Given parts uploaded out of order with part retries, when completion names a - valid order, assert exact concatenated bytes become visible through one - generation without gateway concatenation or part reads. Invariant: completion - is metadata-only, atomic and storage-backed. E2E test. -- Given completed parts with known MD5 values, when completion selects and - reorders them, assert the ETag uses only the selected raw part MD5 values in - completion order and the part count suffix. Invariant: multipart ETag matches - the S3-compatible composite algorithm. Unit test. -- Given missing, duplicated, undersized, checksum-mismatched, or concurrently - replaced parts, when completion runs, assert no object publishes and exact - errors are stable. Invariant: only the validated part set can publish. - Integration test. -- Given response loss during part upload, complete, and abort, when identities - retry after restart, assert the durable outcome is returned without duplicate - generations or cleanup. Invariant: every multipart transition is idempotent. - E2E test. -- Given abort/expiry with readers or cleanup failures, when reconciliation - runs, assert no active upload is reclaimed and unreachable part data is - eventually queued within bounds. Invariant: cleanup follows durable upload - state. Integration test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-s3 --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` diff --git a/doc/backlog/R170-s3-cuobject-rdma.md b/doc/backlog/R170-s3-cuobject-rdma.md index 768341e86..8a09a6d53 100644 --- a/doc/backlog/R170-s3-cuobject-rdma.md +++ b/doc/backlog/R170-s3-cuobject-rdma.md @@ -91,8 +91,8 @@ Client/GPU AccessServer Chunk plan DiskIO A..D publication, range, integrity, or error contracts. - Uses `crowdb-chunk-client`, `crowdb-diskio`, internal authenticated RPC, and native ownership/completion support from `crowdb-rpc-ffi`. -- Accelerated multipart upload additionally depends on R167 and remains - disabled until both requirements land. +- Accelerated multipart upload builds on the existing S3 multipart authority + and remains disabled until this acceleration requirement lands. - NVIDIA cuObject server libraries, compatible drivers, and ConnectX-5-or-newer hardware are optional deployment dependencies. Unsupported deployments keep the basic TCP service unchanged. diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md deleted file mode 100644 index ee422080d..000000000 --- a/doc/backlog/R188-console-group0-authority.md +++ /dev/null @@ -1,155 +0,0 @@ - - - -### R188: console — Group 0 authority and deployment configuration cleanup - -## Problem - -The unreleased `ConsoleConfig` in -[`design-crowdb-console.md`](../design/console/design-crowdb-console.md) -currently mixes cluster topology with host, SSH, binary, port, PID, and local -launch information. CLI and bare-metal Web can write the local file before a -Group 0 mutation succeeds, and some reads and restart paths still accept that -file or a monitor cache as topology authority. A second console can therefore -observe a different cluster, while an outage can resurrect stale topology. - -R187's Docker monitor already owns container process supervision; Group 0 owns -CROWDB system metadata, not container IDs, images, mounts, PIDs, restart -generations, or machine-local launch policy. Completing a cross-mode console -rewrite is not a prerequisite for packaging that monitor and the single-node -profile. The remaining boundary cleanup belongs in this separate requirement. - -At the user's request, this requirement also owns the deferred container crash -diagnostics work: core collection, bounded retention and source-line -symbolization. Existing crash recovery is implemented, but usable diagnostic -dumps depend on the host collector and exact-build symbols. This follow-up does -not block R187 completion. - -## Solution - -1. Keep Group 0 as the durable authority for CROWDB hardware hierarchy, - ownership and binding maps, KV store/group/replica metadata, and service - registration. Do not add Docker or bare-metal process deployment records to - Group 0. Docker process state comes from `crowdb-monitor`; bare-metal launch - policy remains local. Deployment mode changes which lifecycle and hardware - controls are allowed, not the meaning of Group 0 records. -2. Replace the mixed `ConsoleConfig` persistence in - `crowdb-console-shared::config`, `crowdb-web`, and `crowdb-cli` with a - versioned Web process configuration and a separate bare-metal launch-only - registry. Finish wiring the existing `LaunchRegistry` parser to actual - bare-metal deploy/restart operations; remove the unreleased mixed - parser/writer, topology fields, restore path, fixtures, and fallback rather - than adding a compatibility reader. Retain SSH credential references, - binary/config paths, workspace, and auto-start policy locally; never persist - inline secrets or runtime PID as topology. Docker Web rejects a launch - registry and keeps its monitor-owned process path. -3. Unify CLI and bare-metal Web hardware mutations through Group 0-backed - operations in `crowdb-console-shared::ops::hardware`. Confirm writes before - updating a read model; preserve conflicts and uncertain results. Docker Web - continues to reject hardware and process mutations. -4. Complete the common Group 0-backed logical store/group/replica flow in - `crowdb-console-shared::ops::kv_logical` for CLI and both Web modes. Reconcile - lost responses by reading confirmed authority, test multi-node fan-out and - rollback, and remove local logical-topology commits. A failed node-side - deletion must not erase surviving Group 0 membership. -5. Replace config-backed monitor refresh, KV endpoint fallback, and - physical/deployment topology reads with Group 0 membership and live service - registration. A missing or ambiguous live endpoint fails unavailable; a - stopped service may still have local launch policy but is not reported as - live. Neither a local launch registry nor a monitor cache is an authority - fallback during a Group 0 outage. -6. Keep pre-Group-0 bootstrap intent separate. After creation, verify every - committed hardware and logical record and delete the local topology copy. - Persist enough bootstrap identity to resume an interrupted transfer, prove - already committed content, and reject conflict. Destroy/clean must use - confirmed Group 0 state. If nonmember KV processes were launched before - Group 0 exists, propagate usable Group 0 seed hints after initialization - before treating their registration as live; seed hints are not topology. -7. Audit the S3 mini-cluster's local `console.toml` and restart path under the - same authority boundary. Retain only launch inputs and bootstrap seeds - locally after Group 0 cutover; do not replay a local topology copy. -8. Migrate the verified bare-metal deployment and operations material into - dedicated bare-metal deployment documentation, organized by KV cluster, - chunk layer, and data access servers. State that bare-metal is not yet - production-ready. Keep Docker deployment documentation independent. -9. Complete container crash diagnostics without changing host-wide collector - policy. Respect file-based core patterns, Ubuntu Apport, systemd-coredump and - Docker Desktop's Linux VM; document where dumps actually go or why collection - is unavailable. Where file dumps are supported, retain them in a bounded, - private data-volume location. Provide an exact-build source-line - symbolization workflow for child and monitor crashes. Dumps can contain - secrets and user data; diagnostics must not expose them in ordinary logs. - Host acceptance and symbol-distribution choices remain open in the execution - plan; no image-size increase or host configuration change is assumed. - -## Dependencies - -- R187 provides the working single-node Docker profile, monitor-owned process - state, managed Web baseline, and Group 0-backed system metadata. R187 image - verification does not depend on this cross-mode cleanup. -- The existing Group 0 schema and `crowdb-kv-client` service APIs remain the - authority. If a live registration is absent, operations fail unavailable or - wait for registration; local launch policy never substitutes for it. -- The old mixed console file is unreleased. No on-disk compatibility promise or - migration tool is required, but bootstrap replay must not overwrite a - confirmed initialized cluster. - -## Acceptance - -- Given a Docker process restart and a bare-metal process restart, when runtime - state is queried, assert Docker PID/restart state comes from the monitor and - bare-metal launch policy stays local, while neither appears as Group 0 - topology. Invariant: deployment state is not sysdata. Integration test. -- Given a mixed legacy config and valid/invalid launch registries, when Web and - CLI start, assert only versioned process and launch inputs are accepted, no - local topology is restored, Docker rejects the registry, and inline secrets - or topology fields fail validation. Invariant: separated configuration. - Integration test. -- Given two bare-metal consoles and one ready Group 0, when each mutates racks, - nodes, disk groups, or disks and a write conflicts or loses its response, - assert both read one confirmed result and neither commits a local-first - topology change. Invariant: hardware authority. Integration test. -- Given CLI, Docker Web, and bare-metal Web with the same Group 0, when each - performs authenticated logical store/group/replica operations, assert one - shared result, correct fan-out/rollback, and no local logical copy. - Invariant: common logical authority. Integration test. -- Given missing, duplicated, or expired registrations and then a Group 0 - outage, when topology, endpoint, or deployment status is read, assert no - stale local endpoint or monitor snapshot is presented as authoritative. - Invariant: fail-closed discovery. Integration test. -- Given a crash before and after each bootstrap commit and before local - deletion, when startup resumes, assert it proves identity and committed - content, writes only safely missing records, and rejects conflict without - overwriting Group 0. Invariant: replay-safe cutover. Integration test. -- Given nonmember KV processes launched before Group 0 initialization, when - Group 0 is created and seed hints are propagated, assert each process - registers exactly one live node identity before logical operations use it. - Invariant: registration readiness. E2E test. -- Given a persisted S3 mini-cluster and a Group 0 outage, when it restarts or - tears down, assert local launch data cannot recreate or mask old cluster - topology. Invariant: no secondary authority. Integration test. -- Given the two deployment guides and a reader following bare-metal steps, - when the reader deploys KV, chunk services, and Iceberg or S3 access servers, - assert each layer has a verified setup and health check, the non-production - boundary is explicit, and no link targets the removed combined guide. - Invariant: deployment guidance follows its implementation. E2E test. -- Given a disposable container on a supported file-based core collector, when - a child or PID 1 crashes, assert the dump has private ownership, bounded - retention and cleanup, and resolves to source lines using exact-build symbols. - Assert ordinary logs disclose no dump contents or credentials and the - container does not change host-wide collector policy. Invariant: private, - bounded and reproducible crash diagnostics. E2E test. -- Given Apport, systemd-coredump or Docker Desktop collector policies, when - crash collection is attempted, assert the documented host export workflow - locates the dump or explicitly reports unsupported collection, without - claiming an absent data-volume core. Invariant: truthful collector boundary. - Integration test. - -Required gates: - -- `pixi run clean-env && pixi run test-console` -- `pixi run clean-env && pixi run test-console-ui` -- `pixi run test-monitor` -- `pixi run test-single-node-container` -- `pixi run rs-fmt-check` -- `pixi run rs-lint` diff --git a/doc/backlog/R92-chunkdb-in-chunk-gc.md b/doc/backlog/R92-chunkdb-in-chunk-gc.md index 933ff05b3..08b377e4b 100644 --- a/doc/backlog/R92-chunkdb-in-chunk-gc.md +++ b/doc/backlog/R92-chunkdb-in-chunk-gc.md @@ -10,19 +10,8 @@ to avoid global merge overhead. **Solution**: Implement in-chunk GC operations (ReclaimStrip, CollapseStrip, MergeStrips) for shared chunks. Add logical-to-physical offset mapping -to support GC while keeping chunk IDs stable. Add a ChunkDB orphan scanner for -chunks and shared ranges allocated by access uploads that crash or fail before -their complete file/object descriptor is published. R190 intentionally does not -write per-chunk catalog intents or upload-owner records on its write hot path. -The scanner must compare candidates with authoritative published S3 and Iceberg -references and reader protection before reclaiming, and must not infer orphan -status merely from age or a missing intermediate upload record. Account for -in-flight writers and delayed publication so physical ranges are never reused -while a writer or reader can still own them. Report candidate and reclaimed -bytes separately. Include Iceberg MPU Complete's frozen selection payload as -an authoritative reference while completion is in progress: it stores the -selected parts' exact chunk locations, which remain live even if a concurrent -UploadPart replaces the same part number before publication. After publication, -the immutable file descriptor is the authoritative reference. +to support GC while keeping chunk IDs stable. R95 owns the chunk-centered +orphan scan and qualified range deletion; this requirement provides the +in-chunk reclamation operations after R95 proves a range unreachable. **Scope**: Placeholder - detailed design to be refined before implementation. diff --git a/doc/backlog/R95-chunkdb-chunk-range-delete.md b/doc/backlog/R95-chunkdb-chunk-range-delete.md index 6d980c7a4..9839ad963 100644 --- a/doc/backlog/R95-chunkdb-chunk-range-delete.md +++ b/doc/backlog/R95-chunkdb-chunk-range-delete.md @@ -1,14 +1,82 @@ -### R95: chunkdb — Chunk Range Delete +### R95: chunkdb — Qualified chunk range deletion and orphan scan -**Problem**: Shared chunks need partial deletion capability for individual object deletion. Without range delete, entire shared chunks cannot be reclaimed efficiently. +## Problem -**Solution**: Define `DeleteChunkRange(chunk_id, offset, size)` in the chunkdb -protocol, client, and server dispatch now. The initial server implementation -returns an explicit not-implemented result without mutation. The full R95 -implementation adds range validation, used-bitmap management, idempotency, and -in-chunk GC integration before any caller may treat success as reclamation. +Shared chunks contain ranges owned by different objects or multipart parts. +Deleting a whole chunk for one unreachable range can erase live neighbors. +Uploads may also write a chunk and fail before their part or final object +reference is recorded. A cleanup queue populated by each MPU mutation would +add metadata writes to the upload path and still miss those pre-record crashes. -**Scope**: Placeholder - detailed design to be refined before implementation. +## Solution + +1. Complete `DeleteChunkRange(chunk_id, offset, size)` in the chunkdb protocol, + client and server. The existing stub remains explicitly not implemented + until range validation, used-bitmap updates, idempotency and in-chunk GC can + prove that the exact physical range is safe to retire. +2. Use a bounded, chunk-centered scanner to find old chunks and ranges whose + bytes have no live owner. Do not create per-MPU cleanup intents, candidate + records or a durable cleanup queue. Start with a configurable age threshold + of one day; a candidate must be older than the threshold after its last + write. Age alone never authorizes deletion. +3. Compare each candidate against published S3 objects, Iceberg file + descriptors, active multipart sessions and parts, frozen completion + selections, in-flight writers and reader protection. A lost part-publication + response or missing intermediate MPU record is not proof that a chunk is + unused. Refuse deletion when reference or reader state cannot be confirmed. +4. Keep S3 MPU session, current-part and immutable part-generation keys under + a common upload prefix in Chunk-KV so the scanner can enumerate one upload's + references with a bounded prefix scan. The S3 multipart authority owns that + key layout and the metadata-only Abort/expiry transition; old part + generations remain available until the scanner proves their bytes are + unreachable. An aborted or expired upload becomes a candidate only after + the age gate and reference checks. +5. Recheck chunk identity, layout generation and exact range against current + authority immediately before reclaim. Treat lost delete replies + idempotently and keep an in-memory scan cursor and bounded work budget. + Restart may rescan old chunks. Report examined, + eligible, deferred and reclaimed bytes without per-upload metric labels. + +## Dependencies + +- R92 supplies in-chunk strip reclamation after R95 qualifies dead ranges. +- The S3 multipart authority supplies grouped MPU keys and durable session, + part and completion references. The scanner also recognizes Iceberg's frozen + MPU selection. +- Reader protection and published generation references must be queryable + before physical deletion is enabled; R168 may use the qualified range-delete + interface for ordinary S3 object deletion. + +## Acceptance + +- Given a shared chunk with live and unreachable ranges, when the scanner + evaluates a candidate older than one day, assert only the exact unreachable + range is passed to `DeleteChunkRange`. Invariant: a live neighbor is never + reclaimed. E2E test. +- Given an MPU part that was replaced, aborted or expired, when its grouped KV + prefix is scanned, assert old locations remain protected by any active or + frozen selection and become eligible only after the age and reference + checks. Invariant: no extra MPU cleanup record is required. Integration test. +- Given a write that crashed before part metadata publication, when the chunk + scan runs, assert it defers the chunk until the age gate and in-flight writer + proof settle, then discovers the orphan without a part record. Invariant: + missing upload metadata alone cannot reclaim data. Integration test. +- Given a current reader, changed layout generation or uncertain authority, + when range deletion is attempted, assert no bytes are reused. After the + reader releases and identity is confirmed, repeat the same delete through a + lost reply and assert one idempotent result. Invariant: reclamation is fenced + by current chunk and reader authority. Integration test. +- Given a large backlog and foreground load, when scanning runs, assert it + respects item, byte and time budgets and reports deferred/reclaimed progress. + Invariant: cleanup cannot monopolize the data path. Integration test. + +Required gates: + +- `pixi run -- cargo test -p crowdb-chunkdb --all-targets` +- `pixi run -- cargo test -p crowdb-chunk-client --all-targets` +- `pixi run -- cargo test -p crowdb-access-s3 --all-targets` +- `pixi run -- cargo fmt --all -- --check` +- `pixi run rs-lint` diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index 4ff88de49..4f57be269 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -37,12 +37,9 @@ requirement is implemented. ### Planned — S3 data access service R152–R166 delivered the limited basic S3 service, including the restart -acceptance baseline. R167–R169 defer multipart upload and shared-storage GC -without blocking basic large-object deletion. R170 separately adds optional -cuObject/RDMA acceleration after the TCP baseline is correct and measured. -- **[R167](R167-s3-multipart-upload.md)** — multipart upload — Area: access - server / S3 — **Deferred.** Add durable part state, atomic completion, cleanup, - and multipart integrity after the basic milestone stabilizes. +acceptance baseline. Multipart upload is available; R168–R169 defer +shared-storage GC without blocking basic large-object deletion. R170 adds +optional cuObject/RDMA acceleration after the TCP baseline is correct and measured. - **[R168](R168-s3-shared-object-reclamation.md)** — shared small-object reclamation — Area: access server / S3 / chunkdb — **Deferred on R95.** Turn exact pending shared ranges into qualified, restart-safe physical deletion. @@ -80,14 +77,6 @@ Caches, selected ORC and container engine workflows remain separate. engine and optional ingest scenarios against the single-node image; publish only tested compatibility recipes. -### Planned — Console authority and deployment - -- **[R188](R188-console-group0-authority.md)** — Group 0 authority and - deployment configuration cleanup — Area: console / CLI / KV — Separate bare-metal launch policy from - cluster sysdata, remove the mixed local topology fallback, and finish - cross-mode console consistency without moving Docker process state into - Group 0. - ### High Priority - **[R103](R103-chunkdb-range-migration.md)** — chunkdb range ownership diff --git a/doc/design/access-server/s3/design-crowdb-access-s3.md b/doc/design/access-server/s3/design-crowdb-access-s3.md index ffaf486ff..20a9107e8 100644 --- a/doc/design/access-server/s3/design-crowdb-access-s3.md +++ b/doc/design/access-server/s3/design-crowdb-access-s3.md @@ -73,6 +73,27 @@ reused. Listings are ordered and continuation-safe within their documented consistency model. Continuation state is opaque and bound to the original request scope. +Multipart uploads keep a durable session, current part pointers, and immutable +part generations under one upload prefix. Replacing a part number advances its +generation while retaining the previous generation as reference evidence for +the chunk reclamation scan. The upload session records admission bounds, part +accounting, expiration, and completion state. Its pending part mutation is a +durable reservation: a writer stores the immutable generation, reserves the +pointer update with a session compare-and-swap, publishes the pointer, then +clears the reservation. A later request can finish an interrupted reservation. +Completion cannot freeze while one is pending. + +Completion validates the ordered selected part numbers, raw MD5 values, +minimum nonfinal size, and current generations. It freezes the selection under +the session compare-and-swap, composes chunk locations with adjusted logical +offsets, and publishes one object record through a predecessor-fenced +object-key mutation. Publication checks the current pointers and immutable +generation digests again. It never reads or concatenates part bytes. The +multipart ETag is the hexadecimal MD5 of the selected raw part MD5 values in +order, followed by the part count suffix. Abort and expiry mark the session +terminal; physical chunk reclamation follows the ordinary reference and age +checks. + ## 5. Relationship to other access models Iceberg is not implemented as special objects in the S3 namespace. Its catalog, @@ -99,3 +120,6 @@ native topology access remain Dataset semantics. reclaim Iceberg or Dataset authority. - **S3-I7 — Transport equivalence:** ordinary HTTP and accelerated transfer produce the same S3 range, integrity, publication, and error outcome. +- **S3-I8 — Multipart selection fence:** a part pointer cannot advance after + completion freezes its selection. An interrupted pointer reservation is + settled before completion or abort proceeds. diff --git a/doc/design/console/design-crowdb-console.md b/doc/design/console/design-crowdb-console.md index 3f927d880..4aaaf9548 100644 --- a/doc/design/console/design-crowdb-console.md +++ b/doc/design/console/design-crowdb-console.md @@ -21,10 +21,10 @@ design is detailed in the sub-design `design-crowdb-console-ui.md`. - [3.2 Logical (usage) view](#32-logical-usage-view) - [3.3 Source of truth and freshness](#33-source-of-truth-and-freshness) - [3.4 Design decisions](#34-design-decisions) -- [4. Console Backend Persistence and Monitor Task](#4-console-backend-persistence-and-monitor-task) - - [4.1 Persisted state (config file)](#41-persisted-state-config-file) - - [4.2 Monitor task](#42-monitor-task) - - [4.3 Persistent Cluster Config](#43-persistent-cluster-config) +- [4. Console Configuration and Authority](#4-console-configuration-and-authority) + - [4.1 Separated local configuration](#41-separated-local-configuration) + - [4.2 Runtime observation](#42-runtime-observation) + - [4.3 Group 0 authority](#43-group-0-authority) - [4.4 Local runtime namespace](#44-local-runtime-namespace) - [5. Node Access Model](#5-node-access-model) - [5.1 Two transports per node](#51-two-transports-per-node) @@ -32,7 +32,7 @@ design is detailed in the sub-design `design-crowdb-console-ui.md`. - [5.3 Process lifecycle (deploy / start / stop)](#53-process-lifecycle-deploy--start--stop) - [6. Web UI Backend (Axum)](#6-web-ui-backend-axum) - [6.1 Design Rules](#61-design-rules) - - [6.2 Recursive reads (?recursive=)](#62-recursive-reads-recursivedepth) + - [6.2 In-process test API](#62-in-process-test-api) - [6.3 Orchestration semantics](#63-orchestration-semantics) - [6.4 Resolution rules](#64-resolution-rules) - [6.5 Frontend contract](#65-frontend-contract) @@ -47,9 +47,8 @@ design is detailed in the sub-design `design-crowdb-console-ui.md`. - [7.8 S3 mini-clusters and benchmarks](#78-s3-mini-clusters-and-benchmarks) - [8. Error Model and Operation Logging](#8-error-model-and-operation-logging) - [9. Observability](#9-observability) -- [10. Open Questions](#10-open-questions) -- [11. Sysdata sync — rack/node/disk-group/disk handlers](#11-sysdata-sync--racknodedisk-groupdisk-handlers) -- [12. Cluster reset](#12-cluster-reset) +- [10. Hardware mutations](#10-hardware-mutations) +- [11. Cluster teardown and verification](#11-cluster-teardown-and-verification) ## 1. Goals and Non-Goals @@ -60,8 +59,8 @@ design is detailed in the sub-design `design-crowdb-console-ui.md`. ### Non-Goals - Bypassing `crowdb-kv-server` to talk to Paxos / WAL / storage internals. -- Authentication, authorization, multi-tenancy, audit logging. -- Persisting console state beyond local config files. +- Multi-tenancy and a general audit-log service. +- Making a console-local file authoritative for cluster topology. Local deployments use one stable directory per logical server below their runtime namespace. Each server owns its `data/`, `config/`, `log/`, and @@ -102,15 +101,12 @@ path through the shared `ops` module: - **Web**: `user → crowdb-web (Axum) → shared (ops module) → group-0 sysdata + crowdb-kv-server` - **CLI**: `user → crowdb-cli → shared (ops module) → group-0 sysdata + crowdb-kv-server mgmt` -Both frontends build an `OpContext` and call the same `ops::*` -functions. The CLI builds one per invocation from `--system-ip` / -`--system-port`; either endpoint may name any system-group node because the -client discovers the leader. The web backend builds one per request via -`AppState::op_context()`, sharing the cached `CrowdbKvClient` -(topology cache + connection pool) and snapshotting the persisted -`ConsoleConfig`. Mutations inside `ops::*` update the per-request -`OpContext` config snapshot; the handler writes the mutated config -back to `AppState.config` + persists to TOML after `ops::*` returns. +Both frontends build an `OpContext` with Group 0 discovery seeds. The CLI +uses `--system-ip` and `--system-port`; production Web uses a versioned process +configuration. Both read confirmed hardware and logical records through Group 0 +and resolve node management endpoints from live registration. The process +launch registry is local to each bare-metal console. Docker process state is +owned by `crowdb-monitor`. The CLI talks directly to group-0 system metadata via `CrowdbSysmdClient` and to individual `crowdb-kv-server` management @@ -129,8 +125,8 @@ call. ┌───────────────┐ │ shared │ (business logic: │ (lib crate) │ ops module, - └──────┬────────┘ monitor cache, - │ leader resolution, + └──────┬────────┘ leader discovery, + │ registry controls, ┌──────────────┼──────────────┐ SSH session pool) ▼ ▼ ▼ HTTP crowdb-rpc SSH @@ -151,9 +147,10 @@ call. - Both frontends build an `OpContext` and call `shared`'s `ops` module directly — the CLI from `--system-ip` / `--system-port` global flags, the web backend via `AppState::op_context()` (sharing the - cached `CrowdbKvClient` + snapshotting `ConsoleConfig`). -- Both frontends share the same `shared` entry points, so any feature - is reachable from both surfaces by construction. + cached `CrowdbKvClient` and Group 0 management seeds). +- Shared hardware and logical operations provide the same authority and + conditional publication rules to both frontends. Deployment-mode policy + controls which process and hardware mutations Web exposes. ## 3. Data Model @@ -184,23 +181,25 @@ Identity is the parent chain Rooted at **Cluster → Store → Group → Replica…** with a unified replica list (no local/remote split; each replica carries a `node_id`). This is the view that KV traffic, leader resolution, and routine cluster -operations use. The web backend is the only component that needs to -translate logical ids into upstream `(node_id, mgmt_url, rpc_url)` -tuples; the SPA and the CLI never see those. +operations use. Shared operations translate logical IDs into confirmed +membership and live endpoints for both frontends. Identity is `(store_id[, group_id[, replica_id]])`. ### 3.3 Source of truth and freshness -- **Persisted (config file, see §4):** rack/node entries and the - *intended* server deployment record (host, ports, binary path). - These survive restart. -- **Live (rebuilt on every console start):** process state, health, - per-node store/group/replica state, leader hints. The monitor task - (§4) pings each node and fetches per-node state; the logical view is - derived by aggregating those reports. -- **No `ClusterSnapshot` polling endpoint.** The SPA queries - per-resource live endpoints, all served from the monitor cache. +- **Group 0:** rack and node identity, nonsecret SSH connection settings and + credential references, disk hierarchy, bindings, KV stores, groups, + replicas, and service registration. +- **Local process inputs:** a versioned Web process configuration, bare-metal + launch registry, and per-console secret store. Neither topology nor inline + SSH secrets are accepted in the launch registry. +- **Live state:** management health, process identity, and current endpoints. + Docker reads monitor-owned process state; bare-metal process identity is + checked by `LaunchRuntime`. A stopped process is not a live registration. +- **Unavailable authority:** missing or ambiguous registration and Group 0 + outages are reported as unavailable. The monitor and launch registry do not + supply fallback topology. ### 3.4 Design decisions @@ -216,121 +215,49 @@ Identity is `(store_id[, group_id[, replica_id]])`. where possible; the console-side wrapper adds the `node_id` projection that the per-server protocol does not encode. -## 4. Console Backend Persistence and Monitor Task - -### 4.1 Persisted state (config file) - -- Single internal TOML file: - `.crowdb-runtime/persistent/console/crowdb-kv.db.toml`. It is CLI state, not - a user-facing command option. -- Contents: - - `rack` / `node` entries (id, rack_id, host, SSH creds). - - Optional per-node server deployment record: management endpoint, - rpc endpoint, and binary/config path as implementation evolves. - This records the operator's intended deployment target, not - authoritative live state. -- **Plaintext** SSH credentials are acceptable for v1 (internal demo); - a single `ConsoleConfig` struct is the only place that reads / - writes the file, so a future move to OS keychain or libsodium - sealed-box does not touch any caller. -- **Never persisted:** live process state, per-node store/group/ - replica state, leader hints, health flags. These are rebuilt on - every console start. - -### 4.2 Monitor task - -On startup, after loading the rack/node table, `shared` spawns a -long-running monitor task that owns the live cache: - -1. **Ping loop** — every `monitor.ping_interval` (default 2 s), the - task probes each node's `/health` over HTTP (and SSH liveness on - demand for the lifecycle API). It updates `NodeHealth` and - `ProcState` in the cache. -2. **Monitor refresh** — for every node observed `Up`, the task - calls the server's topology-report API to fetch `NodeStore` / - `NodeGroup` data (per-node store, group, local replica, remote - list). The aggregated `StoreView` / `GroupView` / `ReplicaView` - needed by the logical API are derived from these per-node reports. -3. **Event-driven refresh** — every successful mutation through - `shared` (deploy, store create, group create, replica add/remove) - triggers an immediate refresh for the affected nodes so the next - read reflects the change without waiting for the next ping tick. -4. **Cache reads are non-blocking.** API handlers read the most - recent cached value; they do not issue an upstream RPC per - request. A handler that needs a stronger guarantee ("force fresh") - can request an inline refresh, but that is the exception. - -### 4.3 Persistent Cluster Config - -**Problem**: The TOML config file is a single point of failure. Losing -the console host loses the full topology. Per-node server config is also -not persisted independently; a node restart relies on the console to -re-push topology. - -**Solution**: A designated Paxos group, **system group (store 0, -group 0)**, stores the full cluster topology as regular KV entries. -Since it is a Paxos group, the topology is replicated and HA by the -same mechanism that protects user data. No external coordinator -needed. This is the standard industry pattern (closest -analog: CockroachDB system ranges). - -- **Two-phase bootstrap**: - - Phase 1: Console TOML is source of truth (existing behavior). - - Phase 2: `HardwareClient` writes hardware hierarchy (racks, nodes) - and `KVClusterMetaClient` writes KV-cluster topology (stores, - groups, replicas) into group 0 via text-path keys with JSON - values. No readiness flag. diskdb's sync loop treats empty group 0 - as "nothing assigned yet" and retries. - - Console restart: two-way fallback. Group 0 missing → TOML mode; - group 0 exists → group 0 authoritative. - -The TOML file remains available for the whole local-deployment lifecycle. -Group-0 initialization does not make it disposable: subsequent CLI processes -use it to find endpoints and tracked process IDs for status, clean, restart, -and destroy operations. Regression runs keep `console.toml` at the retained -run root after teardown as diagnostic state; it is not stored inside a single -command's invocation directory. - -- **Group-0 sysdata schema** (text-path keys, JSON values): - - `/hw/rack/` — rack metadata (`RackValue`) - - `/hw/node//` — node metadata (`NodeValue`) - - `/hw/dg///` — disk-group metadata - - `/hw/disk////` — disk metadata - - `/hw/owner///` — ownership map - - `/hw/bind///` — bind map - - `/kv/store/` — store metadata (`StoreValue`) - - `/kv/group//` — group metadata (`GroupValue`) - - `/kv/replica///` — replica metadata - - `/srv//` — service registry instances - -- **Per-node config cache** (`conf/node-config.json`): Local cache - derived from the system group. On startup: load cache → create - stores/groups → replay WAL → reconcile with group 0 KV. If cache is - lost, node queries group 0 to rebuild it. - -- **Divergence reconciliation**: On node startup, if group 0 is - reachable and finalized, compare local cache against group 0 KV. - Create missing stores/groups, remove stale ones. If group 0 not - reachable, boot from local cache only (deferred). - -- **Cluster init flow**: `POST /api/cluster/init` on the console - orchestrates: calls `POST /system/init` on selected nodes, wires - remotes for multi-node, persists topology in console config, then - writes hardware + KV-cluster topology into group 0 via - `HardwareClient` + `KVClusterMetaClient`. Data store/group creation - is blocked (`409`) until cluster is initialized. - -- **Management API endpoints** (on `crowdb-kv-server`, internal — only - called by `crowdb-kv-client`'s `KVClusterAdmin`): - - `POST /system/init` — bootstrap store 0 + group 0 on this node - - Lifecycle: `add_store`, `remove_store`, `add_group`, - `remove_group`, `add_remote_replicas`, `remove_remote_replica`, - `step_down`, `join_group_via_snapshot`, `flush_group` - - Query: `GET /topology` (export), `GET /health`, `GET /metrics` - -- **Group 0 membership evolution**: Reuses shipped Model B - reconfiguration (direct HTTP mutation + `membership_epoch` fence). - No new consensus primitive required. +## 4. Console Configuration and Authority + +### 4.1 Separated local configuration + +`WebProcessConfig` contains the listener, Group 0 management seeds, UI and log +paths, deployment mode, and (for Docker) the monitor status path. Production +Web requires this versioned input. Docker rejects a launch registry. + +`LaunchRegistry` contains bare-metal process policy: service, node, host, +binary, service config, workspace and auto-start setting. Runtime PID and +start-time identity are retained separately by `LaunchRuntime`. SSH credential +reference IDs are read from Group 0 and resolved against each console's local +secret store. Group 0 never contains private keys, passwords, PIDs, images or +container IDs. + +The `ConsoleConfig` struct is an ephemeral operation input for bootstrap and +local development. It has no file parser or writer. A sealed `BootstrapIntent` +retains pre-Group-0 identity across interruption and is deleted only after all +committed records are verified. + +### 4.2 Runtime observation + +The production `/api/preview` snapshot reads Group 0 hardware and logical +records and validates live service registrations. Docker overlays monitor +process status, while bare-metal Web uses its local launch runtime. Missing +monitor status makes Docker runtime observation unavailable. The in-process +Web test router keeps a monitor cache for fixture orchestration; that cache is +not production topology authority. + +### 4.3 Group 0 authority + +System group (store 0, group 0) replicates hardware and KV-cluster metadata. +Bootstrap initializes selected KV members, wires peers and conditionally +publishes the rack, node, store, group and replica records. A retry compares +sealed identity and already committed content, writes only missing records, +and rejects conflicting content. Nonmember KV processes receive Group 0 seeds +and must register exactly one live identity before logical operations use them. + +The metadata namespaces are `/hw/rack`, `/hw/node`, `/hw/dg`, `/hw/disk`, +`/hw/owner`, `/hw/bind`, `/kv/store`, `/kv/group`, `/kv/replica`, and `/srv`. +Logical mutations confirm all node-side steps before publishing membership; +conditional writes and confirmed reads reconcile a lost response. A local +launch or monitor record never substitutes for a missing Group 0 result. ### 4.4 Local runtime namespace @@ -370,117 +297,44 @@ namespaces. - Default host: `127.0.0.1` with the current OS user. - Pre-flight: every operation calls `ssh::probe(node)` which performs a real handshake before any side-effecting work. Failure surfaces as `NodeUnreachable { node_id, reason }`. -**SSH credential storage lifecycle** — two phases: - -- **Bootstrap phase** (before group 0 exists) — SSH creds are stored - in the shared TOML config file below - `.crowdb-runtime/persistent/console/` - (via `TomlFileEngine::default_path()` in - `lib/crowdb-console-shared/src/config.rs`). This file stores - rack/node/server/store/group/disk-group/disk entries, with SSH creds - in `NodeEntry` (`ssh_user`, `ssh_key`, `ssh_password`). The CLI and - UI share the same `ConsoleConfig` + `TomlFileEngine` flow — `cluster - rack add` / `cluster node add` write to this file, `kv server deploy` - reads SSH creds from it. No separate CLI-only config file. -- **Steady-state phase** (after group 0 exists) — SSH creds are moved - into group-0 sysdata, encrypted with a default key. Subsequent `kv - server deploy` calls read creds from group-0 sysdata via - `KVClusterMetaClient`. The TOML file is no longer the source of truth - for SSH creds; group 0 is. The TOML file remains as a local cache / - bootstrap fallback. - +**SSH credential boundary:** Group 0 stores only the SSH user, port and +credential reference associated with a node. Each bare-metal console resolves +the reference in its own local secret store. Bootstrap intent rejects inline +private keys and passwords; the launch registry accepts references only. ### 5.3 Process lifecycle (deploy / start / stop) -**SSH path** (`ssh_user` non-empty): -1. SSH into node (`russh` crate, pure Rust async). -2. `nohup crowdb-kv-server --management-addr 127.0.0.1 --management-port

--ports &`; - capture pid via `echo $!`; record in the persisted node server entry. -3. Health-check via the new server's HTTP `/health` until ready or timeout (10 s). +`LaunchRuntime` uses the validated launch registry for local or SSH process +start, restart, stop and readiness checks. It records PID plus process start +time as local runtime identity and refuses to signal an unrelated process. +Auto-start policy is reconciled on Web startup and reload. A successful process +launch is not a substitute for a Group 0 service registration. Docker delegates +child recovery and status to `crowdb-monitor`. -**Local-fork path** (`ssh_user` empty, for tests/dev on `127.0.0.1`): -1. `tokio::process::Command::new(crowdb-kv-server)` with the same args. -2. Stage the binary into the node's stable service directory in the runtime - namespace. -3. Detach the child (do not kill on drop); track the pid. -4. Health-check via `/health`. - -Binary resolution: `$CROWDB_KV_SERVER_BIN` → sibling of current executable → -`$PATH` lookup for `crowdb-kv-server`. +## 6. Web UI Backend (Axum) -(Future: scp the binary to the remote host on first deploy and render -a config template. Not yet implemented. The SSH path assumes the -binary is already present on the remote host.) +### 6.1 Design Rules -`server deploy`, `server restart`, and `server stop` address a node. There is no separate -server id namespace in the console API. +Production Web uses the managed router and a versioned process configuration. +`/api/preview` combines confirmed Group 0 records, live registration, and the +mode-specific process view. `/api/stores/...` provides authenticated logical +mutations through shared operations in both modes. Bare-metal Web additionally +exposes rack, node, disk-group and disk reads and authenticated mutations, plus +registry-backed launch controls and bootstrap. Docker Web does not expose +hardware or process mutation routes. Unknown managed API routes report +unavailable rather than entering an in-memory topology path. -## 6. Web UI Backend (Axum) +A mutation is accepted only after the required node-side steps and Group 0 +publication are confirmed. Authenticated management routes use a bearer token. +The SPA calls the Axum backend; it does not talk directly to KV management +endpoints. -### 6.1 Design Rules +### 6.2 In-process test API -The console-facing API is split along the **two hierarchy views** -defined in §3, and every route lives under exactly one of them. Every -handler builds an `OpContext` via `AppState::op_context()` and -delegates to the matching `ops::*` function — the web backend no -longer hand-rolls orchestration logic (fan-out, rollback, sysdata -sync). The `ops` module owns all multi-step logic; the handler only -parses input, calls `ops::*`, writes back config, and renders output. - -**R1. Two URL trees, one per hierarchy.** -- `/api/racks/...` and `/api/nodes/...` form the **physical** tree. - Every resource is addressed by its parent chain. -- `/api/stores/...` forms the **logical** tree. KV traffic and - cluster-wide operations live here, addressed by - `(store_id[, group_id[, replica_id]])`. Logical-tree responses still - carry `node_id` on every entry so a caller can see placement without - a physical-tree query; only the **path** is node-free. -- A route never crosses trees. - -**R2. Logical reads aggregate; physical reads are per-node.** -The same store, observed through the two trees, returns different -shapes: aggregated `StoreView` vs. that node's local `NodeStore`. -This is how the operator inspects "is the cluster consistent?" vs. -"what does this one node think it has?". - -**R3. Logical writes orchestrate; physical writes act on one node.** -A logical write declares *intent*; the `ops` function fans out -per-node calls and rolls back on partial failure. A physical write -is the low-level primitive. It touches exactly that node, never fans -out. Logical writes are implemented on top of physical primitives. - -**R4. No `server_id` namespace.** -Process lifecycle and reachability probes use -`/api/nodes/:node_id/server/...`. Node identity *is* server identity. - -**R5. `OpContext` per request.** -Each handler builds an `OpContext` from `AppState::op_context()`, -which shares the cached `Arc` (topology cache + -connection pool) and snapshots the persisted `ConsoleConfig`. After -`ops::*` returns, the handler writes the mutated config back to -`AppState.config` (short write-lock, no `await` inside) and persists -via the config engine. On error, the snapshot is discarded — -`AppState.config` is unchanged. - -> **Retired contracts (no compatibility shim):** `?server=` -> query parameter, `/api/servers/:sid/...`, -> `/api/openapi.json?server=`, `/api/cluster/snapshot`, -> `/api/swagger/...`, `/api/nodes/:id/openapi.json`. - -The full endpoint list is defined in the Axum route handlers and the -OpenAPI spec; this section covers design rules only. - -### 6.2 Recursive reads (`?recursive=`) - -Any `GET` in either tree accepts `?recursive=` to inline up to `n` -child levels in one response, avoiding O(N) follow-up requests for -UIs that render a whole sub-tree. `recursive=all` is a capped alias -(default max depth 8) intended for the SPA's initial render. - -Rules: read-only (mutations ignore it), depth counts child hops from -the addressed resource, each tree expands along its own hierarchy, KV -key/value payloads are never inlined, and all responses use the -monitor cache so `recursive` is cheap even at high depth. +The in-process Web router and `--test-mode` retain fixture orchestration for +browser and integration tests. Their recursive physical views and monitor cache +help exercise the UI, but are never selected by a production Web process. +They do not persist a topology file or provide a fallback for managed requests. ### 6.3 Orchestration semantics @@ -506,8 +360,8 @@ these rules: parents; an already absent node-side object permits retry. - **Idempotent retries.** A repeat of the same logical request must converge to the same state. -- **Cache refresh on success.** Every successful mutation triggers an - immediate monitor refresh for the affected nodes. +- **Read after write.** A mutation returns only after the required node-side + and Group 0 confirmation steps complete. ### 6.4 Resolution rules @@ -525,8 +379,8 @@ backend-facing contract here: - Bundle output is `app/crowdb-web/ui/dist/`; `crowdb-web` serves it via SPA fallback. -- The SPA polls per-resource live endpoints on a short interval. No - WebSocket/SSE. All reads are served from the monitor cache. +- The SPA polls the management API on a short interval. An unavailable + authority clears stale logical rows and is shown explicitly. - No `/api/cluster/snapshot` aggregate endpoint. ## 7. CLI Design @@ -557,108 +411,25 @@ this section covers design rules only. ### 7.1 Four-Domain Hierarchy -The CLI is split by service domain into four top-level groups, each -cohesive and focused: - -- **`cluster`** (alias `cls`) — hardware topology (rack, node, - disk-group, disk, including runtime hardware state via - `set-status`) + cluster-level ops (init, reset, clean, status, - topology). `disk-group` and `disk` live here, not under `chunk`, - because they are hardware topology concepts — physical disks grouped - into disk-groups on nodes in racks. The `set-status` / - `set-dg-status` verbs are executed through the diskdb service API, - but the CLI verb belongs under `cluster` because it changes hardware - topology state, not chunk service state. `chunk diskdb` owns only the - diskdb service lifecycle and maintenance (scan/recalc/compact/ - rebuild). -- **`kv`** — KV layer: `kv server` (crowdb-kv-server lifecycle), - `kv store` / `kv group` / `kv replica` (logical concepts), `kv put` - / `get` / `delete` / `scan` / `snapshot` (data-plane). The verb - distinguishes management from data-plane; no `kv` prefix needed on - resource names. -- **`chunk`** — chunk storage service cluster: `chunk diskdb` / - `chunk chunkdb` / `chunk diskio` (server lifecycle + maintenance) + - future chunk data-plane (`allocate` / `free` / `write` / `read` / - `gc`). diskdb (block allocator), chunkdb (chunk metadata), diskio - (disk I/O), and the chunk client lib compose the chunk storage - service cluster; the group name reflects the unified service, not - individual servers. Stubs pending implementation. -- **`bench`** — load injection per layer. - -The four-domain hierarchy is the **standard concept** across the -production system — not CLI-specific. The console UI (`crowdb-web`) -uses the same domain grouping for its navigation and operation -surfaces (see `design-crowdb-console-ui.md`). The operation logic -behind each verb lives in `crowdb-console-shared`'s `ops` module -(§2.2); both frontends call the same shared operations, so CLI and UI -behave identically. +`cluster` owns hardware metadata and bootstrap, clean, destroy and status. +`kv` owns KV server launch controls, logical store/group/replica operations +and KV data commands. `chunk` owns storage-service launch controls and +maintenance. `bench` owns workload runners. The CLI connects to Group 0 +directly and shares the authority operations with production Web. ### 7.2 Command Hierarchy -``` -crowdb-cli -│ -├── cluster (alias: cls) ← hardware topology + cluster-level ops -│ ├── init (--nodes; bootstraps group 0 — §7.3) -│ ├── reset (full teardown — §13) -│ ├── clean (wipe user data, keep metadata + group-0 — §7.4) -│ ├── status -│ ├── topology -│ ├── rack { add, remove, list } -│ ├── node { add, remove, list, ping } -│ ├── disk-group { add, remove, list, set-status } -│ └── disk { add, remove, list, set-status } -│ -├── kv ← KV layer: server + logical concepts + data-plane -│ ├── server { deploy, restart (alias start), stop, delete, list } (delete — §7.5) -│ ├── store { add, remove, list, inspect } -│ ├── group { add, remove, list, inspect } -│ ├── replica { add, remove } -│ ├── put / get / delete / scan -│ └── snapshot { create, list, scan, release } -│ -├── chunk ← chunk storage service cluster (stubs) -│ ├── diskdb { deploy, restart, stop, delete, list, usage, -│ │ scan-status, scan, recalc, compact, rebuild } -│ ├── chunkdb { deploy, restart, stop, delete, list } (future) -│ ├── diskio { deploy, restart, stop, delete, list } (future) -│ └── allocate / free / write / read / gc (future data-plane) -│ -└── bench ← load injection - ├── kv { read, write, scan, mix } - ├── rpc - ├── diskdb { allocate, mix } (future) - ├── chunkdb { allocate, mix } (future) - └── chunk { write, read, mix } (future) -``` - -**Three layers max** — `crowdb-cli ` -(e.g. `kv server deploy`, `kv store add`, `cluster rack list`). -Direct data-plane verbs are two layers (`kv put`, `chunk allocate`). - -**Verb vocabulary:** -- Resource CRUD: `add / remove / list / inspect`. -- Server lifecycle: `deploy / restart / stop / delete` — consistent - across `kv server`, `chunk diskdb`, `chunk chunkdb`, `chunk diskio`. - `start` is an alias of `restart`. Servers are deployed one-per-node - by default; `list` enumerates instances across all nodes. -- Data-plane: `put / get / delete / scan`. The API uses `scan` for - prefix-scan; `list` is management-only (enumerates resources, not - data), never data-plane. -- Hardware state: `set-status` on `cluster disk` / `cluster disk-group`. - -**Logical entity addressing**: store/group/replica/KV commands use -`--store` / `--group`; the backend resolves placement. Server -lifecycle uses `--node`. - -**Leaders are elected, not assigned.** `kv group add` takes no -`--leader` flag; leadership is decided by Paxos election. +The `clap` command enums define the exact verbs and flags. Hardware and +logical commands use Group 0 for identity and membership. Process controls +require `--registry`; `cluster init` additionally requires sealed bootstrap +input for first creation. Development `local-deploy` runs a one-shot loopback +cluster and prints the Group 0 management seed for later CLI invocations. ### 7.3 `cluster init` — bootstrap special case -`cluster init` is the only command that runs before group 0 exists. -It takes `--nodes ` directly (not `--system-ip` / -`--system-port`) and bootstraps group-0/store-0 on those nodes via +`cluster init` requires `--registry` and a versioned `--bootstrap-file` +for first creation, or a sealed retry intent beside the registry. It takes +`--nodes ` and bootstraps group-0/store-0 on those nodes via direct node REST calls (the `POST /system/init` mechanism, §4.3), wires remotes, and writes the hardware + KV-cluster topology into group-0 sysdata. After `cluster init` completes, subsequent commands @@ -673,43 +444,17 @@ registrations rather than treating launch configuration as a live endpoint. ### 7.4 `cluster clean` — data wipe boundary -`cluster clean` wipes user-layer data across all storage services, -keeping services running and group 0 intact: - -- **KV user data** — remove all user stores + groups via the existing - store/group removal flow (cascades to replicas and on-disk WAL/tree - cleanup). group-0/store-0 preserved. -- **chunkdb metadata** — chunkdb stores metadata in CROWDB KV; cleaning - the chunkdb KV store (same as any KV store removal) wipes chunkdb - metadata. -- **diskio data** — diskio writes at positions it points to; later - writes overwrite old data. No explicit clean needed — new writes - supersede old data. -- **diskdb metadata + backing** — remove all diskdb metadata (clean the - diskdb group(s) in KV sysdata). For file-simulated disks, trim or - reset the backing file to reclaim space. For real devices, metadata - removal is sufficient (zones are reclaimed on next allocation). - -Services (`crowdb-kv-server`, `crowdb-diskdb`, `crowdb-chunkdb`, -`crowdb-diskio`) stay running. group-0 leadership continues — leaders -are elected, not assigned; as long as group-0 replicas survive, they -elect a leader. Topology (racks/nodes/disk-groups/disks) is preserved. - -For repeated full-stack benchmarks, `cluster clean --restart-services` -extends the boundary after the KV wipe. The console stops all locally deployed -DiskDB, DiskIO, and ChunkDB processes, then starts DiskDB and DiskIO before -ChunkDB with the same identities, endpoints, working directories, and launch -arguments. It waits for health, service registration, and ChunkDB range -bindings before returning. KV processes remain running so group 0 and hardware -topology survive. Suites with multiple data groups clean every group and request -the service restart on the final clean. A high-volume `mem-block` suite may use -`cluster destroy` followed by a fresh combined deployment for each case. This -process boundary releases the complete in-memory working set and prevents RSS -from accumulating across independent benchmark cases. - -Local auxiliary launch commands are retained in the run-root `console.toml`. -They are diagnostic lifecycle state, are removed with their server entry, and -are cleared by `cluster destroy`. +`cluster clean --store --group ` derives the target replica nodes +from confirmed Group 0 membership and resolves every live KV management +registration. It asks each target to wipe user data, then waits for a new +leader. Group 0 hardware, logical records, and process launch policy remain +intact. A missing group, registration, or acknowledgement fails the operation; +a local launch record cannot justify a wipe. + +`--restart-services` additionally restarts locally configured DiskIO, DiskDB +and ChunkDB processes in dependency order through `LaunchRuntime`. It requires +a validated launch registry before the wipe begins. KV processes stay running +so Group 0 remains available. ### 7.5 `kv server delete` — graceful + require-empty @@ -735,7 +480,9 @@ Verb distinction: - `bench kv ` runs KV workloads against a target store/group. `bench rpc` measures raw RPC transport throughput. -- Both are stubs pending re-wiring to the `ops` module. +- Bench discovery starts from the explicit Group 0 management seed and + resolves metrics hosts from confirmed replica membership and live + registrations. It does not load a console topology file. ### 7.7 Bench lifecycle verbs (deploy / prepare / run / teardown) @@ -787,7 +534,7 @@ path. The location, rather than the caller's global console configuration, is the cluster identity and recovery boundary: - a missing or empty location is initialized as a three-node cluster; -- a location containing `s3-mini-cluster.json` and `console.toml` is restarted +- a location containing `s3-mini-cluster.json` and versioned local launch state is restarted with the same service identities, endpoints, launch commands, KV/WAL/tree directories, and DiskIO files; - a non-empty location without the marker is rejected without modification. @@ -801,11 +548,11 @@ does not remove configuration or storage. `delete` stops the cluster, releases its persistent port claims, and removes the named location. `status` is read-only. -First start is transactional. The complete marker is published only after the -access endpoint is ready; failure stops the processes created by that -invocation. A later start archives an incomplete initialization directory next -to the selected location before retrying, preserving its logs for diagnosis -without treating it as a recoverable cluster. +First start seals bootstrap intent before publishing Group 0 and publishes the +complete marker only after the access endpoint is ready. Interrupted launch +steps replay from retained process and seed inputs; committed Group 0 content +is verified before a missing step is retried. A non-empty foreign directory is +rejected. Local state cannot reconstruct topology during a Group 0 outage. The mini topology is intentionally loopback and places its simulated nodes in one rack, so ChunkDB explicitly permits colocated fragments. This is not the @@ -821,12 +568,10 @@ be used for a non-loopback listener. Bucket and object commands take the same `--root`, discover the persisted endpoint, preserve S3 errors, and do not fall back to another mutation. -The durable record is deliberately small. `console.toml` retains service PIDs -and reproducible launch specifications; `s3-mini-cluster.json` retains only the -format version, storage profile, loopback endpoint, and non-secret tenant name. -The access master key is injected into a child only while it starts and is not -persisted in either record. Runtime liveness is derived from recorded PIDs, -not represented by additional compound cluster states. +The durable local record holds only versioned launch inputs, process +identities, bootstrap seeds, the storage profile, loopback endpoint and +nonsecret tenant name. It does not contain rack, node or logical topology. The +access master key is injected into a child only while it starts. `crowdb-cli bench s3` owns a separate, invocation-scoped memory profile. KV and WAL blocks use memory backing, DiskIO uses memory disks, and chunk-KV keeps its @@ -893,59 +638,23 @@ The following invariants apply: events; metric validation therefore checks metric sections and counters independently of auxiliary log size. -## 10. Open Questions - -- **SSH crate**: `russh` (decided). Defaults to `~/.ssh/*`; `(user, - password)` is an explicit alternative. -- **Frontend bundle**: built on demand; `npm run build` produces `dist/` - which the Axum server serves. The committed repo does not include - `web/dist/`. -- **Credentials storage**: plaintext TOML, accessed only through - `ConsoleConfig` so the source can change later without touching call - sites. -- **Multiple servers per node**: UI and console enforce one; lower - layers remain unrestricted. - -## 11. Sysdata sync — rack/node/disk-group/disk handlers - -Console add/remove handlers for racks, nodes, disk-groups, and disks -delegate to `ops::hardware::*`, which updates the `OpContext` config -snapshot first, then syncs group-0 sysdata via `ctx.sysmd()` -(`HardwareClient`). If group 0 is not yet initialized, the sysdata -sync is skipped — `cluster_init` Phase 5 writes the full hierarchy on -bootstrap. After `ops::*` returns, the handler writes the mutated -config back to `AppState.config` and persists to TOML. - -## 12. Cluster reset - -`cluster destroy` is full teardown. It is implemented in -`crowdb-console-shared`'s `ops::cluster::reset` as a hybrid operation -— group-0 discovery + direct node teardown — so the CLI no longer -depends on a `crowdb-web` endpoint. The flow: - -1. **Discovery** — connect to the system group (via `--system-ip` / - `--system-port`) to enumerate all resources: user stores/groups/ - replicas, diskdb/chunkdb/diskio instances, server entries, topology. -2. **Teardown in dependency order** — erase resources one by one: - remove user groups → user stores → clean group-0 sysdata (rack - cascade + store records + diskdb service unregister) → SIGTERM each - node's processes. -3. **Destroy group 0** — tear down group-0/store-0 itself (last, after - all user resources are gone). -4. **Delete topology** — remove all nodes and racks from - the persistent console configuration. -5. **Fast path** — if group 0 is not created (e.g. `cluster init` - failed or was never run), skip steps 1-3 and use the TOML config - info (rack/node entries) to clean up any stray processes and clear - the config. - -The `POST /internal/reset` endpoint on `crowdb-kv-server` remains for -UI use; the CLI implements its own teardown via the shared `ops` -module. When no KV servers are running, the RPC steps are skipped -(fast path for E2E test fixtures). - -The web backend exposes `POST /api/cluster/reset` (calls -`ops::cluster::reset`) and `POST /api/cluster/clean` (calls -`ops::cluster::clean` — removes orphaned sysdata entries from stopped -servers without full teardown). Both are reachable from the CLI and -the web UI. +## 10. Hardware mutations + +CLI and bare-metal Web use the shared Group 0 hardware operations. Rack, +node, disk-group and disk changes update parent and child records in one +conditional batch where membership changes. Matching retries are confirmed; +conflicting concurrent writes preserve the existing record. Docker Web does +not expose hardware or process mutations. + +## 11. Cluster teardown and verification + +`cluster destroy` requires the local launch registry. It reads confirmed Group +0 membership, removes user stores and groups through shared logical operations, +then removes the system group last. Only after metadata teardown succeeds does +it stop processes named by that console's launch registry. A failed or +unconfirmed step returns an error instead of deleting presumed local topology. + +There is no orphan-guessing reset command. A stopped or unreachable node does +not imply its membership should be deleted. `cluster clean` derives its +replica targets from Group 0, wipes each live target, and waits for a new +leader while preserving topology. diff --git a/doc/dev/crash_debugging.md b/doc/dev/crash_debugging.md new file mode 100644 index 000000000..68027c946 --- /dev/null +++ b/doc/dev/crash_debugging.md @@ -0,0 +1,168 @@ + + + +# Debugging a CROWDB Crash + +Use the core from the crashed process and the exact executable build that +produced it. Core files can contain credentials and user data. Keep them in a +private directory, do not attach them to ordinary logs or issues, and remove +them when the investigation is complete. + +## 1. Find the collector + +On the machine running Docker or a bare-metal service, inspect: + +```sh +cat /proc/sys/kernel/core_pattern +cat /proc/sys/fs/suid_dumpable +``` + +- A relative name such as `core.%e.%p.%t` writes in the crashing process's + working directory. The name must start with `core` for the single-node + container's retention rule to recognize it. +- A leading `|` sends the dump to a host-side program. Ubuntu Apport usually + writes a report under `/var/crash`; `apport-unpack REPORT.crash OUTPUT_DIR` + extracts its `CoreDump`. `systemd-coredump` uses `coredumpctl list` and + `coredumpctl --output=FILE dump`. A collector may reject a container crash. + If there is no report, there is no core to extract. +- An absolute file pattern uses the crashing process's mount namespace and + root. Check the actual destination and access policy on that host. + +The container's data volume does not override a host collector. During the +2026-09-29 Ubuntu development-host check, Apport received the crash but could +not resolve `/opt/crowdb/bin/crowdb-kv-server` in the host filesystem. +The image contains the executable; the failed lookup happens in Apport. No +CROWDB core was saved by that test. + +## 2. Configure a file-based collector on a dedicated development host + +This is a **host-wide change**, not a Docker setting. It replaces Apport's +automatic crash reports for all processes on that host. Other programs with +a nonzero core limit may write private core files in their own working +directories. Do not apply it to a shared or production host without an +operator decision. Run these commands on the Docker host, never inside the +container. Record the original `core_pattern` and Apport service state first. + +On an Ubuntu host where `apport.service` owns `core_pattern`, a persistent +file-based setup is: + +```sh +cat /proc/sys/kernel/core_pattern +systemctl is-enabled apport.service +sudo systemctl disable --now apport.service +printf 'kernel.core_pattern=core.%%e.%%p.%%t\nfs.suid_dumpable=0\n' | + sudo tee /etc/sysctl.d/99-crowdb-core.conf +sudo sysctl -w 'kernel.core_pattern=core.%e.%p.%t' +sudo sysctl -w fs.suid_dumpable=0 +cat /proc/sys/kernel/core_pattern +``` + +`apport.service` sets the pipe pattern when it starts, so a sysctl file alone +does not keep the file pattern after a restart. Verify `core_pattern` again +after reboot. `fs.suid_dumpable=0` permits an ordinary process to write a +relative core; executables with file capabilities may still be excluded. +The container currently gives file capabilities to `crowdb-iceberg` and +`crowdb-access-server` for low ports, so do not assume those two will produce +cores under this setting. Verify the particular crashed service. + +For a one-time investigation, stop Apport and apply the two `sysctl -w` +commands without creating the sysctl file or disabling the service. Restart +Apport after the investigation to restore its collector. For the persistent +setup above, restore the prior Ubuntu behavior with: + +```sh +sudo unlink /etc/sysctl.d/99-crowdb-core.conf +sudo systemctl enable --now apport.service +cat /proc/sys/kernel/core_pattern +``` + +If this host used another collector originally, restore its recorded service +state and exact original pattern instead of starting Apport. + +## 3. Bound and locate the core + +For the single-node container, add a nonzero per-process bound when running +Docker: + +```sh +--ulimit core=1073741824:1073741824 +``` + +The monitor and managed children run from the private mounted directory +`/opt/crowdb/data/crash` (mode `0700`). With a relative `core.*` pattern, a +dump lands there. The monitor keeps the newest regular `core` file after +startup or child recovery. Inspect the directory with `docker exec`; export +the selected core to a private host directory with `docker cp` for GDB. A +1 GiB limit can truncate a larger dump. A piped collector ignores this +`RLIMIT_CORE` bound and uses its own policy. + +For a bare-metal process, set a writable private working directory and a +nonzero core limit in the launcher. A shell launch can use +`ulimit -c 1048576` (KiB); a systemd unit can use `WorkingDirectory=` and +`LimitCORE=1G`. Verify the actual process limit in `/proc/PID/limits`. +There is no container monitor retention rule for bare-metal cores. + +## 4. Open a container core with exact-build symbols + +Use the image that ran the crashed process and its matching optional symbol +archive. From the CROWDB checkout: + +```sh +pixi run -- python tools/symbolize-container-core.py \ + --image 'docker.io/crowdb/crowdb-iceberg:' \ + --symbols '/private/crowdb-symbols--git--linux-amd64.tar.zst' \ + --binary crowdb-kv-server \ + --core /private/core.crowdb-kv-ser.PID.TIME +``` + +The tool checks the image revision, version and binary SHA-256 hashes against +the archive, places the `.debug` files beside the matching stripped ELF files, +and starts GDB with the image's shared libraries. GNU debuglinks let GDB load +the separate symbols. Replace `--binary` with the actual crashed CROWDB +binary. If the archive was not released, build the exact Git tag with +`CROWDB_PACKAGE_SYMBOLS=1 pixi run build-single-node-container` and package +its local `target/container-symbols` directory for the helper: + +```sh +pixi run -- bash -c 'tar -C target/container-symbols -cf - . | zstd -q -o /private/crowdb-symbols-local.tar.zst' +``` + +Use that archive with the image produced by the same build. A different +commit or build is not a safe substitute. + +For an interactive session, keep private copies of the same image's `bin/` +and `lib/`, put each matching `.debug` file beside its stripped binary or +library, and run: + +```sh +pixi run -- gdb -q /private/runtime/bin/crowdb-kv-server /private/core +(gdb) set solib-search-path /private/runtime/lib +(gdb) sharedlibrary +(gdb) set print frame-arguments none +(gdb) thread apply all bt +``` + +Do not use a newly built binary against an older core just because its version +string is unchanged. + +## 5. Open a bare-metal core + +Bare-metal deployment keeps the binary's debug information. Use the exact +binary and shared libraries that were running when the core was made; no +separate container symbol archive is needed: + +```sh +pixi run -- gdb -q /path/to/exact/crowdb-kv-server /private/core +(gdb) set solib-search-path /path/to/exact/lib +(gdb) sharedlibrary +(gdb) set print frame-arguments none +(gdb) thread apply all bt +``` + +If the executable or a shared library has been replaced since the crash, +recover the original build before trusting the stack. A core from one build +must not be interpreted with symbols from another. + +References: [Linux core dump rules](https://man7.org/linux/man-pages/man5/core.5.html), +[kernel `core_pattern`](https://docs.kernel.org/admin-guide/sysctl/kernel.html), +and [Ubuntu Apport](https://documentation.ubuntu.com/project/contributors/debugging/apport/). diff --git a/doc/doc_index.md b/doc/doc_index.md index 8c54e702c..f57984c7a 100644 --- a/doc/doc_index.md +++ b/doc/doc_index.md @@ -47,9 +47,10 @@ Temporary plans live under `doc/working/`; flow analyses live under | Doc | When to read | | ---------------------------- | ---------------------------------------------------------------------- | +| `doc/dev/crash_debugging.md` | Core collection, host setup, GDB, and exact-build symbols. | | `doc/dev/env_setup.md` | Benchmark commands, sentinels, prerequisites, and perf-counter setup. | | `doc/dev/hyper_fork.md` | Hyper fork branches, submodule, build, sync, validation, and recovery. | -| `tools/README.md` | Tool directories, Pixi task entry points, CI checks, and suite timing. | +| `tools/README.md` | Tool directories, Pixi task entry points, CI checks, and suite timing. | ## Project Files (repo root) diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md deleted file mode 100644 index ac0987caa..000000000 --- a/doc/working/plan-console-authority.md +++ /dev/null @@ -1,287 +0,0 @@ - - - -# Console Authority Plan - -Upstream: [R188](../backlog/R188-console-group0-authority.md). - -Goal: make Group 0 the shared CLI/Web authority while retaining only process -and launch inputs locally. - -Status: paused at the verified bootstrap-publication checkpoint by user request -to prioritize the single-node image and merge preparation. Remaining tasks below -are retained for resumption; this requirement is not complete. - -## Registration and acceptance failures - -- [x] **Stable registration across restart**: persist generated instance IDs - under the KV node's config root, reject changed explicit identities and - corruption, and prove restart replaces an unexpired old registration rather - than creating ambiguity. Diagnose the concurrent restart suite without - increasing election timeouts. Files: KV server startup/background identity, - discovery integration and Web incremental restart tests. -- [x] **Pre-bootstrap nonmember registration**: reproduce the missing live - registration for servers started before Group 0 but excluded from its member - set. Propagate discovery seeds after confirmed initialization, retain them - across restart as launch inputs, and wait for exactly one live identity before - declaring the bootstrap complete. Do not create local topology fallback or - restart processes into a different workspace. Files: KV server keepalive and - management modules, console shared cluster initialization and HTTP client, - focused integration tests, UI node-inspection and replica flows. -- [x] **Unavailable logical view**: preserve the explicit unavailable state - before Group 0 exists and during outages, clear stale logical rows, and avoid - treating an expected unavailable response as an unhandled browser exception. - Update the canvas navigation assertions to distinguish unavailable authority - from a confirmed empty cluster. Files: UI logical-tree data hook, KV panel, - shell/canvas/full-chain specs. - -## Common logical operations - -- [x] **Group and replica creation fan-out**: reject missing peer registrations, - missing peer endpoints and failed - remote wiring, roll back created local groups, and publish no Group 0 group - or replica records on these failures. Prove with real Group 0 and controlled - management endpoints. Files: shared `ops/kv_logical.rs` and - `tests/ops_logical_fanout_test.rs`. -- [x] **Logical deletion cleanup**: derive store hosts from both store and - replica membership, confirm every node deletion before removing authority, - and remove descendant records before parents. Test success, sibling - preservation, later replica hosts and node-side failure. Files: shared - `ops/kv_logical.rs`, `tests/ops_logical_delete_test.rs`. -- [x] **Replica creation cleanup**: clean a newly created target store when - local group creation fails, preserve pre-existing target groups, and report - incomplete rollback. Test injected creation and cleanup failures. -- [x] **Confirmed logical mutations**: reconcile lost responses through - confirmed authority, complete replica fan-out/rollback and delete cleanup; - preserve Group 0 membership when node-side deletion fails. Reuse the common - flow in CLI and both Web modes, with no local topology commit. - Conditional publication replaces overwrite writes for stores, groups and - replicas; test concurrent matching/conflicting records and a real RPC reply - dropped after commit. New groups and their initial replicas now use one - conditional batch; confirm the complete member set after a lost reply. - Reconcile failed delete responses with a confirmed absence read. - -## Configuration and hardware operations - -- [x] **Launch registry lifecycle**: wire `WebProcessConfig` and `LaunchRegistry` - into CLI and bare-metal Web deploy/restart paths. Consume binary, service - config, workspace, host and auto-start policy; retain PIDs only in runtime - state and resolve SSH credentials through references. Files: console shared - config/lifecycle, CLI startup, Web startup/state/lifecycle and tests. - First complete launch arguments/readiness inputs, shared local/SSH lifecycle - and runtime-only process identity. Then connect Web auto-start and CLI - deployment/restart callers before removing mixed persistence. - Shared primitives are implemented in `launch.rs` and its local/remote/runtime - modules: private PID/start-time records, idempotent start, referenced service - config and SSH keys, readiness checks, and failure cleanup. Local and real - SSH transport regressions pass, including a native KV launch and refusal to - adopt an unrelated healthy endpoint. Complete Console shared tests, fmt and - clippy pass. Logs: `/tmp/crowdb-launch-{shared,lint,fmt}.log`. - Web now loads and reconciles auto-start policy, exposes authenticated - start/restart/stop and runtime views, and reloads policy on each request. - CLI `--registry` deploy/start/restart/stop/delete uses the same runtime; - deletion checks confirmed replica membership before removing launch policy. - Native Web and CLI integration tests pass, including ignored legacy state, - idempotent start, changed restart identity, and policy edits without a Web - restart. Complete shared/CLI/Web regressions, fmt and clippy pass. - Logs: `/tmp/crowdb-launch-consumers-{full,lint-3,fmt}.log`. - Generic CLI `launch list/start/restart/stop` and chunk diskdb/chunkdb/diskio - deployment controls now share the same launch runtime. Process controls - work before Group 0; chunk service lists still read live registration. - Three-service lifecycle regressions and the complete CLI suite, fmt and - clippy pass. Logs: `/tmp/crowdb-chunk-launch-*.log`. Removal of legacy - startup/restore paths remains coupled to bootstrap cutover below. -- [ ] **Remove mixed persistence**: remove the unreleased `ConsoleConfig` - parser/writer, inline SSH secrets, topology restoration and fixtures after - the launch lifecycle and replay-safe bootstrap paths are wired. Preserve - bootstrap intent independently until verified cutover. Update CLI commands, - Web persistence and S3 mini-cluster callers together; no compatibility reader. -- [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through - shared Group 0 hardware operations; preserve conflicts and uncertain writes - without local-first commits. Docker keeps its hardware restrictions. -- [ ] **Authority-only reads**: replace local monitor/config topology and - endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or - expired registrations remain unavailable. - Versioned bare-metal snapshots now read Group 0 without requiring a Docker - monitor; Docker keeps its monitor requirement and overlay. Validate every - replica host as well as the store's original hosts. Real Group 0 regressions - cover missing, duplicate and expired registrations, recovery, and outage - without stale topology. Docker and launch-route regressions, fmt and clippy - pass. Logs: `/tmp/crowdb-bare-authority-*.log`. Legacy physical routes and - monitor refresh still remain for the mixed-config removal. -- [ ] **Replay-safe bootstrap cutover**: persist bootstrap identity, verify - committed records, write only safely missing content, reject conflicts and - delete topology intent after verified transfer. Clean/destroy use confirmed - authority. Audit S3 mini-cluster persistence against the same contract. - System initialization now confirms an existing replica's identity after a - conflict or lost response, preserves groups for retry after peer failures, - and requires every peer endpoint and remote-wiring request to succeed before - recording membership. Four focused failure cases, complete shared tests, - Web deploy/restart/migration suites, fmt and clippy pass. Logs: - `/tmp/crowdb-bootstrap-replay-*.log`. Durable intent/cutover remain pending. -- [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical - records, accept matching content without rewriting revisions, reject conflicts, - and conditionally create missing records. Reconcile uncertain writes with - confirmed reads; record local membership only after publication is confirmed. - Three real-authority regressions failed before the fix and now pass. Strict - publication exposed missing leader discovery in conditional KV writes: - explicit no-hint not-leader rejections now use the existing bounded retry - policy, while ambiguous dispatch still returns `OutcomeUnknown`. - -## Crash diagnostics follow-up - -Transferred from R187 by user request. This work remains pending while R188 -is paused; it does not block the single-node image requirement. - -- [ ] **Crash dump location and retention**: document and test how Linux - host `core_pattern`, Docker's core ulimit, and the non-root container affect - CROWDB child and PID 1 crashes. Cover a plain relative core-file pattern, - Ubuntu Apport, systemd-coredump, and Docker Desktop's Linux VM. Choose a - bounded, private location under the mounted `/opt/crowdb/data` volume where - the host permits file dumps; otherwise report the host collector location - and provide explicit setup guidance instead of claiming the volume contains - a core. Verify one disposable child crash end to end, retention/cleanup, - secret exposure, and symbolization against the exact binary build. Do not - change the host-wide `core_pattern` from inside the container. Files: - `container/single-node-container/{Dockerfile,entrypoint.sh,tests/**}`, - `container/crowdb-monitor/src/**`, - `container/single-node-container/README.md`. -## Documentation and completion - -- [ ] **Bare-metal documentation**: migrate verified KV, chunk and access - setup into dedicated deployment documentation, state the non-production - boundary, then fix links and remove obsolete combined material. Keep Docker - deployment notes independent. -- [ ] **Acceptance and cleanup**: run affected integration cases, full console - and UI suites, Rust fmt and lint; update the relevant permanent architecture, - then remove the requirement, backlog entry and this plan when complete. - -## Evidence - -- Bootstrap checkpoint passes complete KV client and Console shared/CLI/Web - suites, five affected browser lifecycle/full-chain cases (53.8s), Rust fmt - and workspace clippy. Logs: `/tmp/crowdb-cas-retry-{baseline,suite,lint}.log`, - `/tmp/crowdb-bootstrap-confirmed-{console,ui,lint}.log`. - Lost-response fixtures now advertise their RPC proxy through management - topology, so discovery refresh cannot bypass the injected reply loss. - -- Deletion reconciliation passes all six cases, including a real dropped - metadata reply for both store and group deletion. Complete Console shared, - affected Web migration/replica tests, fmt and clippy pass. - Logs: `/tmp/crowdb-delete-reconcile-*.log`. - -- Group publication baseline gives the group and initial replica different - committed revisions (3 and 4), exposing partial publication on interruption. - Conditional batch publication passes the shared-revision and lost-batch-reply - tests, full Console shared tests, and Web restart/migration/replica tests. - Orphan membership is rejected before local mutation; all three focused - regressions, fmt and clippy pass. Logs: `/tmp/crowdb-group-publication-*.log`. - -- Conditional publication baseline overwrites a competing store record and - reports success. Both matching and conflicting race tests pass after the - CAS change. A real RPC proxy discards the committed write reply; linearizable - confirmation succeeds and exactly one reply is dropped. Full Console shared - and CLI pass. Full Web passes after the restart-fixture correction below, - including all five concurrent restart cases. Rust fmt and clippy pass. - Logs: `/tmp/crowdb-publication-*.log`. -- The complete Console gate reaches a three-node restart failure: no complete - store view within 3s. Its persisted identities are stable; node logs show - repeated Group 0 elections and late registration, including election churn - before restart. This real-process case uses the paused-clock `test` profile - (5ms heartbeat, 30–60ms election) while all larger clusters use `e2e`. - Exact isolation passes in 4.43s; default-concurrency rerun passes in 13.33s, - and serial execution passes. Use the existing `e2e` fixture for the three-node - process case and retain its 3s acceptance assertion. Add the last HTTP - observation to store-wait failures, as already done for group waits. - Original logs remain in `restart-3n-1g-20260927-234030.467` under the ephemeral - Web E2E root; do not clean them during diagnosis. - -- Replica cleanup passes four regressions: failed group creation cleans its - newly created store, cleanup failure is explicit, existing replica hosts - are rejected before mutation, and automatic identity exhaustion returns - validation rather than panicking. Complete Console shared tests, Web replica - tests, fmt and clippy pass; helper extraction also passes all nine focused - creation/fan-out regressions. Logs: `/tmp/crowdb-replica-cleanup-*.log`. - -- Deletion baseline fails three of four cases: later replica hosts are skipped, - node-side failure reports success, and group deletion leaves replica records. - All four now pass, including idempotent node-side 404 and preservation of - sibling groups. Complete Console shared tests, affected Web migration/replica - tests, fmt and workspace clippy pass. Logs: `/tmp/crowdb-delete-*.log`. - -- Group fan-out baseline: both rejected remote wiring and missing peer endpoint - returned success. Both failure-injection cases now pass, and the complete - Console shared/Web gates passed for the group fix. Replica baseline adds - three failures: missing peer registration/address reports success, and a - missing new endpoint leaves its local group behind. Resolve all existing - peers before mutation and roll back the target when its endpoint is missing. - All five regressions pass, along with complete Console shared tests, affected - Web replica/migration tests, six browser store/reconfiguration flows, Rust - fmt and workspace clippy. Logs: `/tmp/crowdb-all-fanout-{shared,web,ui}.log`. - -- Complete Web integration gate passes after stable identity persistence. - Full Console UI passes all 86 component tests and 56 browser tests (4.6m), - including all five original failures. Logs: - `/tmp/crowdb-final-console-server.log`, `/tmp/crowdb-final-console-ui.log`. - -- Discovery integration passes duplicate seed submission, invalid origins, - exactly one live nonmember identity, no accidental membership, and restart - with persisted hints. Rust fmt and workspace clippy pass. -- Affected browser cases now pass: shell dialogs (11.9s), shell health (3.3s), - node inspection (8.3s), node cross-jump (2.7s), full-chain flow (4.8s), and - all three canvas cases (3.2s / 0.8s / 4.2s). The canvas assertion is scoped - to the main panel because the same unavailable text appears in a notification. -- Logical-tree hook regression passes confirmed reads, outage clearing, and - recovery. Full KV Server and Console shared-library gates pass. Complete - CLI/Web and browser gates remain pending for requirement completion. -- Full CLI passes. Web gate reached a failure in - `cluster_restart_incremental_test::restart_6node_2group_overlap`: after all - nodes restarted, group 11/1 did not converge to one leader within 3s. The - unchanged exact test passes alone in 13.27s. Preserve the original timeout; - compare the complete restart suite at default concurrency and serially, - without concurrent release compilation, before attributing the failure. - Logs: `/tmp/crowdb-discovery-console.log`, - `/tmp/crowdb-overlap-restart-isolated.log`. -- Generated registration identity changed across a normal restart in the - focused baseline (deterministic assertion failure). Persisting identity fixes - that case and the missed-unregister case; changed explicit IDs, changed node - IDs and malformed files fail closed. Full KV Server passes. The five restart - cases pass at default concurrency after the fix (13.55s); before the fix, - serial execution passed (44.73s) while default execution failed on different - groups. No election timeout or retry count changed. Full Web still remains. - -- Initial full Console UI baseline: 85 component tests pass; 51 browser tests - pass and five fail. Missing live registration affects node 203 in shell - replica creation and node 262 in node inspection. Two canvas assertions - expect an empty-store view before Group 0 exists; the full-chain flow records - logical-tree fetch errors during that same uninitialized phase. -- Isolated `12-cluster-node-inspect.spec.ts` reproduces HTTP 404 for node 262; - one test passes and one fails in 25.2 seconds. This is not solely an ordering - issue in the full browser suite. Keepalive uses its local bootstrap endpoint - when launched without seeds; cluster initialization currently waits only for - selected Group 0 members and does not propagate seeds to nonmembers. - -## Tests - -- Focused: KV server discovery/keepalive integration, console shared bootstrap - and operation tests, affected shell/node-inspection/canvas/full-chain specs. -- Full: `pixi run clean-env && pixi run test-console` and - `pixi run clean-env && pixi run test-console-ui`, sequentially. -- Crash diagnostics: monitor retention tests and disposable-container crash, - collector/export and exact-build source-line symbolization acceptance through - `pixi run test-monitor` and `pixi run test-single-node-container`. -- Style: `pixi run rs-fmt-check` and `pixi run rs-lint`. - -## Open Questions - -- **Crash collection and symbols:** the current host routes `core_pattern` to - Apport. A container-local file directory/ulimit cannot override that policy, - and changing the host-wide collector is outside container implementation. - Choose acceptance on a disposable Linux host with file-based core collection, - or certify and document a host-collector export workflow. Source-line symbol - distribution also needs a choice: bundle compressed CROWDB line tables and - adjust the measured image-size ceiling, or publish exact-build debug symbols - separately while retaining runtime function names. The existing all-dependency - experiment increased monitor size substantially; neither complete-image option - has yet been measured. Bounded volume retention and end-to-end source-line - symbolization remain incomplete, not claimed acceptance. diff --git a/doc/working/test.md b/doc/working/test.md index 196374ff6..e2647e54a 100644 --- a/doc/working/test.md +++ b/doc/working/test.md @@ -16,7 +16,7 @@ For test strategy, layer scope, and coverage details, see [`design/kv/design-cro ## Current CI Test Design -CI uses ten parallel jobs, grouped by runtime requirements. Component tasks in +Regular CI uses nine parallel jobs, grouped by runtime requirements. Component tasks in `pixi.toml` select library, binary, and integration test targets with `--tests`; benchmark targets are excluded. Group scripts under `tools/pixi-tasks/` define execution order. GitHub Actions calls those group @@ -24,16 +24,15 @@ tasks. See [tools/README.md](../../tools/README.md) for the tooling map. | Job | Group task | Coverage | | ------------- | --------------------------------------- | ------------------------------------------------ | -| Lint | `test-task-coverage`, fmt, clippy | Package assignments and reachable CI tasks | +| Lint | `check-ci-test-tasks`, fmt, clippy | Package assignments and reachable CI tasks | | CppTests | `test-cpp` | C++ and Rust FFI | | UnitTests | `test-unit` | Rust libraries, including `test-access-iceberg` | | ServerTests | `test-server` | Native services, access server and monitor | | S3E2E | `-e s3-e2e test-boto3-e2e` | Access S3, access server and 17 boto3 cases | | IcebergE2E | `-e iceberg-e2e test-iceberg-e2e` | PyIceberg, native storage, GC and crash recovery | -| IcebergSDK | `-e iceberg-e2e test-iceberg-sdk` | Official Java/Rust SDKs and pinned Apache RCK | +| IcebergSDK | `-e iceberg-e2e test-iceberg-sdk` | Official Java SDK and pinned Apache RCK | | ConsoleTests | `test-console` | Shared operations, CLI and Web | | UITests | `test-console-ui` | Vitest and real-backend Playwright | -| DockerPreview | `test-single-node-container` | Linux amd64 image smoke and container E2E | Subprocess suites run sequentially inside each job and clean disposable runtime state. Iceberg jobs use the pinned `iceberg-e2e` Pixi environment for Python, @@ -41,18 +40,27 @@ Maven and Java; Rust/native builds use the default environment. The RCK task fetches and verifies its exact Apache Iceberg source revision. Test-only child listener functions remain ignored and are invoked by their parent crash tests. -`test-suite` runs the host groups, including both Iceberg groups. Docker is a -separate explicit `test-single-node-container` task requiring a Linux amd64 -Docker host. It is always included in the DockerPreview CI job. +`test-suite` runs the host groups, including both Iceberg groups and the Rust +SDK task. The Rust SDK task is available through +`pixi run -e iceberg-e2e test-rust-iceberg-e2e` and the manual-only +`IcebergRustSDK` workflow. It does not run on regular pushes or pull requests. +DockerPreview is a manual-only workflow that runs +`pixi run test-single-node-container` on a Linux amd64 Docker host. The release +workflow also runs this image test before publication. -### Coverage guard +IcebergE2E uses the release profile for PyIceberg and native acceptance. The +access-server component suite runs in ServerTests, so IcebergE2E does not run it +again. -`pixi run test-task-coverage` validates every workspace package against -`TASK_PACKAGES` in `tools/ci-checks/check-test-task-coverage.py`, including the +### CI test-task check + +`pixi run check-ci-test-tasks` validates every workspace package against +`TASK_PACKAGES` in `tools/ci-checks/check-ci-test-tasks.py`, including the test harness's own runtime-namespace tests. -The guard follows Pixi group calls and checked-in shell scripts from CI, so an -existing component task disconnected from its job fails validation. It also -requires explicit CI reachability for the container and client acceptance tasks. +The guard follows Pixi group calls and checked-in shell scripts from CI, so a +required component task disconnected from its job fails validation. It also +checks that DockerPreview and IcebergRustSDK are reachable from their manual +workflows and absent from regular CI. ### Adding tests @@ -62,7 +70,7 @@ requires explicit CI reachability for the container and client acceptance tasks. 3. Add the component to the group script matching its runtime requirements. 4. Feature-gated or ignored tests require explicit task selectors. Do not count compiling an ignored test as executing it; exclude subprocess helper entries. -5. Run `pixi run test-task-coverage`, the affected suites, and workflow validation. +5. Run `pixi run check-ci-test-tasks`, the affected suites, and workflow validation. 6. Measure changed suites with `pixi run bash tools/test-metrics/measure.sh TASK...` and update the timing table. Environment selection is automatic. @@ -75,9 +83,9 @@ reported test counts and exit codes are saved under `.crowdb-runtime/artifacts/measure-tests/`. Timing includes incremental builds and subprocess startup/shutdown, so feature changes and cold builds affect it. Counts are runner-reported cases, not assertions; ignored cases are excluded. -Native Iceberg and Java/Rust/RCK SDK acceptance use release binaries, matching -the published container profile. Component suites retain their default test -profile. The focused debug native 100 MiB multipart upload, completion, +IcebergE2E and Java/Rust/RCK SDK acceptance use release binaries, matching +the published container profile. Other component suites retain their default +test profile. The focused debug native 100 MiB multipart upload, completion, restart, replay, and full read passed on 2026-09-29 in 54.40 s. Its previous 10 s completion deadline failure did not recur after the streaming I/O changes. @@ -146,7 +154,6 @@ All individual tests or test binaries with wall-clock time >= 7 s. | `test-chunk-client` | 18.33 s | `small_object_writer_e2e` — small-write E2E with real ChunkDB + DiskIO (14) | | `test-console-server` | 18.07 s | `cluster_deployer_test` — deployer lifecycle (3 tests) | | `test-console-shared` | 15.12 s | `lifecycle_e2e_test` — lifecycle E2E (1 test) | -| `test-console-server` | 13.53 s | `rolling_upgrade_test` — rolling upgrade (1 test) | | `test-console-ui` | 10.8 s | `50-chunk-capacity-disk-group:428` — assign disk-group to diskdb via UI | | `test-chunk-client` | 10.39 s | `chunk_reader_e2e` — chunk reader E2E with failure injection (6 tests) | | `test-console-server` | 9.93 s | `cluster_restart_incremental_test` — restart cycles (5 tests) | diff --git a/lib/crowdb-access-iceberg/Cargo.toml b/lib/crowdb-access-iceberg/Cargo.toml index 252c27795..a029acced 100644 --- a/lib/crowdb-access-iceberg/Cargo.toml +++ b/lib/crowdb-access-iceberg/Cargo.toml @@ -26,6 +26,7 @@ lz4_flex = { version = "0.11", default-features = false, features = ["std", "saf md-5 = "0.10" crowdb-chunk-kv-client = { path = "../crowdb-chunk-kv-client" } crowdb-chunk-client = { path = "../crowdb-chunk-client" } +crowdb-access-multipart = { path = "../crowdb-access-multipart" } crowdb-common = { workspace = true } crowdb-protocol = { path = "../crowdb-protocol" } flatbuffers = { workspace = true } diff --git a/lib/crowdb-access-iceberg/src/file/multipart.rs b/lib/crowdb-access-iceberg/src/file/multipart.rs index 0c75546cd..505c17d11 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart.rs @@ -2,6 +2,7 @@ use crate::catalog::CatalogContext; use crate::error::ValidationError; use crate::key::{CatalogScope, FileId, IcebergKey, OperationId}; use crate::operation::PayloadReference; +use crowdb_access_multipart::MultipartBounds; use sha2::{Digest, Sha256}; use std::fmt::Write; @@ -20,11 +21,13 @@ impl MultipartLimits { /// # Errors /// Rejects missing or incoherent independent multipart limits. pub fn validate(self) -> Result<(), ValidationError> { - if self.max_parts == 0 - || self.max_parts > 10_000 - || self.max_part_bytes == 0 - || self.max_part_bytes > self.max_file_bytes - || self.max_file_bytes > self.max_staged_bytes + if !(MultipartBounds { + max_parts: self.max_parts, + max_part_bytes: self.max_part_bytes, + max_object_bytes: self.max_file_bytes, + max_staged_bytes: self.max_staged_bytes, + }) + .valid() || self.max_staged_bytes > u64::MAX / 8 || self.ttl_ms == 0 || self.ttl_ms > 7 * 24 * 60 * 60 * 1000 @@ -35,15 +38,7 @@ impl MultipartLimits { } } -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum MultipartPhase { - Open, - Completing, - Publishing, - Published, - Aborted, - Conflicted, -} +pub use crowdb_access_multipart::MultipartPhase; #[derive(Clone, Debug, Eq, PartialEq)] pub struct MultipartCompletion { diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository.rs index fd4a6157b..ff9c08eb2 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository.rs @@ -5,6 +5,7 @@ use crate::error::ValidationError; use crate::key::{CatalogScope, IcebergKey, OperationId}; use crate::operation::mutation_identity; use crate::record::StorageRecord; +use crowdb_access_multipart::{live_at, next_revision}; use super::{MultipartPhase, MultipartSession}; @@ -131,7 +132,7 @@ impl MultipartRepository { } fn check_live(session: &MultipartSession, now_ms: u64) -> Result<(), CatalogError> { - if now_ms < session.created_ms || now_ms >= session.expires_ms { + if !live_at(session.created_ms, session.expires_ms, now_ms) { return Err(CatalogError::Conflict); } Ok(()) @@ -139,7 +140,7 @@ fn check_live(session: &MultipartSession, now_ms: u64) -> Result<(), CatalogErro fn increment(session: &MultipartSession) -> Result { let mut next = session.clone(); - next.revision = session.revision.checked_add(1).ok_or(ValidationError::Record)?; + next.revision = next_revision(session.revision).ok_or(ValidationError::Record)?; Ok(next) } diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs index 178c87ed6..199d8b732 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs @@ -4,6 +4,7 @@ use crate::file::{MultipartPart, MultipartPartMutation, MultipartPhase, Multipar use crate::key::{CatalogScope, IcebergKey}; use crate::operation::mutation_identity; use crate::record::StorageRecord; +use crowdb_access_multipart::{next_part_revision, reserve_part_accounting, PartAccounting}; use super::{check_live, increment, MultipartRepository}; @@ -41,9 +42,7 @@ impl MultipartRepository { before.validate_for(¤t)?; } let mut after = part.clone(); - after.revision = before - .as_ref() - .map_or(Some(1), |before| before.revision.checked_add(1)) + after.revision = next_part_revision(before.as_ref().map(|before| before.revision)) .ok_or(ValidationError::Record)?; after.modified_ms = now_ms; after.validate_for(¤t)?; @@ -180,15 +179,19 @@ impl MultipartRepository { before.validate_for(session)?; } let mut next = increment(session)?; - next.part_count = session - .part_count - .checked_add(u16::from(before.is_none())) - .ok_or(ValidationError::Record)?; - next.staged_bytes = session - .staged_bytes - .checked_sub(before.as_ref().map_or(0, MultipartPart::length)) - .and_then(|bytes| bytes.checked_add(part.length())) - .ok_or(ValidationError::Record)?; + let accounting = reserve_part_accounting( + PartAccounting { + count: session.part_count, + staged_bytes: session.staged_bytes, + }, + before.as_ref().map(MultipartPart::length), + part.length(), + session.limits.max_parts, + session.limits.max_staged_bytes, + ) + .map_err(|_| ValidationError::Record)?; + next.part_count = accounting.count; + next.staged_bytes = accounting.staged_bytes; next.pending = Some(MultipartPartMutation { before, after }); Ok(self.exchange(session, &next).await?.then_some(next)) } diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs index 943407a69..8b334b6c6 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs @@ -6,12 +6,22 @@ use crate::file::{ }; use crate::operation::PayloadStore; use crate::record::StorageRecord; -use crowdb_protocol::chunkdb::rpc::Location; -use md5::{Digest, Md5}; -use std::fmt::Write; +use crowdb_access_multipart::MultipartComposer; use super::{check_live, increment, MultipartRepository}; +fn raw_md5(etag: &str) -> Result<[u8; 16], ValidationError> { + let mut raw = [0_u8; 16]; + if etag.len() != 32 { + return Err(ValidationError::Record); + } + for (byte, pair) in raw.iter_mut().zip(etag.as_bytes().chunks_exact(2)) { + let pair = std::str::from_utf8(pair).map_err(|_| ValidationError::Record)?; + *byte = u8::from_str_radix(pair, 16).map_err(|_| ValidationError::Record)?; + } + Ok(raw) +} + impl MultipartRepository { /// Publishes selected durable part locations without reading or rewriting part bytes. /// # Errors @@ -37,41 +47,24 @@ impl MultipartRepository { .get(&completion.selection) .await?; let selection = MultipartSelection::decode(&bytes)?; - let mut locations = Vec::::new(); - let mut length = 0_u64; - let mut md5 = Md5::new(); + let mut composer = MultipartComposer::new(session.limits.max_file_bytes); for (index, selected) in selection.parts().iter().enumerate() { let snapshot = selection.snapshots().and_then(|snapshots| snapshots.get(index)); let Some(stream) = self.selected_stream(session, selected, snapshot).await? else { return Ok(None); }; let etag = stream.content.etag().ok_or(ValidationError::Record)?; - for pair in etag.as_bytes().chunks_exact(2) { - let pair = std::str::from_utf8(pair).map_err(|_| ValidationError::Record)?; - md5.update([u8::from_str_radix(pair, 16).map_err(|_| ValidationError::Record)?]); - } - for mut location in stream + let raw_md5 = raw_md5(etag)?; + let locations = stream .content .locations(stream.length)? - .ok_or(ValidationError::Record)? - { - location.logical_offset = location - .logical_offset - .checked_add(length) - .ok_or(ValidationError::Record)?; - locations.push(location); - } - length = length - .checked_add(stream.length) - .filter(|length| *length <= session.limits.max_file_bytes) .ok_or(ValidationError::Record)?; + composer + .push(stream.length, raw_md5, &locations) + .map_err(|_| ValidationError::Record)?; } - let mut etag = String::with_capacity(40); - for byte in md5.finalize() { - write!(&mut etag, "{byte:02x}").expect("string write cannot fail"); - } - write!(&mut etag, "-{}", selection.count()).expect("string write cannot fail"); - let content = FileContent::from_locations(&locations, length, etag)?; + let assembled = composer.finish().map_err(|_| ValidationError::Record)?; + let content = FileContent::from_locations(&assembled.locations, assembled.length, assembled.etag)?; let path = session.location.relative_key(); let extension = std::path::Path::new(path).extension(); let has_extension = |wanted: &str| extension.is_some_and(|value| value.eq_ignore_ascii_case(wanted)); @@ -93,7 +86,7 @@ impl MultipartRepository { location: session.location.clone(), kind, format, - length, + length: assembled.length, digest: [0; 32], content, hint: None, @@ -107,7 +100,7 @@ impl MultipartRepository { next.phase = MultipartPhase::Publishing; let completion = next.completion.as_mut().ok_or(ValidationError::Record)?; completion.progress.next_part = selection.count(); - completion.progress.completed_bytes = length; + completion.progress.completed_bytes = assembled.length; completion.publication = Some(publication); Ok(Some(self.exchange(session, &next).await?)) } diff --git a/lib/crowdb-access-iceberg/src/file/multipart_selection.rs b/lib/crowdb-access-iceberg/src/file/multipart_selection.rs index 8ae47ec83..c519035da 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_selection.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_selection.rs @@ -1,6 +1,8 @@ use crate::error::ValidationError; use crate::operation::MAX_PAYLOAD_BYTES; use bincode::Options; +use crowdb_access_multipart::validate_selected_parts; +pub use crowdb_access_multipart::SelectedPart; use serde::{Deserialize, Serialize}; use sha2::{Digest, Sha256}; @@ -10,13 +12,6 @@ const MAGIC_V1: &[u8; 5] = b"ICMS\x01"; const MAGIC_V2: &[u8; 5] = b"ICMS\x02"; const ENTRY_BYTES: usize = 42; -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub struct SelectedPart { - pub number: u16, - pub revision: u64, - pub digest: [u8; 32], -} - #[derive(Clone, Debug, Eq, PartialEq)] pub struct MultipartSelection { parts: Vec, @@ -35,16 +30,7 @@ impl MultipartSelection { /// # Errors /// Rejects empty, oversized, unordered or duplicate part selections. pub fn new(parts: Vec) -> Result { - if parts.is_empty() || parts.len() > 10_000 { - return Err(ValidationError::Record); - } - let mut previous = 0; - for part in &parts { - if part.number <= previous || part.number > 10_000 || part.revision == 0 { - return Err(ValidationError::Record); - } - previous = part.number; - } + validate_selected_parts(&parts, 10_000).map_err(|_| ValidationError::Record)?; let count = u16::try_from(parts.len()).map_err(|_| ValidationError::Record)?; Ok(Self { parts, diff --git a/lib/crowdb-access-multipart/Cargo.toml b/lib/crowdb-access-multipart/Cargo.toml new file mode 100644 index 000000000..9b2177de0 --- /dev/null +++ b/lib/crowdb-access-multipart/Cargo.toml @@ -0,0 +1,17 @@ +[package] +name = "crowdb-access-multipart" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +description = "Protocol-neutral multipart data composition for CROWDB access services." + +[lints] +workspace = true + +[dependencies] +crowdb-protocol = { path = "../crowdb-protocol" } +md-5 = "0.10" +serde = { version = "1", features = ["derive"] } +thiserror = { workspace = true } diff --git a/lib/crowdb-access-multipart/src/lib.rs b/lib/crowdb-access-multipart/src/lib.rs new file mode 100644 index 000000000..e10c4e2ec --- /dev/null +++ b/lib/crowdb-access-multipart/src/lib.rs @@ -0,0 +1,136 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Storage-backed multipart composition shared by access protocols. +//! +//! Part bytes remain in their original chunks. A completed object contains +//! the selected parts' locations with adjusted logical offsets, while its +//! composite `ETag` uses only the parts' previously recorded raw MD5 digests. + +use std::fmt::Write as _; + +use crowdb_protocol::chunkdb::rpc::Location; +use md5::{Digest, Md5}; + +mod state; + +pub use state::{ + live_at, next_part_revision, next_revision, reserve_part_accounting, validate_selected_parts, + MultipartBounds, MultipartPhase, PartAccounting, SelectedPart, StateError, +}; + +#[derive(Debug, thiserror::Error, PartialEq, Eq)] +pub enum ComposeError { + #[error("multipart part count exceeds 10000")] + TooManyParts, + #[error("multipart part locations do not cover its logical length")] + InvalidLocations, + #[error("multipart object length exceeds its limit")] + LengthLimit, + #[error("multipart location offset overflow")] + OffsetOverflow, + #[error("multipart completion has no parts")] + Empty, +} + +/// A completed metadata-only multipart object. +pub struct ComposedObject { + pub locations: Vec, + pub length: u64, + pub etag: String, +} + +/// Incrementally composes selected parts in completion order. +pub struct MultipartComposer { + locations: Vec, + length: u64, + max_length: u64, + part_count: u16, + md5: Md5, +} + +impl MultipartComposer { + #[must_use] + pub fn new(max_length: u64) -> Self { + Self { + locations: Vec::new(), + length: 0, + max_length, + part_count: 0, + md5: Md5::new(), + } + } + + /// Adds one already durable part without reading its data. + /// + /// # Errors + /// Rejects invalid or noncontiguous part locations, offset overflow, + /// too many parts, or an object that exceeds `max_length`. On error the + /// composer remains unchanged. + pub fn push( + &mut self, + length: u64, + raw_md5: [u8; 16], + locations: &[Location], + ) -> Result<(), ComposeError> { + if self.part_count >= 10_000 { + return Err(ComposeError::TooManyParts); + } + let next_length = self + .length + .checked_add(length) + .ok_or(ComposeError::OffsetOverflow)?; + if next_length > self.max_length { + return Err(ComposeError::LengthLimit); + } + let mut cursor = 0_u64; + for location in locations { + if location.chunk_id.is_none() + || location.length == 0 + || location.logical_length == 0 + || location.logical_offset != cursor + || location.offset.checked_add(location.length).is_none() + { + return Err(ComposeError::InvalidLocations); + } + cursor = cursor + .checked_add(location.logical_length) + .ok_or(ComposeError::OffsetOverflow)?; + self.length + .checked_add(location.logical_offset) + .ok_or(ComposeError::OffsetOverflow)?; + } + if cursor != length { + return Err(ComposeError::InvalidLocations); + } + for location in locations { + let mut adjusted = location.clone(); + adjusted.logical_offset += self.length; + self.locations.push(adjusted); + } + self.md5.update(raw_md5); + self.length = next_length; + self.part_count += 1; + Ok(()) + } + + /// Completes the composite `ETag` from the selected raw part MD5 values. + /// + /// # Errors + /// Rejects a completion with no selected parts. + pub fn finish(self) -> Result { + if self.part_count == 0 { + return Err(ComposeError::Empty); + } + let mut etag = String::with_capacity(40); + for byte in self.md5.finalize() { + write!(&mut etag, "{byte:02x}").expect("string write cannot fail"); + } + write!(&mut etag, "-{}", self.part_count).expect("string write cannot fail"); + Ok(ComposedObject { + locations: self.locations, + length: self.length, + etag, + }) + } +} diff --git a/lib/crowdb-access-multipart/src/state.rs b/lib/crowdb-access-multipart/src/state.rs new file mode 100644 index 000000000..81b68bdfb --- /dev/null +++ b/lib/crowdb-access-multipart/src/state.rs @@ -0,0 +1,138 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Protocol-neutral part selection and reservation invariants. + +/// Durable multipart transition phase shared by S3 and Iceberg. +#[derive(Clone, Copy, Debug, Eq, PartialEq, serde::Serialize, serde::Deserialize)] +pub enum MultipartPhase { + Open, + Completing, + Publishing, + Published, + Aborted, + Conflicted, +} + +/// Protocol-neutral admission bounds for a durable multipart upload. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct MultipartBounds { + pub max_parts: u16, + pub max_part_bytes: u64, + pub max_object_bytes: u64, + pub max_staged_bytes: u64, +} + +impl MultipartBounds { + #[must_use] + pub const fn valid(self) -> bool { + self.max_parts > 0 + && self.max_parts <= 10_000 + && self.max_part_bytes > 0 + && self.max_part_bytes <= self.max_object_bytes + && self.max_object_bytes <= self.max_staged_bytes + } +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq, serde::Serialize, serde::Deserialize)] +pub struct SelectedPart { + pub number: u16, + pub revision: u64, + pub digest: [u8; 32], +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct PartAccounting { + pub count: u16, + pub staged_bytes: u64, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] +pub enum StateError { + #[error("multipart selection has no parts")] + EmptySelection, + #[error("multipart part number is invalid or out of order")] + InvalidPartNumber, + #[error("multipart selected revision is zero")] + InvalidRevision, + #[error("multipart part count exceeds its limit")] + PartLimit, + #[error("multipart staged byte total exceeds its limit")] + StagedLimit, + #[error("multipart part counters are inconsistent")] + InvalidAccounting, +} + +/// Admission time is inclusive at creation and exclusive at expiry. +#[must_use] +pub const fn live_at(created_ms: u64, expires_ms: u64, now_ms: u64) -> bool { + created_ms <= now_ms && now_ms < expires_ms +} + +/// Returns the next durable session revision without wrapping. +#[must_use] +pub const fn next_revision(current: u64) -> Option { + current.checked_add(1) +} + +/// Returns the first or replacement part revision without wrapping. +#[must_use] +pub const fn next_part_revision(previous: Option) -> Option { + match previous { + Some(current) => next_revision(current), + None => Some(1), + } +} + +/// Validates one ordered completion selection independent of wire format. +/// +/// # Errors +/// Rejects empty, duplicate, descending, oversized or zero-revision entries. +pub fn validate_selected_parts(parts: &[SelectedPart], max_parts: u16) -> Result<(), StateError> { + if parts.is_empty() { + return Err(StateError::EmptySelection); + } + if parts.len() > usize::from(max_parts.min(10_000)) { + return Err(StateError::PartLimit); + } + let mut previous = 0; + for part in parts { + if part.number <= previous || part.number > max_parts.min(10_000) { + return Err(StateError::InvalidPartNumber); + } + if part.revision == 0 { + return Err(StateError::InvalidRevision); + } + previous = part.number; + } + Ok(()) +} + +/// Computes the counters to fence one new or replacement part publication. +/// +/// # Errors +/// Rejects count, byte and arithmetic limit violations before any metadata +/// mutation. A replacement keeps the part count and subtracts its old bytes. +pub fn reserve_part_accounting( + current: PartAccounting, + old_length: Option, + new_length: u64, + max_parts: u16, + max_staged_bytes: u64, +) -> Result { + let count = current + .count + .checked_add(u16::from(old_length.is_none())) + .filter(|count| *count <= max_parts.min(10_000)) + .ok_or(StateError::PartLimit)?; + let staged_bytes = current + .staged_bytes + .checked_sub(old_length.unwrap_or(0)) + .ok_or(StateError::InvalidAccounting)? + .checked_add(new_length) + .ok_or(StateError::StagedLimit)?; + if staged_bytes > max_staged_bytes { + return Err(StateError::StagedLimit); + } + Ok(PartAccounting { count, staged_bytes }) +} diff --git a/lib/crowdb-access-multipart/tests/composition_test.rs b/lib/crowdb-access-multipart/tests/composition_test.rs new file mode 100644 index 000000000..bdbd9a402 --- /dev/null +++ b/lib/crowdb-access-multipart/tests/composition_test.rs @@ -0,0 +1,72 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_multipart::{ComposeError, MultipartComposer}; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; + +fn location(part: u64, logical_length: u64) -> Location { + Location { + chunk_id: Some(ChunkId { high: 1, low: part }), + offset: 100, + length: logical_length + 34, + logical_offset: 0, + logical_length, + } +} + +#[test] +fn completion_composes_locations_and_saved_md5_in_selected_order() { + let mut composer = MultipartComposer::new(12); + composer.push(7, [0x11; 16], &[location(2, 7)]).unwrap(); + composer.push(5, [0x22; 16], &[location(1, 5)]).unwrap(); + let object = composer.finish().unwrap(); + assert_eq!(object.length, 12); + assert_eq!(object.locations[0].logical_offset, 0); + assert_eq!(object.locations[1].logical_offset, 7); + assert_eq!(object.locations[0].chunk_id.as_ref().unwrap().low, 2); + assert_eq!(object.locations[1].chunk_id.as_ref().unwrap().low, 1); + assert_eq!(object.etag, "b4ab393b73e0e71830bf2bf0e63c4d91-2"); +} + +#[test] +fn malformed_part_does_not_advance_composition() { + let mut composer = MultipartComposer::new(12); + let mut invalid = location(1, 5); + invalid.logical_offset = 1; + assert_eq!( + composer.push(5, [0x11; 16], &[invalid]), + Err(ComposeError::InvalidLocations) + ); + composer.push(5, [0x22; 16], &[location(2, 5)]).unwrap(); + let object = composer.finish().unwrap(); + assert_eq!(object.length, 5); + assert_eq!(object.locations[0].logical_offset, 0); + assert!(object.etag.ends_with("-1")); +} + +#[test] +fn length_and_offset_overflow_are_rejected() { + let mut composer = MultipartComposer::new(u64::MAX); + let first = Location { + chunk_id: Some(ChunkId { high: 1, low: 1 }), + offset: 100, + length: 1, + logical_offset: 0, + logical_length: u64::MAX - 1, + }; + composer.push(u64::MAX - 1, [0; 16], &[first]).unwrap(); + assert_eq!( + composer.push(2, [0; 16], &[location(2, 2)]), + Err(ComposeError::OffsetOverflow) + ); + assert_eq!(composer.finish().unwrap().length, u64::MAX - 1); +} + +#[test] +fn empty_selection_is_rejected() { + assert!(matches!( + MultipartComposer::new(1).finish(), + Err(ComposeError::Empty) + )); +} diff --git a/lib/crowdb-access-multipart/tests/state_test.rs b/lib/crowdb-access-multipart/tests/state_test.rs new file mode 100644 index 000000000..e122fcfcd --- /dev/null +++ b/lib/crowdb-access-multipart/tests/state_test.rs @@ -0,0 +1,99 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_multipart::{ + live_at, next_part_revision, next_revision, reserve_part_accounting, validate_selected_parts, + MultipartBounds, PartAccounting, SelectedPart, StateError, +}; + +#[test] +fn both_adapters_share_lifetime_and_revision_edges() { + assert!(!live_at(100, 200, 99)); + assert!(live_at(100, 200, 100)); + assert!(live_at(100, 200, 199)); + assert!(!live_at(100, 200, 200)); + assert_eq!(next_revision(1), Some(2)); + assert_eq!(next_revision(u64::MAX), None); + assert_eq!(next_part_revision(None), Some(1)); + assert_eq!(next_part_revision(Some(1)), Some(2)); + assert_eq!(next_part_revision(Some(u64::MAX)), None); +} + +fn selected(number: u16, revision: u64) -> SelectedPart { + SelectedPart { + number, + revision, + digest: [1; 32], + } +} + +#[test] +fn both_protocols_share_the_same_multipart_admission_bounds() { + let valid = MultipartBounds { + max_parts: 10_000, + max_part_bytes: 10, + max_object_bytes: 100, + max_staged_bytes: 200, + }; + assert!(valid.valid()); + assert!(!MultipartBounds { + max_parts: 10_001, + ..valid + } + .valid()); + assert!(!MultipartBounds { + max_part_bytes: 101, + ..valid + } + .valid()); + assert!(!MultipartBounds { + max_staged_bytes: 99, + ..valid + } + .valid()); +} + +#[test] +fn selected_parts_require_strict_order_and_stable_revisions() { + assert_eq!(validate_selected_parts(&[], 10), Err(StateError::EmptySelection)); + assert_eq!( + validate_selected_parts(&[selected(2, 1), selected(2, 2)], 10), + Err(StateError::InvalidPartNumber) + ); + assert_eq!( + validate_selected_parts(&[selected(2, 1), selected(1, 2)], 10), + Err(StateError::InvalidPartNumber) + ); + assert_eq!( + validate_selected_parts(&[selected(1, 0)], 10), + Err(StateError::InvalidRevision) + ); + assert_eq!( + validate_selected_parts(&[selected(2, 1), selected(9, 4)], 10), + Ok(()) + ); +} + +#[test] +fn replacing_a_part_preserves_count_and_reclaims_its_old_credit() { + let current = PartAccounting { + count: 2, + staged_bytes: 15, + }; + let next = reserve_part_accounting(current, Some(10), 8, 3, 20).unwrap(); + assert_eq!( + next, + PartAccounting { + count: 2, + staged_bytes: 13 + } + ); + assert_eq!( + reserve_part_accounting(current, None, 6, 3, 20), + Err(StateError::StagedLimit) + ); + assert_eq!( + reserve_part_accounting(current, Some(16), 1, 3, 20), + Err(StateError::InvalidAccounting) + ); +} diff --git a/lib/crowdb-access-s3/Cargo.toml b/lib/crowdb-access-s3/Cargo.toml index 8c76e1f9a..80ded8296 100644 --- a/lib/crowdb-access-s3/Cargo.toml +++ b/lib/crowdb-access-s3/Cargo.toml @@ -19,6 +19,7 @@ base64 = "0.22" bincode = "1" chrono = { version = "0.4", default-features = false, features = ["std"] } crowdb-chunk-client = { path = "../crowdb-chunk-client" } +crowdb-access-multipart = { path = "../crowdb-access-multipart" } crowdb-chunk-kv-client = { path = "../crowdb-chunk-kv-client" } crowdb-protocol = { path = "../crowdb-protocol" } flatbuffers = { workspace = true } @@ -29,6 +30,7 @@ md5 = "0.7" percent-encoding = "2" libc = "0.2" rand = "0.8" +serde = { version = "1", features = ["derive"] } sha2 = "0.10" subtle = "2" thiserror = { workspace = true } @@ -39,4 +41,4 @@ zeroize = { version = "1", features = ["derive"] } [build-dependencies] [dev-dependencies] -tokio = { workspace = true, features = ["macros", "rt-multi-thread"] } +tokio = { workspace = true, features = ["macros", "rt-multi-thread", "sync"] } diff --git a/lib/crowdb-access-s3/src/error.rs b/lib/crowdb-access-s3/src/error.rs index 00dcbf3ae..69a750467 100644 --- a/lib/crowdb-access-s3/src/error.rs +++ b/lib/crowdb-access-s3/src/error.rs @@ -13,8 +13,12 @@ pub enum S3ErrorCode { NotImplemented, NoSuchBucket, NoSuchKey, + NoSuchUpload, BucketNotEmpty, InvalidRequest, + InvalidPart, + InvalidPartOrder, + EntityTooSmall, InvalidRange, PreconditionFailed, SlowDown, @@ -35,8 +39,12 @@ impl S3ErrorCode { Self::NotImplemented => "NotImplemented", Self::NoSuchBucket => "NoSuchBucket", Self::NoSuchKey => "NoSuchKey", + Self::NoSuchUpload => "NoSuchUpload", Self::BucketNotEmpty => "BucketNotEmpty", Self::InvalidRequest => "InvalidRequest", + Self::InvalidPart => "InvalidPart", + Self::InvalidPartOrder => "InvalidPartOrder", + Self::EntityTooSmall => "EntityTooSmall", Self::InvalidRange => "InvalidRange", Self::PreconditionFailed => "PreconditionFailed", Self::SlowDown => "SlowDown", @@ -57,8 +65,12 @@ impl S3ErrorCode { Self::NotImplemented => NOT_IMPLEMENTED_MESSAGE, Self::NoSuchBucket => "The specified bucket does not exist.", Self::NoSuchKey => "The specified key does not exist.", + Self::NoSuchUpload => "The specified multipart upload does not exist.", Self::BucketNotEmpty => "The bucket you tried to delete is not empty.", Self::InvalidRequest => "The request is not valid for this service.", + Self::InvalidPart => "One or more of the specified parts could not be found or matched.", + Self::InvalidPartOrder => "The list of parts was not in ascending order.", + Self::EntityTooSmall => "A nonfinal multipart part is smaller than the minimum size.", Self::InvalidRange => "The requested range is not satisfiable.", Self::PreconditionFailed => "At least one precondition failed.", Self::SlowDown => "Please reduce your request rate.", @@ -83,9 +95,12 @@ impl S3ErrorCode { const fn status(self) -> StatusCode { match self { Self::NotImplemented => StatusCode::NOT_IMPLEMENTED, - Self::NoSuchBucket | Self::NoSuchKey => StatusCode::NOT_FOUND, + Self::NoSuchBucket | Self::NoSuchKey | Self::NoSuchUpload => StatusCode::NOT_FOUND, Self::BucketNotEmpty => StatusCode::CONFLICT, Self::InvalidRequest + | Self::InvalidPart + | Self::InvalidPartOrder + | Self::EntityTooSmall | Self::RequestTimeTooSkewed | Self::InvalidDigest | Self::BadDigest diff --git a/lib/crowdb-access-s3/src/integrity.rs b/lib/crowdb-access-s3/src/integrity.rs index 208390972..ecaf782fb 100644 --- a/lib/crowdb-access-s3/src/integrity.rs +++ b/lib/crowdb-access-s3/src/integrity.rs @@ -24,8 +24,8 @@ pub enum IntegrityError { /// /// The `ETag` is lowercase hexadecimal MD5 of the logical object bytes. It is /// intentionally calculated before metadata publication, never from physical -/// chunks, so frame and EC boundaries cannot change it. Multipart has its own -/// future contract. +/// chunks, so frame and EC boundaries cannot change it. Multipart uses the +/// selected parts' raw MD5 digests to calculate a composite `ETag`. pub struct SinglePartIntegrity { md5: md5::Context, sha256: Option, @@ -112,3 +112,32 @@ impl SinglePartIntegrity { Ok(result) } } + +/// Encodes a composite multipart `ETag` as a distinct 18-byte metadata +/// checksum marker: 16 digest bytes followed by the part count. +#[must_use] +pub fn multipart_checksum_marker(etag: &str) -> Option<[u8; 18]> { + let (digest, count) = etag.split_once('-')?; + if digest.len() != 32 || count.starts_with('0') { + return None; + } + let count: u16 = count.parse().ok()?; + if count == 0 || count > 10_000 { + return None; + } + let mut marker = [0_u8; 18]; + for (byte, pair) in marker[..16].iter_mut().zip(digest.as_bytes().chunks_exact(2)) { + let pair = std::str::from_utf8(pair).ok()?; + if pair.bytes().any(|byte| byte.is_ascii_uppercase()) { + return None; + } + *byte = u8::from_str_radix(pair, 16).ok()?; + } + marker[16..].copy_from_slice(&count.to_be_bytes()); + Some(marker) +} + +#[must_use] +pub fn is_multipart_checksum(checksum: &[u8], etag: &str) -> bool { + checksum.len() == 18 && multipart_checksum_marker(etag).is_some_and(|marker| checksum == marker) +} diff --git a/lib/crowdb-access-s3/src/metadata.rs b/lib/crowdb-access-s3/src/metadata.rs index decd1ea70..e1f20472c 100644 --- a/lib/crowdb-access-s3/src/metadata.rs +++ b/lib/crowdb-access-s3/src/metadata.rs @@ -4,6 +4,8 @@ //! Durable S3 namespace metadata. mod key; +mod multipart; +mod multipart_repository; mod namespace; mod record; mod store; @@ -21,6 +23,14 @@ mod generated { } pub use key::{BucketId, MetadataKey, MetadataKeyError, TenantId}; +pub use multipart::{ + new_upload_id, MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord, + PendingPartMutation, +}; +pub use multipart_repository::{ + CompletionPart, MultipartExpiryPage, MultipartPartPage, MultipartRepository, MultipartRepositoryError, + MultipartUploadPage, +}; pub use namespace::{BucketDeleteOutcome, BucketNamespace, BucketNamespaceError}; pub use record::{BucketNameRecord, MetadataRecordError, ObjectRecord}; pub use store::{ChunkKvMetadataStore, MetadataStoreError, PutIfAbsentOutcome}; diff --git a/lib/crowdb-access-s3/src/metadata/key.rs b/lib/crowdb-access-s3/src/metadata/key.rs index cfdc7492e..96ac50b45 100644 --- a/lib/crowdb-access-s3/src/metadata/key.rs +++ b/lib/crowdb-access-s3/src/metadata/key.rs @@ -5,6 +5,11 @@ use std::fmt; const BUCKET_NAME_KIND: u8 = 1; const OBJECT_KIND: u8 = 2; +const MULTIPART_INDEX_KIND: u8 = 3; +const MULTIPART_UPLOAD_KIND: u8 = 4; +const MULTIPART_SESSION_ENTRY: u8 = 0; +const MULTIPART_PART_ENTRY: u8 = 1; +const MULTIPART_GENERATION_ENTRY: u8 = 2; const MAX_KEY_BYTES: usize = 1024; /// A tenant namespace identity. @@ -30,7 +35,7 @@ impl TenantId { } /// An immutable bucket identity. -#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd)] +#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd, serde::Serialize, serde::Deserialize)] pub struct BucketId([u8; 16]); impl BucketId { @@ -150,12 +155,155 @@ impl MetadataKey { end.push(OBJECT_KIND + 1); end } + + /// Starts the ordered multipart listing index for one bucket. + #[must_use] + pub fn multipart_session_prefix(tenant: &TenantId, bucket: BucketId) -> Vec { + let mut key = namespace_prefix(tenant); + key.extend_from_slice(bucket.as_bytes()); + key.push(MULTIPART_INDEX_KIND); + key + } + + /// Ends the multipart upload interval for one bucket. + #[must_use] + pub fn multipart_session_end(tenant: &TenantId, bucket: BucketId) -> Vec { + let mut key = namespace_prefix(tenant); + key.extend_from_slice(bucket.as_bytes()); + key.push(MULTIPART_INDEX_KIND + 1); + key + } + + /// Starts the upload interval whose object names share a byte prefix. + /// + /// # Errors + /// Rejects a prefix longer than an object key. + pub fn multipart_session_key_prefix( + tenant: &TenantId, + bucket: BucketId, + object_prefix: &[u8], + ) -> Result, MetadataKeyError> { + if object_prefix.len() > MAX_KEY_BYTES { + return Err(MetadataKeyError::TooLong("object key prefix")); + } + let mut key = Self::multipart_session_prefix(tenant, bucket); + append_ordered_bytes_prefix(&mut key, object_prefix); + Ok(key) + } + + /// Ends the upload interval whose object names share a byte prefix. + /// + /// # Errors + /// Rejects a prefix longer than an object key. + pub fn multipart_session_key_prefix_end( + tenant: &TenantId, + bucket: BucketId, + object_prefix: &[u8], + ) -> Result, MetadataKeyError> { + if object_prefix.is_empty() { + return Ok(Self::multipart_session_end(tenant, bucket)); + } + let mut end = Self::multipart_session_key_prefix(tenant, bucket, object_prefix)?; + increment_lexicographic(&mut end); + Ok(end) + } + + /// Identifies one immutable listing entry by object key and upload ID. + /// + /// # Errors + /// Rejects an empty or oversized object key. + pub fn multipart_upload_index( + tenant: &TenantId, + bucket: BucketId, + object: &[u8], + upload_id: &[u8; 16], + ) -> Result, MetadataKeyError> { + validate_component("object key", object)?; + let mut key = Self::multipart_session_prefix(tenant, bucket); + append_ordered_bytes(&mut key, object); + key.extend_from_slice(upload_id); + Ok(key) + } + + /// Starts the session and all part records for one upload. + #[must_use] + pub fn multipart_upload_prefix(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { + let mut key = namespace_prefix(tenant); + key.extend_from_slice(bucket.as_bytes()); + key.push(MULTIPART_UPLOAD_KIND); + key.extend_from_slice(upload_id); + key + } + + /// Identifies the mutable session inside one upload interval. + #[must_use] + pub fn multipart_session(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { + let mut key = Self::multipart_upload_prefix(tenant, bucket, upload_id); + key.push(MULTIPART_SESSION_ENTRY); + key + } + + /// Starts the current-part interval of one upload. + #[must_use] + pub fn multipart_part_prefix(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { + let mut key = Self::multipart_upload_prefix(tenant, bucket, upload_id); + key.push(MULTIPART_PART_ENTRY); + key + } + + /// Ends the part interval of one upload. + #[must_use] + pub fn multipart_part_end(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { + let mut key = Self::multipart_part_prefix(tenant, bucket, upload_id); + increment_lexicographic(&mut key); + key + } + + /// Identifies one part number independently of its replacement revision. + /// + /// # Errors + /// Rejects part numbers outside the S3 multipart range. + pub fn multipart_part( + tenant: &TenantId, + bucket: BucketId, + upload_id: &[u8; 16], + number: u16, + ) -> Result, MetadataKeyError> { + if number == 0 || number > 10_000 { + return Err(MetadataKeyError::InvalidPartNumber); + } + let mut key = Self::multipart_part_prefix(tenant, bucket, upload_id); + key.extend_from_slice(&number.to_be_bytes()); + Ok(key) + } + + /// Identifies an immutable generation retained after part-number replacement. + /// + /// # Errors + /// Rejects invalid part numbers or a zero revision. + pub fn multipart_part_generation( + tenant: &TenantId, + bucket: BucketId, + upload_id: &[u8; 16], + number: u16, + revision: u64, + ) -> Result, MetadataKeyError> { + if number == 0 || number > 10_000 || revision == 0 { + return Err(MetadataKeyError::InvalidPartNumber); + } + let mut key = Self::multipart_upload_prefix(tenant, bucket, upload_id); + key.push(MULTIPART_GENERATION_ENTRY); + key.extend_from_slice(&number.to_be_bytes()); + key.extend_from_slice(&revision.to_be_bytes()); + Ok(key) + } } #[derive(Clone, Debug, Eq, PartialEq)] pub enum MetadataKeyError { Empty(&'static str), TooLong(&'static str), + InvalidPartNumber, } impl fmt::Display for MetadataKeyError { @@ -163,6 +311,7 @@ impl fmt::Display for MetadataKeyError { match self { Self::Empty(component) => write!(formatter, "{component} cannot be empty"), Self::TooLong(component) => write!(formatter, "{component} exceeds {MAX_KEY_BYTES} bytes"), + Self::InvalidPartNumber => write!(formatter, "multipart part number must be in 1..=10000"), } } } diff --git a/lib/crowdb-access-s3/src/metadata/multipart.rs b/lib/crowdb-access-s3/src/metadata/multipart.rs new file mode 100644 index 000000000..dadef0295 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart.rs @@ -0,0 +1,289 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Versioned durable S3 multipart session and part values. + +use bincode::Options as _; +use crowdb_access_multipart::{ + next_part_revision, validate_selected_parts, MultipartBounds, MultipartComposer, SelectedPart, +}; +use crowdb_protocol::chunkdb::rpc::Location; +use serde::{Deserialize, Serialize}; +use sha2::{Digest as _, Sha256}; + +use super::BucketId; + +const SESSION_MAGIC: [u8; 5] = *b"S3MS\x02"; +const PART_MAGIC: [u8; 5] = *b"S3MP\x01"; +const MAX_RECORD_BYTES: u64 = 1024 * 1024; +const MAX_OBJECT_KEY_BYTES: usize = 1024; +const MAX_CONTENT_TYPE_BYTES: usize = 1024; + +pub use crowdb_access_multipart::MultipartPhase; + +/// Creates a random upload identity whose byte order follows initiation time. +/// Uploads created in the same millisecond have an unspecified relative order. +#[must_use] +pub fn new_upload_id(now_ms: u64) -> [u8; 16] { + let mut id = *uuid::Uuid::new_v4().as_bytes(); + id[..8].copy_from_slice(&now_ms.to_be_bytes()); + id +} + +#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] +pub struct MultipartSessionRecord { + pub bucket_id: BucketId, + pub object_key: Vec, + pub upload_id: [u8; 16], + pub revision: u64, + pub phase: MultipartPhase, + pub created_ms: u64, + pub expires_ms: u64, + pub content_type: String, + pub max_parts: u16, + pub max_part_bytes: u64, + pub max_object_bytes: u64, + pub max_staged_bytes: u64, + pub part_count: u16, + pub staged_bytes: u64, + pub pending: Option, + pub selection: Option>, + pub completion_request_digest: Option<[u8; 32]>, + pub publication_ms: Option, + pub object_predecessor: Option>, + pub etag: Option, +} + +/// A durable session fence for publishing one current-part pointer. +/// The immutable after-generation is stored before reserving this mutation. +#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] +pub struct PendingPartMutation { + pub number: u16, + pub before_revision: Option, + pub before_digest: Option<[u8; 32]>, + pub after_revision: u64, + pub after_digest: [u8; 32], + pub after_length: u64, +} + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct MultipartPartRecord { + pub bucket_id: BucketId, + pub upload_id: [u8; 16], + pub number: u16, + pub revision: u64, + pub modified_ms: u64, + pub length: u64, + pub raw_md5: [u8; 16], + pub locations: Vec, +} + +#[derive(Debug, Eq, PartialEq, thiserror::Error)] +pub enum MultipartRecordError { + #[error("invalid multipart record")] + Invalid, + #[error("multipart record exceeds its byte limit")] + TooLarge, + #[error("multipart record belongs to another key")] + Identity, +} + +impl MultipartSessionRecord { + /// Encodes one validated session with a distinct schema version. + /// + /// # Errors + /// Rejects incoherent phases, bounds and oversized records. + pub fn encode(&self) -> Result, MultipartRecordError> { + self.validate()?; + encode(SESSION_MAGIC, self) + } + + /// Decodes and binds one session to the requested bucket, key and upload. + /// + /// # Errors + /// Rejects corrupt, oversized, foreign or incoherent values. + pub fn decode( + bytes: &[u8], + bucket: BucketId, + object: &[u8], + upload_id: &[u8; 16], + ) -> Result { + let record: Self = decode(SESSION_MAGIC, bytes)?; + record.validate()?; + if record.bucket_id != bucket || record.object_key != object || &record.upload_id != upload_id { + return Err(MultipartRecordError::Identity); + } + Ok(record) + } + + pub(crate) fn decode_unbound(bytes: &[u8]) -> Result { + let record: Self = decode(SESSION_MAGIC, bytes)?; + record.validate()?; + Ok(record) + } + + fn validate(&self) -> Result<(), MultipartRecordError> { + if self.object_key.is_empty() + || self.object_key.len() > MAX_OBJECT_KEY_BYTES + || self.upload_id == [0; 16] + || self.revision == 0 + || self.created_ms >= self.expires_ms + || self.content_type.len() > MAX_CONTENT_TYPE_BYTES + || !(MultipartBounds { + max_parts: self.max_parts, + max_part_bytes: self.max_part_bytes, + max_object_bytes: self.max_object_bytes, + max_staged_bytes: self.max_staged_bytes, + }) + .valid() + || self.part_count > self.max_parts + || self.staged_bytes > self.max_staged_bytes + { + return Err(MultipartRecordError::Invalid); + } + if let Some(selection) = &self.selection { + validate_selected_parts(selection, self.max_parts).map_err(|_| MultipartRecordError::Invalid)?; + if selection.len() > usize::from(self.part_count) { + return Err(MultipartRecordError::Invalid); + } + } + if let Some(pending) = &self.pending { + if self.phase != MultipartPhase::Open + || self.part_count == 0 + || pending.number == 0 + || pending.number > self.max_parts + || pending.after_length > self.max_part_bytes + || self.staged_bytes < pending.after_length + || pending.before_revision.is_some() != pending.before_digest.is_some() + || next_part_revision(pending.before_revision) != Some(pending.after_revision) + { + return Err(MultipartRecordError::Invalid); + } + } + let selected_count = self.selection.as_ref().map_or(0, Vec::len); + match self.phase { + MultipartPhase::Open + if self.selection.is_none() + && self.completion_request_digest.is_none() + && self.publication_ms.is_none() + && self.object_predecessor.is_none() + && self.etag.is_none() => + { + Ok(()) + } + MultipartPhase::Completing + if self.selection.is_some() + && self.completion_request_digest.is_some() + && self.publication_ms.is_none() + && self.object_predecessor.is_none() + && self.etag.is_none() => + { + Ok(()) + } + MultipartPhase::Publishing | MultipartPhase::Published + if self.selection.is_some() + && self.completion_request_digest.is_some() + && self + .publication_ms + .is_some_and(|time| time >= self.created_ms && time < self.expires_ms) + && self.object_predecessor.is_some() + && self + .etag + .as_ref() + .is_some_and(|etag| valid_etag(etag, selected_count)) => + { + Ok(()) + } + MultipartPhase::Aborted if self.etag.is_none() => Ok(()), + _ => Err(MultipartRecordError::Invalid), + } + } +} + +impl MultipartPartRecord { + /// Binds a completion selection to this exact persisted part generation. + /// + /// # Errors + /// Rejects an invalid part record before calculating its identity. + pub fn selection_digest(&self) -> Result<[u8; 32], MultipartRecordError> { + Ok(Sha256::digest(self.encode()?).into()) + } + + /// Encodes one immutable selected part generation. + /// + /// # Errors + /// Rejects invalid identity, locations or oversized records. + pub fn encode(&self) -> Result, MultipartRecordError> { + self.validate()?; + encode(PART_MAGIC, self) + } + + /// Decodes and binds a part to the requested bucket, upload and number. + /// + /// # Errors + /// Rejects corrupt, oversized, foreign or incoherent values. + pub fn decode( + bytes: &[u8], + bucket: BucketId, + upload_id: &[u8; 16], + number: u16, + ) -> Result { + let record: Self = decode(PART_MAGIC, bytes)?; + record.validate()?; + if record.bucket_id != bucket || &record.upload_id != upload_id || record.number != number { + return Err(MultipartRecordError::Identity); + } + Ok(record) + } + + fn validate(&self) -> Result<(), MultipartRecordError> { + if self.upload_id == [0; 16] || self.number == 0 || self.number > 10_000 || self.revision == 0 { + return Err(MultipartRecordError::Invalid); + } + let mut composer = MultipartComposer::new(self.length); + composer + .push(self.length, self.raw_md5, &self.locations) + .map_err(|_| MultipartRecordError::Invalid)?; + composer.finish().map_err(|_| MultipartRecordError::Invalid)?; + Ok(()) + } +} + +fn valid_etag(etag: &str, selected_count: usize) -> bool { + let Some((digest, count)) = etag.split_once('-') else { + return false; + }; + digest.len() == 32 + && digest + .bytes() + .all(|byte| byte.is_ascii_hexdigit() && !byte.is_ascii_uppercase()) + && count + .parse::() + .is_ok_and(|count| count > 0 && usize::from(count) == selected_count) +} + +fn encode(magic: [u8; 5], value: &T) -> Result, MultipartRecordError> { + let mut bytes = magic.to_vec(); + let encoded = bincode::DefaultOptions::new() + .with_fixint_encoding() + .with_limit(MAX_RECORD_BYTES - 5) + .serialize(value) + .map_err(|_| MultipartRecordError::TooLarge)?; + bytes.extend_from_slice(&encoded); + Ok(bytes) +} + +fn decode Deserialize<'de>>(magic: [u8; 5], bytes: &[u8]) -> Result { + if bytes.len() as u64 > MAX_RECORD_BYTES { + return Err(MultipartRecordError::TooLarge); + } + if bytes.get(..5) != Some(&magic[..]) { + return Err(MultipartRecordError::Invalid); + } + bincode::DefaultOptions::new() + .with_fixint_encoding() + .with_limit(MAX_RECORD_BYTES - 5) + .reject_trailing_bytes() + .deserialize(&bytes[5..]) + .map_err(|_| MultipartRecordError::Invalid) +} diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs new file mode 100644 index 000000000..da625fa1a --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -0,0 +1,231 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! CAS-backed multipart authority in the S3 Chunk-KV namespace. + +use crowdb_access_multipart::next_revision; +use std::sync::Arc; + +use super::{ + ChunkKvMetadataStore, MetadataKey, MetadataKeyError, MetadataStoreError, MultipartPartRecord, + MultipartPhase, MultipartRecordError, MultipartSessionRecord, PutIfAbsentOutcome, TenantId, +}; + +mod completion; +mod listing; +mod parts; +mod publication; +mod terminal; + +pub use completion::CompletionPart; +pub use listing::{MultipartPartPage, MultipartUploadPage}; +pub use terminal::MultipartExpiryPage; + +#[derive(Debug, thiserror::Error)] +pub enum MultipartRepositoryError { + #[error(transparent)] + Key(#[from] MetadataKeyError), + #[error(transparent)] + Record(#[from] MultipartRecordError), + #[error(transparent)] + Store(#[from] MetadataStoreError), + #[error("multipart operation conflicts with durable state")] + Conflict, + #[error("multipart mutation is still settling")] + Busy, + #[error("multipart completion references a missing or changed part")] + InvalidPart, + #[error("a nonfinal multipart part is smaller than 5 MiB")] + EntityTooSmall, + #[error("multipart listing exhausted its bounded scan budget")] + ScanBudgetExhausted, +} + +pub struct MultipartRepository { + store: Arc, + tenant: TenantId, +} + +impl MultipartRepository { + #[must_use] + pub fn new(store: Arc, tenant: TenantId) -> Self { + Self { store, tenant } + } + + /// Creates one upload or confirms the exact record after a lost reply. + /// + /// # Errors + /// Rejects a different record at the same upload key or unavailable store. + pub async fn begin( + &self, + session: &MultipartSessionRecord, + ) -> Result { + if session.phase != MultipartPhase::Open || session.revision != 1 { + return Err(MultipartRepositoryError::Conflict); + } + let value = session.encode()?; + let key = self.session_key(session); + let index = MetadataKey::multipart_upload_index( + &self.tenant, + session.bucket_id, + &session.object_key, + &session.upload_id, + )?; + let indexed = self.store.put_if_absent(index.clone(), key.clone()).await; + match indexed { + Ok(PutIfAbsentOutcome::Inserted { .. }) => {} + Ok(PutIfAbsentOutcome::Existing(existing)) if existing.value == key => {} + Ok(PutIfAbsentOutcome::Existing(_)) => return Err(MultipartRepositoryError::Conflict), + Err(error) => { + if !self + .store + .get(index) + .await? + .is_some_and(|entry| entry.value == key) + { + return Err(error.into()); + } + } + } + let result = self.store.put_if_absent(key, value.clone()).await; + match result { + Ok(PutIfAbsentOutcome::Inserted { .. }) => Ok(session.clone()), + Ok(PutIfAbsentOutcome::Existing(existing)) if existing.value == value => Ok(session.clone()), + Ok(PutIfAbsentOutcome::Existing(_)) => Err(MultipartRepositoryError::Conflict), + Err(error) => { + if self.load(session).await?.as_ref() == Some(session) { + Ok(session.clone()) + } else { + Err(error.into()) + } + } + } + } + + /// Reads one upload through its exact bucket/object/upload identity. + /// + /// # Errors + /// Rejects corrupt or foreign values and storage failures. + pub async fn load( + &self, + identity: &MultipartSessionRecord, + ) -> Result, MultipartRepositoryError> { + self.load_identity(identity.bucket_id, &identity.object_key, &identity.upload_id) + .await + } + + /// Reads one session through its bucket, object and upload identity. + /// + /// # Errors + /// Rejects corrupt or foreign values and storage failures. + pub async fn load_identity( + &self, + bucket: super::BucketId, + object: &[u8], + upload_id: &[u8; 16], + ) -> Result, MultipartRepositoryError> { + let key = MetadataKey::multipart_upload_index(&self.tenant, bucket, object, upload_id)?; + let Some(index) = self.store.get(key).await? else { + return Ok(None); + }; + let expected = MetadataKey::multipart_session(&self.tenant, bucket, upload_id); + if index.value != expected { + return Err(MultipartRepositoryError::Conflict); + } + self.store + .get(expected) + .await? + .map(|value| { + MultipartSessionRecord::decode(&value.value, bucket, object, upload_id).map_err(Into::into) + }) + .transpose() + } + + /// Moves one session phase under an exact-value CAS fence. + /// + /// # Errors + /// Rejects invalid revisions, changed identity and uncertain storage + /// writes that cannot be confirmed by a matching read. + pub async fn exchange( + &self, + previous: &MultipartSessionRecord, + next: &MultipartSessionRecord, + ) -> Result { + if previous.bucket_id != next.bucket_id + || previous.object_key != next.object_key + || previous.upload_id != next.upload_id + || next_revision(previous.revision) != Some(next.revision) + { + return Err(MultipartRepositoryError::Conflict); + } + let key = self.session_key(previous); + let expected = previous.encode()?; + let value = next.encode()?; + match self.store.compare_exchange(key, expected, value).await { + Ok(true) => Ok(true), + Ok(false) => Ok(self.load(next).await?.as_ref() == Some(next)), + Err(error) => { + if self.load(next).await?.as_ref() == Some(next) { + Ok(true) + } else { + Err(error.into()) + } + } + } + } + + /// Reads one current part generation. + /// + /// # Errors + /// Rejects corrupt or foreign records and storage failures. + pub async fn part( + &self, + session: &MultipartSessionRecord, + number: u16, + ) -> Result, MultipartRepositoryError> { + let key = MetadataKey::multipart_part(&self.tenant, session.bucket_id, &session.upload_id, number)?; + self.store + .get(key) + .await? + .map(|value| { + MultipartPartRecord::decode(&value.value, session.bucket_id, &session.upload_id, number) + .map_err(Into::into) + }) + .transpose() + } + + /// Reads one immutable part generation, including replaced generations. + /// + /// # Errors + /// Rejects corrupt or foreign generation bytes and unavailable storage. + pub async fn part_generation( + &self, + session: &MultipartSessionRecord, + number: u16, + revision: u64, + ) -> Result, MultipartRepositoryError> { + let key = MetadataKey::multipart_part_generation( + &self.tenant, + session.bucket_id, + &session.upload_id, + number, + revision, + )?; + self.store + .get(key) + .await? + .map(|value| { + let part = + MultipartPartRecord::decode(&value.value, session.bucket_id, &session.upload_id, number)?; + if part.revision != revision { + return Err(MultipartRepositoryError::InvalidPart); + } + Ok(part) + }) + .transpose() + } + + fn session_key(&self, session: &MultipartSessionRecord) -> Vec { + MetadataKey::multipart_session(&self.tenant, session.bucket_id, &session.upload_id) + } +} diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs new file mode 100644 index 000000000..f288a5cc2 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs @@ -0,0 +1,144 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Freeze the exact selected part generations before object publication. + +use crowdb_access_multipart::{ + live_at, next_revision, validate_selected_parts, MultipartComposer, SelectedPart, +}; +use sha2::{Digest as _, Sha256}; + +use super::{MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord}; + +const MIN_NONFINAL_PART_BYTES: u64 = 5 * 1024 * 1024; + +/// One S3 `CompleteMultipartUpload` part reference in request order. +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct CompletionPart { + pub number: u16, + pub etag: String, +} + +impl MultipartRepository { + /// Validates and freezes an ordered part selection under the session CAS. + /// + /// A reserved part mutation settles before this phase write. Publication + /// rechecks the selected pointer and generation evidence. + /// + /// # Errors + /// Rejects missing, duplicate, undersized or mismatched parts and + /// conflicting/expired uploads. No object is published by this step. + pub async fn freeze_completion( + &self, + session: &MultipartSessionRecord, + requested: &[CompletionPart], + now_ms: u64, + ) -> Result, MultipartRepositoryError> { + let request_digest = request_digest(requested)?; + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if matches!( + current.phase, + MultipartPhase::Publishing | MultipartPhase::Published + ) && current.completion_request_digest == Some(request_digest) + { + return Ok(Some(current)); + } + if current.pending.is_some() { + self.settle_pending_part(¤t).await?; + return Ok(None); + } + if current.phase != MultipartPhase::Open + || !live_at(current.created_ms, current.expires_ms, now_ms) + || requested.is_empty() + || requested.len() > usize::from(current.max_parts) + { + return Err(MultipartRepositoryError::Conflict); + } + let mut composer = MultipartComposer::new(current.max_object_bytes); + let mut selection = Vec::with_capacity(requested.len()); + let mut previous = 0; + for (index, request) in requested.iter().enumerate() { + if request.number <= previous || request.number > current.max_parts { + return Err(MultipartRepositoryError::InvalidPart); + } + previous = request.number; + let part = self + .part(¤t, request.number) + .await? + .ok_or(MultipartRepositoryError::InvalidPart)?; + if index + 1 < requested.len() && part.length < MIN_NONFINAL_PART_BYTES { + return Err(MultipartRepositoryError::EntityTooSmall); + } + if !etag_matches(&request.etag, &part.raw_md5) { + return Err(MultipartRepositoryError::InvalidPart); + } + composer + .push(part.length, part.raw_md5, &part.locations) + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + selection.push(SelectedPart { + number: part.number, + revision: part.revision, + digest: part.selection_digest()?, + }); + } + validate_selected_parts(&selection, current.max_parts) + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + let assembled = composer + .finish() + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + let mut next = current.clone(); + next.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; + next.phase = MultipartPhase::Publishing; + next.part_count = + u16::try_from(selection.len()).map_err(|_| MultipartRepositoryError::InvalidPart)?; + next.staged_bytes = assembled.length; + next.selection = Some(selection); + next.completion_request_digest = Some(request_digest); + next.publication_ms = Some(now_ms); + let object_key = super::MetadataKey::object(&self.tenant, current.bucket_id, ¤t.object_key)?; + next.object_predecessor = Some( + self.store + .get(object_key) + .await? + .map(|value| Sha256::digest(value.value).into()), + ); + next.etag = Some(assembled.etag); + Ok(self.exchange(¤t, &next).await?.then_some(next)) + } +} + +fn etag_matches(value: &str, raw_md5: &[u8; 16]) -> bool { + parse_raw_md5(value).is_some_and(|actual| &actual == raw_md5) +} + +fn request_digest(requested: &[CompletionPart]) -> Result<[u8; 32], MultipartRepositoryError> { + let mut sha = Sha256::new(); + sha.update( + u16::try_from(requested.len()) + .map_err(|_| MultipartRepositoryError::InvalidPart)? + .to_be_bytes(), + ); + for part in requested { + sha.update(part.number.to_be_bytes()); + sha.update(parse_raw_md5(&part.etag).ok_or(MultipartRepositoryError::InvalidPart)?); + } + Ok(sha.finalize().into()) +} + +fn parse_raw_md5(value: &str) -> Option<[u8; 16]> { + let unquoted = value + .strip_prefix('"') + .and_then(|value| value.strip_suffix('"')) + .unwrap_or(value); + if unquoted.len() != 32 { + return None; + } + let mut raw = [0_u8; 16]; + for (output, pair) in raw.iter_mut().zip(unquoted.as_bytes().chunks_exact(2)) { + *output = u8::from_str_radix(std::str::from_utf8(pair).ok()?, 16).ok()?; + } + Some(raw) +} diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs new file mode 100644 index 000000000..3a7b3624d --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs @@ -0,0 +1,203 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Bounded listing of current part-number generations. + +use super::{ + MetadataKey, MultipartPartRecord, MultipartPhase, MultipartRepository, MultipartRepositoryError, + MultipartSessionRecord, +}; +use crate::metadata::BucketId; + +const MAX_LIST_PARTS: usize = 1_000; +const MAX_SCAN_BYTES: usize = 4 * 1024 * 1024; +const MAX_UPLOAD_SCAN_PAGES: usize = 8; +const MAX_UPLOAD_SCAN_ITEMS: usize = 4_096; + +pub struct MultipartPartPage { + pub parts: Vec, + pub next_part_number_marker: Option, +} + +pub struct MultipartUploadPage { + pub uploads: Vec, + pub next: Option<(Vec, [u8; 16])>, +} + +impl MultipartRepository { + /// Lists active uploads in object-key and upload-ID order with bounded scans. + /// + /// Callers must create upload IDs in initiation-time order for the same key. + /// A dense interval of terminal records returns a scan-budget error rather + /// than an incomplete success page. + /// + /// # Errors + /// Rejects invalid markers, corrupt records and exhausted scan budgets. + pub async fn list_uploads( + &self, + bucket: BucketId, + prefix: &[u8], + key_marker: Option<&[u8]>, + upload_marker: Option<&[u8; 16]>, + max_uploads: usize, + now_ms: u64, + ) -> Result { + if max_uploads == 0 + || max_uploads > MAX_LIST_PARTS + || (upload_marker.is_some() && key_marker.is_none()) + { + return Err(MultipartRepositoryError::Conflict); + } + let mut start = MetadataKey::multipart_session_key_prefix(&self.tenant, bucket, prefix)?; + let end = MetadataKey::multipart_session_key_prefix_end(&self.tenant, bucket, prefix)?; + if let Some(key) = key_marker { + let mut after = MetadataKey::multipart_upload_index( + &self.tenant, + bucket, + key, + upload_marker.unwrap_or(&[u8::MAX; 16]), + )?; + after.push(0); + start = start.max(after); + } + if start >= end { + return Ok(MultipartUploadPage { + uploads: Vec::new(), + next: None, + }); + } + let mut uploads = Vec::with_capacity(max_uploads + 1); + let mut continuation = None; + let mut scanned = 0; + for _ in 0..MAX_UPLOAD_SCAN_PAGES { + let page = self + .store + .scan_page( + start.clone(), + end.clone(), + (MAX_UPLOAD_SCAN_ITEMS - scanned).min(MAX_LIST_PARTS), + MAX_SCAN_BYTES, + continuation, + ) + .await?; + scanned += page.items.len(); + for item in page.items { + let Some(record) = self.store.get(item.value).await? else { + continue; + }; + let session = MultipartSessionRecord::decode_unbound(&record.value)?; + if session.bucket_id != bucket + || MetadataKey::multipart_upload_index( + &self.tenant, + bucket, + &session.object_key, + &session.upload_id, + )? != item.key + || MetadataKey::multipart_session(&self.tenant, bucket, &session.upload_id) != record.key + { + return Err(MultipartRepositoryError::Conflict); + } + if session.phase == MultipartPhase::Open + && session.created_ms <= now_ms + && now_ms < session.expires_ms + { + uploads.push(session); + if uploads.len() > max_uploads { + uploads.pop(); + let last = uploads.last().ok_or(MultipartRepositoryError::Conflict)?; + let next = Some((last.object_key.clone(), last.upload_id)); + return Ok(MultipartUploadPage { uploads, next }); + } + } + } + match page.continuation { + None => return Ok(MultipartUploadPage { uploads, next: None }), + Some(next) if scanned < MAX_UPLOAD_SCAN_ITEMS => continuation = Some(next), + Some(_) => return Err(MultipartRepositoryError::ScanBudgetExhausted), + } + } + Err(MultipartRepositoryError::ScanBudgetExhausted) + } + + /// Lists current, visible part generations in ascending part-number order. + /// + /// # Errors + /// Rejects a closed upload, invalid marker/limit or malformed scan data. + pub async fn list_parts( + &self, + session: &MultipartSessionRecord, + after_number: u16, + max_parts: usize, + ) -> Result { + if max_parts == 0 || max_parts > MAX_LIST_PARTS || after_number > 10_000 { + return Err(MultipartRepositoryError::Conflict); + } + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.phase != MultipartPhase::Open { + return Err(MultipartRepositoryError::Conflict); + } + if current.pending.is_some() { + self.settle_pending_part(¤t).await?; + return Err(MultipartRepositoryError::Busy); + } + if after_number == 10_000 { + return Ok(MultipartPartPage { + parts: Vec::new(), + next_part_number_marker: None, + }); + } + let prefix = MetadataKey::multipart_part_prefix(&self.tenant, current.bucket_id, ¤t.upload_id); + let start = if after_number == 0 { + prefix.clone() + } else { + MetadataKey::multipart_part( + &self.tenant, + current.bucket_id, + ¤t.upload_id, + after_number + 1, + )? + }; + let end = MetadataKey::multipart_part_end(&self.tenant, current.bucket_id, ¤t.upload_id); + let page = self + .store + .scan_page(start, end, max_parts + 1, MAX_SCAN_BYTES, None) + .await?; + let mut parts = Vec::with_capacity(page.items.len().min(max_parts)); + let has_more = page.continuation.is_some() || page.items.len() > max_parts; + for item in page.items.into_iter().take(max_parts) { + let suffix = item + .key + .strip_prefix(prefix.as_slice()) + .ok_or(MultipartRepositoryError::InvalidPart)?; + let number = match suffix { + [high, low] => u16::from_be_bytes([*high, *low]), + _ => return Err(MultipartRepositoryError::InvalidPart), + }; + if number <= after_number + || parts + .last() + .is_some_and(|prior: &MultipartPartRecord| prior.number >= number) + { + return Err(MultipartRepositoryError::InvalidPart); + } + parts.push(MultipartPartRecord::decode( + &item.value, + current.bucket_id, + ¤t.upload_id, + number, + )?); + } + let next_part_number_marker = if has_more { + Some(parts.last().ok_or(MultipartRepositoryError::InvalidPart)?.number) + } else { + None + }; + Ok(MultipartPartPage { + parts, + next_part_number_marker, + }) + } +} diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/parts.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/parts.rs new file mode 100644 index 000000000..36a7ce580 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/parts.rs @@ -0,0 +1,253 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Durable part-pointer reservations shared with completion's session fence. + +use crate::metadata::PendingPartMutation; +use crowdb_access_multipart::{ + live_at, next_part_revision, next_revision, reserve_part_accounting, PartAccounting, +}; +use sha2::{Digest as _, Sha256}; + +use super::{ + MetadataKey, MultipartPartRecord, MultipartPhase, MultipartRepository, MultipartRepositoryError, + MultipartSessionRecord, PutIfAbsentOutcome, +}; + +impl MultipartRepository { + /// Publishes a streamed part only while the upload remains open. + /// + /// The immutable generation is written first. A session CAS then reserves + /// the current-part pointer mutation, so completion cannot freeze a stale + /// pointer while an earlier part writer publishes. An abandoned generation + /// remains available to R95's chunk-centered reclamation. + /// + /// # Errors + /// Rejects closed, expired or foreign sessions, invalid parts and + /// unconfirmed storage errors. A competing writer returns `None`. + pub async fn put_stream_part( + &self, + session: &MultipartSessionRecord, + part: &MultipartPartRecord, + now_ms: u64, + ) -> Result, MultipartRepositoryError> { + for _ in 0..2 { + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.pending.is_some() { + self.settle_pending_part(¤t).await?; + continue; + } + return self.reserve_stream_part(¤t, part, now_ms).await; + } + Ok(None) + } + + async fn reserve_stream_part( + &self, + current: &MultipartSessionRecord, + part: &MultipartPartRecord, + now_ms: u64, + ) -> Result, MultipartRepositoryError> { + if current.phase != MultipartPhase::Open + || !live_at(current.created_ms, current.expires_ms, now_ms) + || part.bucket_id != current.bucket_id + || part.upload_id != current.upload_id + || part.number == 0 + || part.number > current.max_parts + || part.length > current.max_part_bytes + { + return Err(MultipartRepositoryError::Conflict); + } + let before = self.part(current, part.number).await?; + if let Some(existing) = &before { + if existing.length == part.length && existing.raw_md5 == part.raw_md5 { + return Ok(Some(existing.clone())); + } + } + let mut after = part.clone(); + after.revision = next_part_revision(before.as_ref().map(|record| record.revision)) + .ok_or(MultipartRepositoryError::Conflict)?; + after.modified_ms = now_ms; + let accounting = reserve_part_accounting( + PartAccounting { + count: current.part_count, + staged_bytes: current.staged_bytes, + }, + before.as_ref().map(|record| record.length), + after.length, + current.max_parts, + current.max_staged_bytes, + ) + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + if !self.persist_generation(current, &after).await? { + return Ok(None); + } + let mut reserved = current.clone(); + reserved.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; + reserved.part_count = accounting.count; + reserved.staged_bytes = accounting.staged_bytes; + let before_digest = before + .as_ref() + .map(|record| record.encode().map(|value| Sha256::digest(value).into())) + .transpose()?; + reserved.pending = Some(PendingPartMutation { + number: after.number, + before_revision: before.as_ref().map(|record| record.revision), + before_digest, + after_revision: after.revision, + after_digest: Sha256::digest(after.encode()?).into(), + after_length: after.length, + }); + if !self.exchange(current, &reserved).await? { + return Ok(None); + } + self.settle_pending_part(&reserved).await + } + + async fn persist_generation( + &self, + session: &MultipartSessionRecord, + after: &MultipartPartRecord, + ) -> Result { + let key = MetadataKey::multipart_part_generation( + &self.tenant, + session.bucket_id, + &session.upload_id, + after.number, + after.revision, + )?; + let value = after.encode()?; + match self.store.put_if_absent(key.clone(), value.clone()).await { + Ok(PutIfAbsentOutcome::Inserted { .. }) => Ok(true), + Ok(PutIfAbsentOutcome::Existing(existing)) if existing.value == value => Ok(true), + Ok(PutIfAbsentOutcome::Existing(_)) => Ok(false), + Err(error) => { + if self + .store + .get(key) + .await? + .is_some_and(|entry| entry.value == value) + { + Ok(true) + } else { + Err(error.into()) + } + } + } + } + + /// Helps a reserved pointer mutation after response loss or restart. + /// + /// # Errors + /// Rejects changed immutable evidence or an unconfirmed metadata write. + pub async fn settle_pending_part( + &self, + session: &MultipartSessionRecord, + ) -> Result, MultipartRepositoryError> { + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current != *session { + return Ok(None); + } + let pending = current + .pending + .as_ref() + .ok_or(MultipartRepositoryError::Conflict)?; + let after = self + .part_generation(¤t, pending.number, pending.after_revision) + .await? + .ok_or(MultipartRepositoryError::InvalidPart)?; + let digest: [u8; 32] = Sha256::digest(after.encode()?).into(); + if after.length != pending.after_length || digest != pending.after_digest { + return Err(MultipartRepositoryError::InvalidPart); + } + self.publish_reserved_pointer(¤t, pending, &after).await?; + let mut settled = current.clone(); + settled.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; + settled.pending = None; + if self.exchange(¤t, &settled).await? { + return Ok(Some(after)); + } + let latest = self + .load(¤t) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if latest.pending.is_none() && self.part(&latest, after.number).await?.as_ref() == Some(&after) { + Ok(Some(after)) + } else { + Ok(None) + } + } + + async fn publish_reserved_pointer( + &self, + session: &MultipartSessionRecord, + pending: &PendingPartMutation, + after: &MultipartPartRecord, + ) -> Result<(), MultipartRepositoryError> { + let current = self.part(session, pending.number).await?; + if current.as_ref() == Some(after) { + return Ok(()); + } + let before_digest = current + .as_ref() + .map(|record| record.encode().map(|value| Sha256::digest(value).into())) + .transpose()?; + if current.as_ref().map(|record| record.revision) != pending.before_revision + || before_digest != pending.before_digest + { + return Err(MultipartRepositoryError::InvalidPart); + } + let key = MetadataKey::multipart_part( + &self.tenant, + session.bucket_id, + &session.upload_id, + pending.number, + )?; + let value = after.encode()?; + let mutation = match current { + Some(before) => { + self.store + .compare_exchange(key.clone(), before.encode()?, value.clone()) + .await + } + None => self + .store + .put_if_absent(key.clone(), value.clone()) + .await + .map(|outcome| matches!(outcome, PutIfAbsentOutcome::Inserted { .. })), + }; + match mutation { + Ok(true) => Ok(()), + Ok(false) => { + if self + .store + .get(key) + .await? + .is_some_and(|entry| entry.value == value) + { + Ok(()) + } else { + Err(MultipartRepositoryError::InvalidPart) + } + } + Err(error) => { + if self + .store + .get(key) + .await? + .is_some_and(|entry| entry.value == value) + { + Ok(()) + } else { + Err(error.into()) + } + } + } + } +} diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs new file mode 100644 index 000000000..76a52fcf3 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs @@ -0,0 +1,138 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Publish the frozen multipart object through one object-key mutation. + +use crowdb_access_multipart::{next_revision, MultipartComposer}; +use sha2::{Digest as _, Sha256}; + +use super::{ + MetadataKey, MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, + PutIfAbsentOutcome, +}; +use crate::integrity::multipart_checksum_marker; +use crate::metadata::ObjectRecord; + +impl MultipartRepository { + /// Publishes selected durable part locations without rereading part bytes. + /// + /// # Errors + /// Rejects changed part generations or object predecessors and unconfirmed + /// storage failures. A retry confirms only an exact published generation. + pub async fn publish_completion( + &self, + session: &MultipartSessionRecord, + ) -> Result { + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.phase == MultipartPhase::Published { + return Ok(current); + } + if current.phase != MultipartPhase::Publishing { + return Err(MultipartRepositoryError::Conflict); + } + let selection = current + .selection + .as_ref() + .ok_or(MultipartRepositoryError::Conflict)?; + let mut composer = MultipartComposer::new(current.max_object_bytes); + for selected in selection { + if self + .part(¤t, selected.number) + .await? + .as_ref() + .map(|part| part.revision) + != Some(selected.revision) + { + return Err(MultipartRepositoryError::InvalidPart); + } + let part = self + .part_generation(¤t, selected.number, selected.revision) + .await? + .ok_or(MultipartRepositoryError::InvalidPart)?; + if part.revision != selected.revision || part.selection_digest()? != selected.digest { + return Err(MultipartRepositoryError::InvalidPart); + } + composer + .push(part.length, part.raw_md5, &part.locations) + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + } + let assembled = composer + .finish() + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + if assembled.length != current.staged_bytes || Some(&assembled.etag) != current.etag.as_ref() { + return Err(MultipartRepositoryError::InvalidPart); + } + let published_at = current.publication_ms.ok_or(MultipartRepositoryError::Conflict)?; + let marker = multipart_checksum_marker(&assembled.etag).ok_or(MultipartRepositoryError::Conflict)?; + let object = ObjectRecord { + bucket_id: current.bucket_id, + key: current.object_key.clone(), + logical_length: assembled.length, + checksum: marker.to_vec(), + etag: assembled.etag, + created_at_ms: published_at, + modified_at_ms: published_at, + content_type: current.content_type.clone(), + attributes: Vec::new(), + data_reference: bincode::serialize(&assembled.locations) + .map_err(|_| MultipartRepositoryError::Conflict)?, + data_length: assembled.length, + }; + let value = object.encode().map_err(|_| MultipartRepositoryError::Conflict)?; + let key = MetadataKey::object(&self.tenant, current.bucket_id, ¤t.object_key)?; + let observed = self.store.get(key.clone()).await?; + if observed.as_ref().is_some_and(|entry| entry.value == value) { + return self.finish_publication(¤t).await; + } + let expected_digest = current + .object_predecessor + .ok_or(MultipartRepositoryError::Conflict)?; + if observed.as_ref().map(|entry| Sha256::digest(&entry.value).into()) != expected_digest { + return Err(MultipartRepositoryError::Conflict); + } + let mutation = if let Some(before) = observed { + self.store + .compare_exchange(key.clone(), before.value, value.clone()) + .await + } else { + self.store + .put_if_absent(key.clone(), value.clone()) + .await + .map(|outcome| matches!(outcome, PutIfAbsentOutcome::Inserted { .. })) + }; + match mutation { + Ok(true) => {} + Ok(false) => { + if self.store.get(key).await?.as_ref().map(|entry| &entry.value) != Some(&value) { + return Err(MultipartRepositoryError::Conflict); + } + } + Err(error) => { + if self.store.get(key).await?.as_ref().map(|entry| &entry.value) != Some(&value) { + return Err(error.into()); + } + } + } + self.finish_publication(¤t).await + } + + async fn finish_publication( + &self, + current: &MultipartSessionRecord, + ) -> Result { + let mut next = current.clone(); + next.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; + next.phase = MultipartPhase::Published; + if self.exchange(current, &next).await? { + Ok(next) + } else { + self.load(current) + .await? + .filter(|record| record.phase == MultipartPhase::Published) + .ok_or(MultipartRepositoryError::Conflict) + } + } +} diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs new file mode 100644 index 000000000..1d09f075c --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs @@ -0,0 +1,135 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Terminal multipart session transitions. + +use super::{ + MetadataKey, MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, +}; +use crate::metadata::BucketId; +use crowdb_access_multipart::next_revision; + +const MAX_EXPIRY_PAGE: usize = 1_000; +const MAX_EXPIRY_SCAN_BYTES: usize = 4 * 1024 * 1024; + +pub struct MultipartExpiryPage { + pub next: Option>, + pub expired: usize, +} + +impl MultipartRepository { + /// Marks expired open uploads terminal while retaining all part evidence. + /// + /// Returns a key cursor for the next bounded listing-index page. A worker + /// can resume from that key without an additional per-upload cleanup queue. + /// + /// # Errors + /// Rejects an invalid cursor, corrupt index data or unavailable metadata. + pub async fn expire_page( + &self, + bucket: BucketId, + after: Option<&[u8]>, + now_ms: u64, + max_items: usize, + ) -> Result { + if max_items == 0 || max_items > MAX_EXPIRY_PAGE { + return Err(MultipartRepositoryError::Conflict); + } + let prefix = MetadataKey::multipart_session_prefix(&self.tenant, bucket); + let end = MetadataKey::multipart_session_end(&self.tenant, bucket); + let start = match after { + Some(after) + if after.starts_with(&prefix) && after.len() > prefix.len() && after < end.as_slice() => + { + let mut next = after.to_vec(); + next.push(0); + next + } + Some(_) => return Err(MultipartRepositoryError::Conflict), + None => prefix, + }; + let page = self + .store + .scan_page(start, end, max_items, MAX_EXPIRY_SCAN_BYTES, None) + .await?; + let next = if page.continuation.is_some() { + Some( + page.items + .last() + .ok_or(MultipartRepositoryError::Conflict)? + .key + .clone(), + ) + } else { + None + }; + let mut expired = 0; + for index in page.items { + let Some(session_value) = self.store.get(index.value).await? else { + continue; + }; + let session = MultipartSessionRecord::decode_unbound(&session_value.value)?; + if session.bucket_id != bucket + || MetadataKey::multipart_upload_index( + &self.tenant, + bucket, + &session.object_key, + &session.upload_id, + )? != index.key + || MetadataKey::multipart_session(&self.tenant, bucket, &session.upload_id) + != session_value.key + { + return Err(MultipartRepositoryError::Conflict); + } + if session.phase == MultipartPhase::Open && session.expires_ms <= now_ms { + match self.abort(&session).await { + Ok(_) => expired += 1, + Err(MultipartRepositoryError::Conflict) => {} + Err(error) => return Err(error), + } + } + } + Ok(MultipartExpiryPage { next, expired }) + } + + /// Logically aborts an open upload, retaining part evidence for cleanup. + /// + /// # Errors + /// Rejects an upload whose completion or publication already won. + pub async fn abort( + &self, + session: &MultipartSessionRecord, + ) -> Result { + let mut current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.pending.is_some() { + self.settle_pending_part(¤t).await?; + current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.pending.is_some() { + return Err(MultipartRepositoryError::Busy); + } + } + if current.phase == MultipartPhase::Aborted { + return Ok(current); + } + if current.phase != MultipartPhase::Open { + return Err(MultipartRepositoryError::Conflict); + } + let mut next = current.clone(); + next.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; + next.phase = MultipartPhase::Aborted; + if self.exchange(¤t, &next).await? { + Ok(next) + } else { + self.load(session) + .await? + .filter(|record| record.phase == MultipartPhase::Aborted) + .ok_or(MultipartRepositoryError::Conflict) + } + } +} diff --git a/lib/crowdb-access-s3/src/metadata/record.rs b/lib/crowdb-access-s3/src/metadata/record.rs index 814cd6b5d..4cac5e5de 100644 --- a/lib/crowdb-access-s3/src/metadata/record.rs +++ b/lib/crowdb-access-s3/src/metadata/record.rs @@ -172,6 +172,10 @@ fn validate_object(record: &ObjectRecord) -> Result<(), MetadataRecordError> { required("etag", record.etag.as_bytes())?; required("content_type", record.content_type.as_bytes())?; required("data_reference", &record.data_reference)?; + if record.checksum.len() == 18 && !crate::integrity::is_multipart_checksum(&record.checksum, &record.etag) + { + return Err(MetadataRecordError::Invalid); + } if record.logical_length != record.data_length { return Err(MetadataRecordError::DataLength); } diff --git a/lib/crowdb-access-s3/src/metadata/store.rs b/lib/crowdb-access-s3/src/metadata/store.rs index 4e8fcc82a..df583ac4d 100644 --- a/lib/crowdb-access-s3/src/metadata/store.rs +++ b/lib/crowdb-access-s3/src/metadata/store.rs @@ -3,7 +3,9 @@ use std::sync::Arc; -use crowdb_chunk_kv_client::{ChunkKvClient, ClientError, MultiScanPage, MultiScanRequest}; +use crowdb_chunk_kv_client::{ + ChunkKvClient, ClientError, MultiScanContinuation, MultiScanPage, MultiScanRequest, +}; use crowdb_protocol::chunk_kv::{ ChunkKvResponse, OperationResult, RpcCompareCondition, RpcFailure, RpcValue, ScanDirection, }; @@ -182,6 +184,24 @@ impl ChunkKvMetadataStore { max_items: usize, max_bytes: usize, ) -> Result, MetadataStoreError> { + Ok(self + .scan_page(start, end, max_items, max_bytes, None) + .await? + .items) + } + + /// Scans a bounded page while preserving its exact continuation fence. + /// + /// # Errors + /// Returns routing, storage or terminal scan failures. + pub async fn scan_page( + &self, + start: Vec, + end: Vec, + max_items: usize, + max_bytes: usize, + continuation: Option, + ) -> Result { let page = self .client .scan(MultiScanRequest { @@ -190,13 +210,13 @@ impl ChunkKvMetadataStore { direction: ScanDirection::Forward, max_items, max_bytes, - continuation: None, + continuation, }) .await?; - if let Some(terminal_failure) = page.terminal_failure { - return Err(failure(&terminal_failure)); + if let Some(terminal_failure) = &page.terminal_failure { + return Err(failure(terminal_failure)); } - Ok(page.items) + Ok(page) } } diff --git a/lib/crowdb-access-s3/src/metrics.rs b/lib/crowdb-access-s3/src/metrics.rs index 4084ab3ce..473430e6d 100644 --- a/lib/crowdb-access-s3/src/metrics.rs +++ b/lib/crowdb-access-s3/src/metrics.rs @@ -13,7 +13,7 @@ use crowdb_chunk_client::{ use crate::native_buffer::{NativeBodyAllocator, NativeBufferMetricsSnapshot}; -const OPERATION_COUNT: usize = 9; +const OPERATION_COUNT: usize = 15; const OUTCOME_COUNT: usize = 6; const OPERATION_NAMES: [&str; OPERATION_COUNT] = [ "create_bucket", @@ -25,6 +25,12 @@ const OPERATION_NAMES: [&str; OPERATION_COUNT] = [ "get_object", "list_objects_v2", "delete_object", + "create_multipart_upload", + "upload_part", + "list_parts", + "complete_multipart_upload", + "abort_multipart_upload", + "list_multipart_uploads", ]; const OUTCOME_NAMES: [&str; OUTCOME_COUNT] = [ "success", diff --git a/lib/crowdb-access-s3/src/retrieval.rs b/lib/crowdb-access-s3/src/retrieval.rs index 732e24361..901d6a298 100644 --- a/lib/crowdb-access-s3/src/retrieval.rs +++ b/lib/crowdb-access-s3/src/retrieval.rs @@ -126,7 +126,9 @@ pub fn prepare_get( inner: client .read_range_stream(&locations, interval.start, interval.end) .map_err(|_| RetrievalError::ChunkRead)?, - integrity: requested.is_none().then(SinglePartIntegrity::default), + integrity: (requested.is_none() + && !crate::integrity::is_multipart_checksum(&record.checksum, &record.etag)) + .then(SinglePartIntegrity::default), expected_checksum: record.checksum.clone(), terminal: false, }) diff --git a/lib/crowdb-access-s3/src/route.rs b/lib/crowdb-access-s3/src/route.rs index 9562d1595..afe295a3c 100644 --- a/lib/crowdb-access-s3/src/route.rs +++ b/lib/crowdb-access-s3/src/route.rs @@ -6,6 +6,10 @@ use hyper::{HeaderMap, Method, Uri}; use percent_encoding::percent_decode_str; +mod multipart; + +pub use multipart::{classify_multipart, MultipartOperation, MultipartRoute}; + #[repr(usize)] #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub enum S3Operation { @@ -18,6 +22,12 @@ pub enum S3Operation { GetObject, ListObjectsV2, DeleteObject, + CreateMultipartUpload, + UploadPart, + ListParts, + CompleteMultipartUpload, + AbortMultipartUpload, + ListMultipartUploads, } #[derive(Clone, Debug, Eq, PartialEq)] @@ -25,6 +35,8 @@ pub struct S3Route { pub operation: S3Operation, pub bucket: Option>, pub key: Option>, + pub upload_id: Option<[u8; 16]>, + pub part_number: Option, } #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -50,6 +62,8 @@ pub fn classify(method: &Method, uri: &Uri) -> Result { operation: S3Operation::ListBuckets, bucket: None, key: None, + upload_id: None, + part_number: None, }) .ok_or(RouteError::Invalid); } @@ -76,6 +90,8 @@ pub fn classify(method: &Method, uri: &Uri) -> Result { operation, bucket: Some(bucket), key, + upload_id: None, + part_number: None, }) } @@ -107,6 +123,23 @@ pub fn classify_request(method: &Method, uri: &Uri, headers: &HeaderMap) -> Resu }) { return Err(RouteError::NotImplemented); } + if let Some(multipart) = classify_multipart(method, uri)? { + let operation = match multipart.operation { + MultipartOperation::Create => S3Operation::CreateMultipartUpload, + MultipartOperation::UploadPart => S3Operation::UploadPart, + MultipartOperation::ListParts => S3Operation::ListParts, + MultipartOperation::Complete => S3Operation::CompleteMultipartUpload, + MultipartOperation::Abort => S3Operation::AbortMultipartUpload, + MultipartOperation::ListUploads => S3Operation::ListMultipartUploads, + }; + return Ok(S3Route { + operation, + bucket: Some(multipart.bucket), + key: multipart.key, + upload_id: multipart.upload_id, + part_number: multipart.part_number, + }); + } classify(method, uri) } @@ -122,6 +155,7 @@ fn selects_extension(query: Option<&str>) -> bool { name, "uploads" | "uploadId" + | "partNumber" | "versionId" | "tagging" | "lifecycle" diff --git a/lib/crowdb-access-s3/src/route/multipart.rs b/lib/crowdb-access-s3/src/route/multipart.rs new file mode 100644 index 000000000..e6df6d665 --- /dev/null +++ b/lib/crowdb-access-s3/src/route/multipart.rs @@ -0,0 +1,103 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Parse the authenticated S3 multipart query surface. + +use hyper::{Method, Uri}; + +use super::{decode, RouteError}; + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum MultipartOperation { + Create, + UploadPart, + ListParts, + Complete, + Abort, + ListUploads, +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct MultipartRoute { + pub operation: MultipartOperation, + pub bucket: Vec, + pub key: Option>, + pub upload_id: Option<[u8; 16]>, + pub part_number: Option, +} + +/// Classifies a multipart query after the common `SigV4` authentication step. +/// +/// # Errors +/// Rejects duplicate selectors, malformed identities and unsupported method +/// or path combinations. Returns `None` for ordinary basic S3 requests. +pub fn classify_multipart(method: &Method, uri: &Uri) -> Result, RouteError> { + let mut uploads = false; + let mut upload_id = None; + let mut part_number = None; + for pair in uri.query().unwrap_or_default().split('&') { + let (name, value) = pair.split_once('=').unwrap_or((pair, "")); + match name { + "uploads" if !uploads && value.is_empty() => uploads = true, + "uploadId" if upload_id.is_none() => upload_id = Some(parse_upload_id(value)?), + "partNumber" if part_number.is_none() => { + let number = value.parse::().map_err(|_| RouteError::Invalid)?; + if number == 0 || number > 10_000 { + return Err(RouteError::Invalid); + } + part_number = Some(number); + } + "uploads" | "uploadId" | "partNumber" => return Err(RouteError::Invalid), + _ => {} + } + } + if !uploads && upload_id.is_none() && part_number.is_some() { + return Err(RouteError::Invalid); + } + if !uploads && upload_id.is_none() { + return Ok(None); + } + if uploads && (upload_id.is_some() || part_number.is_some()) { + return Err(RouteError::Invalid); + } + let path = uri.path().strip_prefix('/').ok_or(RouteError::Invalid)?; + let (bucket, key) = path + .split_once('/') + .map_or((path, None), |(bucket, key)| (bucket, Some(key))); + let bucket = decode(bucket); + if bucket.is_empty() { + return Err(RouteError::Invalid); + } + let key = key.map(decode); + let operation = match (method, key.as_deref(), uploads, upload_id, part_number) { + (&Method::POST, Some(_), true, None, None) => MultipartOperation::Create, + (&Method::GET, None, true, None, None) => MultipartOperation::ListUploads, + (&Method::PUT, Some(_), false, Some(_), Some(_)) => MultipartOperation::UploadPart, + (&Method::GET, Some(_), false, Some(_), None) => MultipartOperation::ListParts, + (&Method::POST, Some(_), false, Some(_), None) => MultipartOperation::Complete, + (&Method::DELETE, Some(_), false, Some(_), None) => MultipartOperation::Abort, + _ => return Err(RouteError::Invalid), + }; + Ok(Some(MultipartRoute { + operation, + bucket, + key, + upload_id, + part_number, + })) +} + +fn parse_upload_id(value: &str) -> Result<[u8; 16], RouteError> { + if value.len() != 32 || !value.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(RouteError::Invalid); + } + let mut id = [0_u8; 16]; + for (output, pair) in id.iter_mut().zip(value.as_bytes().chunks_exact(2)) { + *output = u8::from_str_radix(std::str::from_utf8(pair).map_err(|_| RouteError::Invalid)?, 16) + .map_err(|_| RouteError::Invalid)?; + } + if id == [0; 16] { + return Err(RouteError::Invalid); + } + Ok(id) +} diff --git a/lib/crowdb-access-s3/src/wire.rs b/lib/crowdb-access-s3/src/wire.rs index 532f7e1aa..0fcffd59e 100644 --- a/lib/crowdb-access-s3/src/wire.rs +++ b/lib/crowdb-access-s3/src/wire.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crate::metadata::{BucketNameRecord, ObjectRecord}; +use crate::metadata::{BucketNameRecord, MultipartPartPage, MultipartUploadPage, ObjectRecord}; use crate::object::ListObjectsV2Page; #[must_use] @@ -22,6 +22,129 @@ pub fn list_buckets(tenant: &[u8], buckets: &[BucketNameRecord]) -> String { output } +#[must_use] +pub fn create_multipart_upload(bucket: &[u8], key: &[u8], upload_id: &[u8; 16]) -> String { + let mut output = xml_start("InitiateMultipartUploadResult"); + element(&mut output, "Bucket", &String::from_utf8_lossy(bucket)); + element(&mut output, "Key", &String::from_utf8_lossy(key)); + element(&mut output, "UploadId", &hex_upload_id(upload_id)); + output.push_str(""); + output +} + +#[must_use] +pub fn complete_multipart_upload(location: &str, bucket: &[u8], key: &[u8], etag: &str) -> String { + let mut output = xml_start("CompleteMultipartUploadResult"); + element(&mut output, "Location", location); + element(&mut output, "Bucket", &String::from_utf8_lossy(bucket)); + element(&mut output, "Key", &String::from_utf8_lossy(key)); + element(&mut output, "ETag", &format!("\"{etag}\"")); + output.push_str(""); + output +} + +#[must_use] +pub fn list_multipart_parts( + bucket: &[u8], + key: &[u8], + upload_id: &[u8; 16], + marker: u16, + max_parts: usize, + page: &MultipartPartPage, +) -> String { + let mut output = xml_start("ListPartsResult"); + element(&mut output, "Bucket", &String::from_utf8_lossy(bucket)); + element(&mut output, "Key", &String::from_utf8_lossy(key)); + element(&mut output, "UploadId", &hex_upload_id(upload_id)); + element(&mut output, "PartNumberMarker", &marker.to_string()); + if let Some(next) = page.next_part_number_marker { + element(&mut output, "NextPartNumberMarker", &next.to_string()); + } + element(&mut output, "MaxParts", &max_parts.to_string()); + element( + &mut output, + "IsTruncated", + if page.next_part_number_marker.is_some() { + "true" + } else { + "false" + }, + ); + for part in &page.parts { + output.push_str(""); + element(&mut output, "PartNumber", &part.number.to_string()); + element(&mut output, "LastModified", &iso8601(part.modified_ms)); + element( + &mut output, + "ETag", + &format!("\"{:x}\"", md5::Digest(part.raw_md5)), + ); + element(&mut output, "Size", &part.length.to_string()); + output.push_str(""); + } + output.push_str(""); + output +} + +#[must_use] +pub fn list_multipart_uploads( + bucket: &[u8], + prefix: &[u8], + key_marker: Option<&[u8]>, + upload_marker: Option<&[u8; 16]>, + max_uploads: usize, + owner_id: &str, + page: &MultipartUploadPage, +) -> String { + let mut output = xml_start("ListMultipartUploadsResult"); + element(&mut output, "Bucket", &String::from_utf8_lossy(bucket)); + element( + &mut output, + "KeyMarker", + &String::from_utf8_lossy(key_marker.unwrap_or_default()), + ); + element( + &mut output, + "UploadIdMarker", + &upload_marker.map(hex_upload_id).unwrap_or_default(), + ); + if let Some((key, id)) = &page.next { + element(&mut output, "NextKeyMarker", &String::from_utf8_lossy(key)); + element(&mut output, "NextUploadIdMarker", &hex_upload_id(id)); + } + element(&mut output, "Prefix", &String::from_utf8_lossy(prefix)); + element(&mut output, "MaxUploads", &max_uploads.to_string()); + element( + &mut output, + "IsTruncated", + if page.next.is_some() { "true" } else { "false" }, + ); + for session in &page.uploads { + output.push_str(""); + element(&mut output, "Key", &String::from_utf8_lossy(&session.object_key)); + element(&mut output, "UploadId", &hex_upload_id(&session.upload_id)); + output.push_str(""); + element(&mut output, "ID", owner_id); + output.push_str(""); + element(&mut output, "ID", owner_id); + output.push_str(""); + element(&mut output, "StorageClass", "STANDARD"); + element(&mut output, "Initiated", &iso8601(session.created_ms)); + output.push_str(""); + } + output.push_str(""); + output +} + +fn hex_upload_id(upload_id: &[u8; 16]) -> String { + use std::fmt::Write as _; + let mut result = String::with_capacity(32); + for byte in upload_id { + write!(&mut result, "{byte:02x}").expect("string write cannot fail"); + } + result +} + #[must_use] pub fn list_objects( bucket: &[u8], diff --git a/lib/crowdb-access-s3/tests/error_test.rs b/lib/crowdb-access-s3/tests/error_test.rs index 1a49bc3bb..1a12dd1f9 100644 --- a/lib/crowdb-access-s3/tests/error_test.rs +++ b/lib/crowdb-access-s3/tests/error_test.rs @@ -52,6 +52,18 @@ fn every_public_error_class_has_a_stable_status_and_code() { ), (S3ErrorCode::NoSuchBucket, StatusCode::NOT_FOUND, "NoSuchBucket"), (S3ErrorCode::NoSuchKey, StatusCode::NOT_FOUND, "NoSuchKey"), + (S3ErrorCode::NoSuchUpload, StatusCode::NOT_FOUND, "NoSuchUpload"), + (S3ErrorCode::InvalidPart, StatusCode::BAD_REQUEST, "InvalidPart"), + ( + S3ErrorCode::InvalidPartOrder, + StatusCode::BAD_REQUEST, + "InvalidPartOrder", + ), + ( + S3ErrorCode::EntityTooSmall, + StatusCode::BAD_REQUEST, + "EntityTooSmall", + ), ( S3ErrorCode::BucketNotEmpty, StatusCode::CONFLICT, diff --git a/lib/crowdb-access-s3/tests/integrity_test.rs b/lib/crowdb-access-s3/tests/integrity_test.rs index 9588e569b..c03c729e9 100644 --- a/lib/crowdb-access-s3/tests/integrity_test.rs +++ b/lib/crowdb-access-s3/tests/integrity_test.rs @@ -4,6 +4,22 @@ use crowdb_access_s3::integrity::{IntegrityError, SinglePartIntegrity}; use hyper::body::Bytes; +#[test] +fn multipart_marker_binds_composite_digest_and_part_count() { + use crowdb_access_s3::integrity::{is_multipart_checksum, multipart_checksum_marker}; + + let etag = "b4ab393b73e0e71830bf2bf0e63c4d91-2"; + let marker = multipart_checksum_marker(etag).unwrap(); + assert_eq!(marker.len(), 18); + assert_eq!(&marker[16..], &[0, 2]); + assert!(is_multipart_checksum(&marker, etag)); + assert!(!is_multipart_checksum( + &marker, + "b4ab393b73e0e71830bf2bf0e63c4d91-3" + )); + assert!(multipart_checksum_marker("b4ab393b73e0e71830bf2bf0e63c4d91-0").is_none()); +} + #[test] fn single_part_etag_is_independent_of_body_frame_boundaries() { let mut fragmented = SinglePartIntegrity::default(); diff --git a/lib/crowdb-access-s3/tests/metadata_key_test.rs b/lib/crowdb-access-s3/tests/metadata_key_test.rs index c774e3852..76cc3e527 100644 --- a/lib/crowdb-access-s3/tests/metadata_key_test.rs +++ b/lib/crowdb-access-s3/tests/metadata_key_test.rs @@ -61,3 +61,30 @@ fn maximum_binary_object_key_stays_within_its_bucket_interval() { assert!(MetadataKey::object_prefix(&tenant, bucket) < encoded); assert!(encoded < MetadataKey::object_end(&tenant, bucket)); } + +#[test] +fn multipart_keys_group_session_current_parts_and_generations_by_upload() { + let tenant = TenantId::new(b"tenant".to_vec()).unwrap(); + let bucket = BucketId::new([5; 16]); + let upload = [9; 16]; + let index = MetadataKey::multipart_upload_index(&tenant, bucket, b"a\0", &upload).unwrap(); + let start = MetadataKey::multipart_session_prefix(&tenant, bucket); + let end = MetadataKey::multipart_session_end(&tenant, bucket); + assert!(start < index && index < end); + let upload_prefix = MetadataKey::multipart_upload_prefix(&tenant, bucket, &upload); + let session = MetadataKey::multipart_session(&tenant, bucket, &upload); + let part_start = MetadataKey::multipart_part_prefix(&tenant, bucket, &upload); + let part_end = MetadataKey::multipart_part_end(&tenant, bucket, &upload); + let first = MetadataKey::multipart_part(&tenant, bucket, &upload, 1).unwrap(); + let last = MetadataKey::multipart_part(&tenant, bucket, &upload, 10_000).unwrap(); + assert!(part_start < first && first < last && last < part_end); + assert!(MetadataKey::multipart_part(&tenant, bucket, &upload, 0).is_err()); + assert!(MetadataKey::multipart_part(&tenant, bucket, &upload, 10_001).is_err()); + assert!(MetadataKey::object_end(&tenant, bucket) <= start); + assert!(end <= upload_prefix); + assert!(session.starts_with(&upload_prefix)); + assert!(part_start.starts_with(&upload_prefix)); + assert!(session < part_start); + let generation = MetadataKey::multipart_part_generation(&tenant, bucket, &upload, 1, 2).unwrap(); + assert!(part_end < generation && generation.starts_with(&upload_prefix)); +} diff --git a/lib/crowdb-access-s3/tests/metadata_multipart_test.rs b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs new file mode 100644 index 000000000..00aecd014 --- /dev/null +++ b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs @@ -0,0 +1,130 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_multipart::SelectedPart; +use crowdb_access_s3::metadata::{ + new_upload_id, BucketId, MultipartPartRecord, MultipartPhase, MultipartRecordError, + MultipartSessionRecord, +}; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; + +fn session() -> MultipartSessionRecord { + MultipartSessionRecord { + bucket_id: BucketId::new([3; 16]), + object_key: b"key\0part".to_vec(), + upload_id: [7; 16], + revision: 1, + phase: MultipartPhase::Open, + created_ms: 100, + expires_ms: 200, + content_type: "application/octet-stream".into(), + max_parts: 10, + max_part_bytes: 100, + max_object_bytes: 500, + max_staged_bytes: 1_000, + part_count: 0, + staged_bytes: 0, + pending: None, + selection: None, + completion_request_digest: None, + publication_ms: None, + object_predecessor: None, + etag: None, + } +} + +fn part() -> MultipartPartRecord { + MultipartPartRecord { + bucket_id: BucketId::new([3; 16]), + upload_id: [7; 16], + number: 1, + revision: 2, + modified_ms: 150, + length: 5, + raw_md5: [9; 16], + locations: vec![Location { + chunk_id: Some(ChunkId { high: 1, low: 2 }), + offset: 10, + length: 39, + logical_offset: 0, + logical_length: 5, + }], + } +} + +#[test] +fn generated_upload_ids_sort_by_initiation_millisecond() { + let first = new_upload_id(100); + let second = new_upload_id(101); + assert!(first < second); + assert_ne!(first, [0; 16]); + assert_ne!(second, [0; 16]); +} + +#[test] +fn session_and_part_records_round_trip_only_under_their_own_keys() { + let session = session(); + let bytes = session.encode().unwrap(); + assert_eq!( + MultipartSessionRecord::decode(&bytes, session.bucket_id, &session.object_key, &session.upload_id) + .unwrap(), + session + ); + assert_eq!( + MultipartSessionRecord::decode(&bytes, session.bucket_id, b"other", &session.upload_id), + Err(MultipartRecordError::Identity) + ); + + let part = part(); + let bytes = part.encode().unwrap(); + assert_eq!( + MultipartPartRecord::decode(&bytes, part.bucket_id, &part.upload_id, part.number).unwrap(), + part + ); + assert_eq!( + MultipartPartRecord::decode(&bytes, part.bucket_id, &part.upload_id, 2), + Err(MultipartRecordError::Identity) + ); +} + +#[test] +fn completed_session_requires_a_matching_selected_count_and_etag() { + let mut session = session(); + session.phase = MultipartPhase::Publishing; + session.part_count = 1; + session.staged_bytes = 5; + session.selection = Some(vec![SelectedPart { + number: 1, + revision: 2, + digest: [4; 32], + }]); + session.completion_request_digest = Some([5; 32]); + session.publication_ms = Some(150); + session.object_predecessor = Some(None); + session.etag = Some("11111111111111111111111111111111-2".into()); + assert_eq!(session.encode(), Err(MultipartRecordError::Invalid)); + session.etag = Some("11111111111111111111111111111111-1".into()); + assert!(session.encode().is_ok()); + session.phase = MultipartPhase::Open; + assert_eq!(session.encode(), Err(MultipartRecordError::Invalid)); +} + +#[test] +fn malformed_or_unbounded_records_fail_before_exposure() { + let mut invalid_part = part(); + invalid_part.locations[0].logical_offset = 1; + assert_eq!(invalid_part.encode(), Err(MultipartRecordError::Invalid)); + let part = part(); + let mut bytes = part.encode().unwrap(); + bytes.push(0); + assert_eq!( + MultipartPartRecord::decode(&bytes, part.bucket_id, &part.upload_id, part.number), + Err(MultipartRecordError::Invalid) + ); + let oversized = vec![0; 1024 * 1024 + 1]; + assert_eq!( + MultipartSessionRecord::decode(&oversized, session().bucket_id, b"key", &[7; 16]), + Err(MultipartRecordError::TooLarge) + ); +} diff --git a/lib/crowdb-access-s3/tests/metrics_test.rs b/lib/crowdb-access-s3/tests/metrics_test.rs index d74ae8d8d..15c00eb25 100644 --- a/lib/crowdb-access-s3/tests/metrics_test.rs +++ b/lib/crowdb-access-s3/tests/metrics_test.rs @@ -93,7 +93,7 @@ fn exported_request_series_have_fixed_cardinality_and_no_namespace_labels() { .lines() .filter(|line| line.starts_with("crowdb_s3_requests_total{")) .count(), - 9 * 6 + 15 * 6 ); assert!(!rendered.contains("bucket=")); assert!(!rendered.contains("key=")); diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs new file mode 100644 index 000000000..07bf1b95b --- /dev/null +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -0,0 +1,815 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::HashMap; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::Arc; +use std::time::Duration; + +use async_trait::async_trait; +use crowdb_access_multipart::SelectedPart; +use crowdb_access_s3::metadata::{ + BucketId, ChunkKvMetadataStore, CompletionPart, MetadataKey, MultipartPartRecord, MultipartPhase, + MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, ObjectRecord, PendingPartMutation, + TenantId, +}; +use crowdb_chunk_kv_client::{ + ChunkKvClient, ChunkKvRangeCatalogSource, ChunkKvTransport, ClientConfig, ClientError, Result, +}; +use crowdb_protocol::chunk_kv::{ + ChunkKvRangeCatalogEntry, ChunkKvRangeCatalogHead, ChunkKvRangeCatalogPage, ChunkKvRangeCatalogPageRef, + ChunkKvRangeCatalogPartitionState, ChunkKvResponse, Id128, KeyRange, OperationResult, OwnerDescriptor, + PartitionArtifact, PointOperation, PointRequest, RpcCompareCondition, RpcValue, ScanContinuation, + ScanRequest, +}; +use crowdb_protocol::chunk_stream::StreamName; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; +use sha2::{Digest as _, Sha256}; +use tokio::sync::{mpsc, oneshot}; + +struct Catalog(ChunkKvRangeCatalogHead, Vec); + +#[async_trait] +impl ChunkKvRangeCatalogSource for Catalog { + async fn load(&self) -> Result<(ChunkKvRangeCatalogHead, Vec)> { + Ok((self.0.clone(), self.1.clone())) + } +} + +enum ActorRequest { + Point(PointOperation, oneshot::Sender>), + Scan(ScanRequest, oneshot::Sender>), +} + +struct ActorTransport(mpsc::UnboundedSender); + +#[async_trait] +impl ChunkKvTransport for ActorTransport { + async fn point(&self, _: &str, request: &PointRequest) -> Result { + let (response, receiver) = oneshot::channel(); + self.0 + .send(ActorRequest::Point(request.operation.clone(), response)) + .expect("actor is running"); + receiver.await.expect("actor replies") + } + + async fn seek(&self, _: &str, _: &crowdb_protocol::chunk_kv::SeekRequest) -> Result { + unreachable!() + } + + async fn scan(&self, _: &str, request: &ScanRequest) -> Result { + let (response, receiver) = oneshot::channel(); + self.0 + .send(ActorRequest::Scan(request.clone(), response)) + .expect("actor is running"); + receiver.await.expect("actor replies") + } +} + +fn reply(operation: PointOperation, values: &mut HashMap, RpcValue>) -> ChunkKvResponse { + let result = match operation { + PointOperation::Get { key } => OperationResult::Value(values.get(&key).cloned()), + PointOperation::PutIfAbsent { key, value } => { + let observed = values.get(&key).cloned(); + let applied = observed.is_none(); + if applied { + values.insert( + key.clone(), + RpcValue { + key, + value, + revision: 1, + }, + ); + } + OperationResult::Mutation { + applied, + revision: applied.then_some(1), + observed, + } + } + PointOperation::CompareExchange { + key, + condition: RpcCompareCondition::Value(expected), + value, + } => { + let observed = values.get(&key).cloned(); + let applied = observed + .as_ref() + .is_some_and(|observed| observed.value == expected); + let revision = observed.as_ref().map_or(1, |observed| observed.revision + 1); + if applied { + values.insert(key.clone(), RpcValue { key, value, revision }); + } + OperationResult::Mutation { + applied, + revision: applied.then_some(revision), + observed, + } + } + other => panic!("unexpected operation: {other:?}"), + }; + ChunkKvResponse { + map_revision: 1, + journal_position: None, + result: Ok(result), + } +} + +fn scan_reply(request: &ScanRequest, values: &HashMap, RpcValue>) -> ChunkKvResponse { + let mut ordered: Vec = values + .values() + .filter(|entry| { + request.start.as_ref().map_or(true, |start| entry.key >= *start) + && request.end.as_ref().map_or(true, |end| entry.key < *end) + && request + .continuation + .as_ref() + .map_or(true, |token| entry.key > token.last_key) + }) + .cloned() + .collect(); + ordered.sort_by(|left, right| left.key.cmp(&right.key)); + let limit = usize::try_from(request.limit).unwrap(); + let truncated = ordered.len() > limit; + ordered.truncate(limit); + let continuation = if truncated { + Some(ScanContinuation { + direction: request.direction, + last_key: ordered.last().unwrap().key.clone(), + partition_id: request.routing.partition_id, + owner_epoch: request.routing.owner_epoch, + map_revision: request.routing.map_revision, + }) + } else { + None + }; + ChunkKvResponse { + map_revision: 1, + journal_position: None, + result: Ok(OperationResult::Scan { + items: ordered, + continuation, + }), + } +} + +async fn repository() -> (MultipartRepository, Arc, Arc) { + let (sender, mut receiver) = mpsc::unbounded_channel::(); + let lose_reply = Arc::new(AtomicBool::new(false)); + let actor_lose_reply = Arc::clone(&lose_reply); + tokio::spawn(async move { + let mut values = HashMap::new(); + while let Some(request) = receiver.recv().await { + match request { + ActorRequest::Point(operation, response) => { + let is_mutation = !matches!(operation, PointOperation::Get { .. }); + let result = reply(operation, &mut values); + if is_mutation && actor_lose_reply.swap(false, Ordering::SeqCst) { + let _ = response.send(Err(ClientError::Transport("committed reply lost".into()))); + } else { + let _ = response.send(Ok(result)); + } + } + ActorRequest::Scan(request, response) => { + let _ = response.send(Ok(scan_reply(&request, &values))); + } + } + } + }); + let mut page = ChunkKvRangeCatalogPage { + generation: 1, + page_index: 0, + entries: vec![ChunkKvRangeCatalogEntry { + partition_id: Id128 { high: 1, low: 1 }, + range: KeyRange { + start: Vec::new(), + end: None, + }, + owner: OwnerDescriptor { + instance_id: 1, + rpc_endpoint: "owner".into(), + }, + owner_epoch: 1, + state: ChunkKvRangeCatalogPartitionState::Serving, + artifact: PartitionArtifact { + tree_id: 1, + stream_name: StreamName { high: 1, low: 1 }, + tail_overlay: None, + }, + transition_id: None, + }], + checksum: [0; 32], + }; + page.seal().unwrap(); + let mut head = ChunkKvRangeCatalogHead { + generation: 1, + previous_generation: None, + pages: vec![ChunkKvRangeCatalogPageRef { + page_generation: 1, + page_index: 0, + first_key: Vec::new(), + page_checksum: page.checksum, + }], + checksum: [0; 32], + }; + head.seal().unwrap(); + let client = Arc::new( + ChunkKvClient::new( + ClientConfig { + operation_timeout: Duration::from_secs(1), + retry_backoff: Duration::from_millis(1), + ..ClientConfig::default() + }, + Arc::new(Catalog(head, vec![page])), + Arc::new(ActorTransport(sender)), + ) + .unwrap(), + ); + client.refresh_catalog().await.unwrap(); + let store = Arc::new(ChunkKvMetadataStore::new(client)); + ( + MultipartRepository::new(Arc::clone(&store), TenantId::new(b"tenant".to_vec()).unwrap()), + lose_reply, + store, + ) +} + +fn session() -> MultipartSessionRecord { + MultipartSessionRecord { + bucket_id: BucketId::new([3; 16]), + object_key: b"object".to_vec(), + upload_id: [7; 16], + revision: 1, + phase: MultipartPhase::Open, + created_ms: 100, + expires_ms: 200, + content_type: "application/octet-stream".into(), + max_parts: 10, + max_part_bytes: 100, + max_object_bytes: 500, + max_staged_bytes: 1_000, + part_count: 0, + staged_bytes: 0, + pending: None, + selection: None, + completion_request_digest: None, + publication_ms: None, + object_predecessor: None, + etag: None, + } +} + +fn part() -> MultipartPartRecord { + MultipartPartRecord { + bucket_id: BucketId::new([3; 16]), + upload_id: [7; 16], + number: 1, + revision: 1, + modified_ms: 0, + length: 5, + raw_md5: [9; 16], + locations: vec![Location { + chunk_id: Some(ChunkId { high: 1, low: 2 }), + offset: 10, + length: 39, + logical_offset: 0, + logical_length: 5, + }], + } +} + +#[tokio::test] +async fn session_cas_and_independent_part_replacement_obey_the_freeze() { + let (repository, lose_reply, _) = repository().await; + let session = session(); + assert_eq!(repository.begin(&session).await.unwrap(), session); + assert_eq!(repository.begin(&session).await.unwrap(), session); + let mut changed = session.clone(); + changed.content_type = "text/plain".into(); + assert!(matches!( + repository.begin(&changed).await, + Err(MultipartRepositoryError::Conflict) + )); + + let first = repository + .put_stream_part(&session, &part(), 110) + .await + .unwrap() + .unwrap(); + assert_eq!(first.revision, 1); + let replay = repository + .put_stream_part(&session, &part(), 111) + .await + .unwrap() + .unwrap(); + assert_eq!(replay, first); + let mut relocated_retry = part(); + relocated_retry.locations[0].offset += 10; + assert_eq!( + repository + .put_stream_part(&session, &relocated_retry, 111) + .await + .unwrap(), + Some(first.clone()) + ); + let mut replacement = part(); + replacement.locations[0].offset += 39; + replacement.raw_md5 = [8; 16]; + let second = repository + .put_stream_part(&session, &replacement, 112) + .await + .unwrap() + .unwrap(); + assert_eq!(second.revision, 2); + assert_eq!(repository.part(&session, 1).await.unwrap(), Some(second)); + assert_eq!( + repository.part_generation(&session, 1, 1).await.unwrap(), + Some(first) + ); + + let current = repository.load(&session).await.unwrap().unwrap(); + let mut frozen = current.clone(); + frozen.revision += 1; + frozen.phase = MultipartPhase::Completing; + frozen.part_count = 1; + frozen.staged_bytes = 5; + frozen.selection = Some(vec![SelectedPart { + number: 1, + revision: 2, + digest: [4; 32], + }]); + frozen.completion_request_digest = Some([5; 32]); + frozen.publication_ms = None; + lose_reply.store(true, Ordering::SeqCst); + assert!(repository.exchange(¤t, &frozen).await.unwrap()); + assert!(repository.exchange(¤t, &frozen).await.unwrap()); + assert!(matches!( + repository.put_stream_part(&session, &part(), 112).await, + Err(MultipartRepositoryError::Conflict) + )); +} + +#[tokio::test] +async fn completion_freezes_exact_part_revision_and_replays_the_same_request() { + let (repository, _, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + let first = repository + .put_stream_part(&session, &part(), 110) + .await + .unwrap() + .unwrap(); + let requested = [CompletionPart { + number: 1, + etag: format!("\"{}\"", "09".repeat(16)), + }]; + let frozen = repository + .freeze_completion(&session, &requested, 120) + .await + .unwrap() + .unwrap(); + assert_eq!(frozen.phase, MultipartPhase::Publishing); + assert_eq!(frozen.selection.as_ref().unwrap()[0].revision, first.revision); + assert!(frozen.etag.as_ref().unwrap().ends_with("-1")); + assert_eq!( + repository + .freeze_completion(&session, &requested, 121) + .await + .unwrap(), + Some(frozen.clone()) + ); + assert!(matches!( + repository.put_stream_part(&session, &part(), 122).await, + Err(MultipartRepositoryError::Conflict) + )); + let published = repository.publish_completion(&frozen).await.unwrap(); + assert_eq!(published.phase, MultipartPhase::Published); + assert_eq!(repository.publish_completion(&session).await.unwrap(), published); + let key = MetadataKey::object( + &TenantId::new(b"tenant".to_vec()).unwrap(), + session.bucket_id, + &session.object_key, + ) + .unwrap(); + let object = ObjectRecord::decode(&store.get(key).await.unwrap().unwrap().value).unwrap(); + assert_eq!(object.logical_length, 5); + assert_eq!(object.etag, frozen.etag.unwrap()); + assert_eq!(object.checksum.len(), 18); + let locations: Vec = bincode::deserialize(&object.data_reference).unwrap(); + assert_eq!(locations, part().locations); +} + +#[tokio::test] +async fn completion_settles_an_interrupted_part_reservation_before_freezing() { + let (repository, _, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + let first = repository + .put_stream_part(&session, &part(), 110) + .await + .unwrap() + .unwrap(); + let mut replacement = first.clone(); + replacement.revision += 1; + replacement.modified_ms = 111; + replacement.raw_md5 = [8; 16]; + replacement.locations[0].offset += 39; + let generation_key = MetadataKey::multipart_part_generation( + &TenantId::new(b"tenant".to_vec()).unwrap(), + session.bucket_id, + &session.upload_id, + 1, + replacement.revision, + ) + .unwrap(); + store + .put_if_absent(generation_key, replacement.encode().unwrap()) + .await + .unwrap(); + let current = repository.load(&session).await.unwrap().unwrap(); + let mut reserved = current.clone(); + reserved.revision += 1; + reserved.pending = Some(PendingPartMutation { + number: 1, + before_revision: Some(first.revision), + before_digest: Some(Sha256::digest(first.encode().unwrap()).into()), + after_revision: replacement.revision, + after_digest: Sha256::digest(replacement.encode().unwrap()).into(), + after_length: replacement.length, + }); + assert!(repository.exchange(¤t, &reserved).await.unwrap()); + let request = [CompletionPart { + number: 1, + etag: "08".repeat(16), + }]; + assert!(repository + .freeze_completion(&session, &request, 120) + .await + .unwrap() + .is_none()); + assert_eq!( + repository.part(&session, 1).await.unwrap(), + Some(replacement.clone()) + ); + assert!(repository + .load(&session) + .await + .unwrap() + .unwrap() + .pending + .is_none()); + let frozen = repository + .freeze_completion(&session, &request, 120) + .await + .unwrap() + .unwrap(); + assert_eq!( + frozen.selection.as_ref().unwrap()[0].revision, + replacement.revision + ); +} + +#[tokio::test] +async fn completion_rejects_undersized_nonfinal_and_wrong_etag() { + let (repository, _, _) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + let mut second = part(); + second.number = 2; + repository.put_stream_part(&session, &second, 111).await.unwrap(); + let etag = "09".repeat(16); + let request = [ + CompletionPart { + number: 1, + etag: etag.clone(), + }, + CompletionPart { number: 2, etag }, + ]; + assert!(matches!( + repository.freeze_completion(&session, &request, 120).await, + Err(MultipartRepositoryError::EntityTooSmall) + )); + let wrong = [CompletionPart { + number: 1, + etag: "00".repeat(16), + }]; + assert!(matches!( + repository.freeze_completion(&session, &wrong, 120).await, + Err(MultipartRepositoryError::InvalidPart) + )); + let current = repository.load(&session).await.unwrap().unwrap(); + assert_eq!(current.phase, MultipartPhase::Open); + assert_eq!(current.part_count, 2); + assert!(current.pending.is_none()); +} + +#[tokio::test] +async fn publication_recovers_a_lost_reply_without_replacing_a_competing_object() { + let (repository, lose_reply, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + let request = [CompletionPart { + number: 1, + etag: "09".repeat(16), + }]; + let frozen = repository + .freeze_completion(&session, &request, 120) + .await + .unwrap() + .unwrap(); + lose_reply.store(true, Ordering::SeqCst); + let published = repository.publish_completion(&frozen).await.unwrap(); + assert_eq!(published.phase, MultipartPhase::Published); + assert_eq!(repository.publish_completion(&frozen).await.unwrap(), published); + + let mut second = session.clone(); + second.upload_id = [8; 16]; + repository.begin(&second).await.unwrap(); + repository + .put_stream_part(&second, &part_for(&second), 130) + .await + .unwrap(); + let frozen_second = repository + .freeze_completion(&second, &request, 140) + .await + .unwrap() + .unwrap(); + let key = MetadataKey::object( + &TenantId::new(b"tenant".to_vec()).unwrap(), + second.bucket_id, + &second.object_key, + ) + .unwrap(); + let previous = store.get(key.clone()).await.unwrap().unwrap(); + assert!(store + .compare_exchange(key.clone(), previous.value, b"competing generation".to_vec()) + .await + .unwrap()); + assert!(matches!( + repository.publish_completion(&frozen_second).await, + Err(MultipartRepositoryError::Conflict) + )); + assert_eq!( + store.get(key).await.unwrap().unwrap().value, + b"competing generation" + ); +} + +#[tokio::test] +async fn frozen_part_generation_rejects_a_late_pointer_change() { + let (repository, _, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + let mut replacement = part(); + replacement.locations[0].offset += 39; + replacement.raw_md5 = [8; 16]; + let selected = repository + .put_stream_part(&session, &replacement, 111) + .await + .unwrap() + .unwrap(); + let request = [CompletionPart { + number: 1, + etag: "08".repeat(16), + }]; + let frozen = repository + .freeze_completion(&session, &request, 120) + .await + .unwrap() + .unwrap(); + let pointer_key = MetadataKey::multipart_part( + &TenantId::new(b"tenant".to_vec()).unwrap(), + session.bucket_id, + &session.upload_id, + 1, + ) + .unwrap(); + let mut late = selected.clone(); + late.revision = 3; + late.locations[0].offset = 77; + let previous = store.get(pointer_key.clone()).await.unwrap().unwrap(); + assert!(store + .compare_exchange(pointer_key, previous.value, late.encode().unwrap()) + .await + .unwrap()); + assert!(matches!( + repository.publish_completion(&frozen).await, + Err(MultipartRepositoryError::InvalidPart) + )); + let object_key = MetadataKey::object( + &TenantId::new(b"tenant".to_vec()).unwrap(), + session.bucket_id, + &session.object_key, + ) + .unwrap(); + assert!(store.get(object_key).await.unwrap().is_none()); +} + +fn part_for(session: &MultipartSessionRecord) -> MultipartPartRecord { + let mut value = part(); + value.upload_id = session.upload_id; + value +} + +#[tokio::test] +async fn abort_confirms_lost_reply_and_rejects_part_publication() { + let (repository, lose_reply, _) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + lose_reply.store(true, Ordering::SeqCst); + let aborted = repository.abort(&session).await.unwrap(); + assert_eq!(aborted.phase, MultipartPhase::Aborted); + assert_eq!(repository.abort(&session).await.unwrap(), aborted); + assert!(matches!( + repository.put_stream_part(&session, &part(), 120).await, + Err(MultipartRepositoryError::Conflict) + )); + assert_eq!( + repository + .part_generation(&session, 1, 1) + .await + .unwrap() + .unwrap() + .length, + 5 + ); +} + +#[tokio::test] +async fn part_listing_paginates_current_generations_in_number_order() { + let (repository, _, _) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + for number in [3, 1, 2] { + let mut value = part(); + value.number = number; + repository.put_stream_part(&session, &value, 110).await.unwrap(); + } + let mut replacement = part(); + replacement.number = 2; + replacement.locations[0].offset += 39; + replacement.raw_md5 = [8; 16]; + repository + .put_stream_part(&session, &replacement, 111) + .await + .unwrap(); + let first = repository.list_parts(&session, 0, 2).await.unwrap(); + assert_eq!( + first.parts.iter().map(|part| part.number).collect::>(), + [1, 2] + ); + assert_eq!(first.parts[1].revision, 2); + assert_eq!(first.next_part_number_marker, Some(2)); + let second = repository.list_parts(&session, 2, 2).await.unwrap(); + assert_eq!( + second.parts.iter().map(|part| part.number).collect::>(), + [3] + ); + assert_eq!(second.next_part_number_marker, None); +} + +#[tokio::test] +async fn upload_prefix_retains_session_and_replaced_part_generations() { + let (repository, _, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + assert_eq!( + repository + .load_identity(session.bucket_id, &session.object_key, &session.upload_id) + .await + .unwrap(), + Some(session.clone()) + ); + assert!(repository + .load_identity(session.bucket_id, b"another", &session.upload_id) + .await + .unwrap() + .is_none()); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + let mut replacement = part(); + replacement.locations[0].offset += 39; + replacement.raw_md5 = [8; 16]; + repository + .put_stream_part(&session, &replacement, 111) + .await + .unwrap(); + repository.abort(&session).await.unwrap(); + + let tenant = TenantId::new(b"tenant".to_vec()).unwrap(); + let prefix = MetadataKey::multipart_upload_prefix(&tenant, session.bucket_id, &session.upload_id); + let mut end = prefix.clone(); + end.push(u8::MAX); + let records = store.scan(prefix, end, 10, 1024 * 1024).await.unwrap(); + assert_eq!(records.len(), 4); + assert_eq!( + records[0].key, + MetadataKey::multipart_session(&tenant, session.bucket_id, &session.upload_id) + ); + assert_eq!( + records[1].key, + MetadataKey::multipart_part(&tenant, session.bucket_id, &session.upload_id, 1).unwrap() + ); + for revision in [1, 2] { + assert!(records.iter().any(|record| record.key + == MetadataKey::multipart_part_generation( + &tenant, + session.bucket_id, + &session.upload_id, + 1, + revision, + ) + .unwrap())); + } +} + +#[tokio::test] +async fn upload_listing_filters_terminal_records_and_resumes_same_key() { + let (repository, _, _) = repository().await; + let mut first = session(); + first.object_key = b"pre/a".to_vec(); + first.upload_id = [1; 16]; + let mut second = first.clone(); + second.upload_id = [2; 16]; + second.created_ms = 101; + let mut third = first.clone(); + third.object_key = b"pre/b".to_vec(); + third.upload_id = [3; 16]; + let mut terminal = first.clone(); + terminal.object_key = b"pre/aborted".to_vec(); + terminal.upload_id = [4; 16]; + for upload in [&first, &second, &third, &terminal] { + repository.begin(upload).await.unwrap(); + } + repository.abort(&terminal).await.unwrap(); + + let page = repository + .list_uploads(first.bucket_id, b"pre/", None, None, 1, 110) + .await + .unwrap(); + assert_eq!(page.uploads, [first.clone()]); + assert_eq!(page.next, Some((first.object_key.clone(), first.upload_id))); + let (key, id) = page.next.unwrap(); + let page = repository + .list_uploads(first.bucket_id, b"pre/", Some(&key), Some(&id), 1, 110) + .await + .unwrap(); + assert_eq!(page.uploads, [second.clone()]); + assert_eq!(page.next, Some((second.object_key.clone(), second.upload_id))); + let page = repository + .list_uploads(first.bucket_id, b"pre/", Some(b"pre/a"), None, 10, 110) + .await + .unwrap(); + assert_eq!(page.uploads, [third]); + assert!(page.next.is_none()); + assert!(repository + .list_uploads(first.bucket_id, b"pre/", None, Some(&id), 1, 110) + .await + .is_err()); +} + +#[tokio::test] +async fn expiry_pages_mark_old_uploads_terminal_without_removing_part_evidence() { + let (repository, _, store) = repository().await; + let first = session(); + let mut second = first.clone(); + second.upload_id = [8; 16]; + second.object_key = b"later".to_vec(); + second.expires_ms = 300; + repository.begin(&first).await.unwrap(); + repository.begin(&second).await.unwrap(); + repository.put_stream_part(&first, &part(), 110).await.unwrap(); + + let page = repository + .expire_page(first.bucket_id, None, 250, 1) + .await + .unwrap(); + assert_eq!(page.expired, 0); + let next = page.next.expect("another index page remains"); + let page = repository + .expire_page(first.bucket_id, Some(&next), 250, 1) + .await + .unwrap(); + assert_eq!(page.expired, 1); + assert!(page.next.is_none()); + assert_eq!( + repository.load(&first).await.unwrap().unwrap().phase, + MultipartPhase::Aborted + ); + assert_eq!( + repository.load(&second).await.unwrap().unwrap().phase, + MultipartPhase::Open + ); + assert!(repository.part_generation(&first, 1, 1).await.unwrap().is_some()); + + let tenant = TenantId::new(b"tenant".to_vec()).unwrap(); + let index = + MetadataKey::multipart_upload_index(&tenant, first.bucket_id, &first.object_key, &first.upload_id) + .unwrap(); + assert!(store.get(index).await.unwrap().is_some()); +} diff --git a/lib/crowdb-access-s3/tests/route_test.rs b/lib/crowdb-access-s3/tests/route_test.rs index 68a88a453..ab51f2b97 100644 --- a/lib/crowdb-access-s3/tests/route_test.rs +++ b/lib/crowdb-access-s3/tests/route_test.rs @@ -1,7 +1,9 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crowdb_access_s3::route::{classify, classify_request, RouteError, S3Operation}; +use crowdb_access_s3::route::{ + classify, classify_multipart, classify_request, MultipartOperation, RouteError, S3Operation, +}; use hyper::header::HeaderValue; use hyper::{HeaderMap, Method, Uri}; @@ -32,6 +34,10 @@ fn rejects_extensions_before_dispatch() { classify(&Method::POST, &"/bucket/key?uploads".parse().unwrap()), Err(RouteError::NotImplemented) ); + assert_eq!( + classify(&Method::PUT, &"/bucket/key?partNumber=7".parse().unwrap()), + Err(RouteError::NotImplemented) + ); } #[test] @@ -43,3 +49,73 @@ fn rejects_headers_that_select_excluded_behavior() { Err(RouteError::NotImplemented) ); } + +#[test] +fn multipart_queries_have_unambiguous_paths_and_identities() { + let id = "ab".repeat(16); + let cases = [ + ( + Method::POST, + "/bucket/key?uploads".to_string(), + MultipartOperation::Create, + ), + ( + Method::GET, + "/bucket?uploads".to_string(), + MultipartOperation::ListUploads, + ), + ( + Method::PUT, + format!("/bucket/key?partNumber=7&uploadId={id}"), + MultipartOperation::UploadPart, + ), + ( + Method::GET, + format!("/bucket/key?uploadId={id}"), + MultipartOperation::ListParts, + ), + ( + Method::POST, + format!("/bucket/key?uploadId={id}"), + MultipartOperation::Complete, + ), + ( + Method::DELETE, + format!("/bucket/key?uploadId={id}"), + MultipartOperation::Abort, + ), + ]; + for (method, uri, expected) in cases { + let uri = uri.parse().unwrap(); + let route = classify_multipart(&method, &uri).unwrap().unwrap(); + assert_eq!(route.operation, expected); + assert_eq!(route.bucket, b"bucket"); + let authenticated = classify_request(&method, &uri, &HeaderMap::new()).unwrap(); + assert_eq!(authenticated.bucket.as_deref(), Some(b"bucket".as_slice())); + assert_eq!(authenticated.upload_id, route.upload_id); + assert_eq!(authenticated.part_number, route.part_number); + } + assert_eq!( + classify_multipart(&Method::GET, &"/bucket/key".parse().unwrap()), + Ok(None) + ); + assert_eq!( + classify_multipart( + &Method::PUT, + &format!("/bucket/key?uploadId={id}").parse().unwrap() + ), + Err(RouteError::Invalid) + ); + assert_eq!( + classify_multipart(&Method::POST, &"/bucket/key?uploadId=bad".parse().unwrap()), + Err(RouteError::Invalid) + ); + assert_eq!( + classify_multipart(&Method::GET, &"/bucket?uploads&uploads".parse().unwrap()), + Err(RouteError::Invalid) + ); + assert_eq!( + classify_multipart(&Method::PUT, &"/bucket/key?partNumber=7".parse().unwrap()), + Err(RouteError::Invalid) + ); +} diff --git a/lib/crowdb-access-s3/tests/wire_test.rs b/lib/crowdb-access-s3/tests/wire_test.rs index eef424bdc..2c189547e 100644 --- a/lib/crowdb-access-s3/tests/wire_test.rs +++ b/lib/crowdb-access-s3/tests/wire_test.rs @@ -1,7 +1,10 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crowdb_access_s3::metadata::{BucketId, BucketNameRecord, ObjectRecord, TenantId}; +use crowdb_access_s3::metadata::{ + BucketId, BucketNameRecord, MultipartPartPage, MultipartPartRecord, MultipartPhase, + MultipartSessionRecord, MultipartUploadPage, ObjectRecord, TenantId, +}; use crowdb_access_s3::object::ListObjectsV2Page; use crowdb_access_s3::wire; @@ -46,3 +49,75 @@ fn list_objects_serializes_stable_etag_time_and_continuation() { assert!(xml.contains("true")); assert!(xml.contains("opaque")); } + +#[test] +fn multipart_xml_escapes_names_and_reports_selected_part_metadata() { + let id = [0xab; 16]; + let created = wire::create_multipart_upload(b"b&", b"k<", &id); + assert!(created.contains("b&")); + assert!(created.contains("k<")); + assert!(created.contains(&format!("{}", "ab".repeat(16)))); + + let page = MultipartPartPage { + parts: vec![MultipartPartRecord { + bucket_id: BucketId::new([1; 16]), + upload_id: id, + number: 4, + revision: 1, + modified_ms: 1_000, + length: 8, + raw_md5: [9; 16], + locations: Vec::new(), + }], + next_part_number_marker: Some(4), + }; + let listed = wire::list_multipart_parts(b"b&", b"k<", &id, 2, 1, &page); + assert!(listed.contains("2")); + assert!(listed.contains("4")); + assert!(listed.contains(""09090909090909090909090909090909"")); + assert!(listed.contains("8")); + let completed = wire::complete_multipart_upload("http://host/b&/k<", b"b&", b"k<", "abc-1"); + assert!(completed.contains("http://host/b&/k<")); + assert!(completed.contains(""abc-1"")); +} + +#[test] +fn multipart_upload_listing_emits_stable_markers_and_initiation_time() { + let id = [0xab; 16]; + let session = MultipartSessionRecord { + bucket_id: BucketId::new([1; 16]), + object_key: b"prefix/k&".to_vec(), + upload_id: id, + revision: 1, + phase: MultipartPhase::Open, + created_ms: 1_000, + expires_ms: 2_000, + content_type: "application/octet-stream".into(), + max_parts: 1, + max_part_bytes: 5, + max_object_bytes: 5, + max_staged_bytes: 5, + part_count: 0, + staged_bytes: 0, + pending: None, + selection: None, + completion_request_digest: None, + publication_ms: None, + object_predecessor: None, + etag: None, + }; + let page = MultipartUploadPage { + uploads: vec![session], + next: Some((b"prefix/k&".to_vec(), id)), + }; + let xml = wire::list_multipart_uploads(b"bucket", b"prefix/", None, None, 1, "owner&", &page); + assert!(xml.contains("prefix/k&")); + assert!(xml.contains("prefix/k&")); + assert!(xml.contains(&format!( + "{}", + "ab".repeat(16) + ))); + assert!(xml.contains("1970-01-01T00:00:01Z")); + assert!(xml.contains("owner&")); + assert!(xml.contains("true")); +} diff --git a/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs b/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs index 458452bfc..7a9ef5532 100644 --- a/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs +++ b/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs @@ -216,7 +216,12 @@ async fn read_data_image(stack: &E2eStack, chunk: &Chunk, location: &Location) - let end = start + u64::from(strip.capacity) * KIB as u64; start <= location.offset && location.offset < end }) - .expect("location strip"); + .unwrap_or_else(|| { + panic!( + "location strip missing while reading image: chunk_id={:?}, offset={}, length={}, chunk={chunk:#?}", + location.chunk_id, location.offset, location.length + ) + }); let strip_start = u64::from(strip.chunk_offset) * KIB as u64; let unit_bytes = u64::from(strip.unit_kb) * KIB as u64; let segment = match strip.strip.as_ref().expect("strip body") { diff --git a/lib/crowdb-chunk-stream/tests/production_restart_e2e.rs b/lib/crowdb-chunk-stream/tests/production_restart_e2e.rs index 2773e23b7..533c7ba00 100644 --- a/lib/crowdb-chunk-stream/tests/production_restart_e2e.rs +++ b/lib/crowdb-chunk-stream/tests/production_restart_e2e.rs @@ -106,6 +106,7 @@ async fn seed_restart_hardware(hardware: &HardwareClient) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![node_id], + ..Default::default() }, ) .await @@ -120,6 +121,7 @@ async fn seed_restart_hardware(hardware: &HardwareClient) { disk_group_ids: vec![disk_group_id], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await @@ -132,6 +134,7 @@ async fn seed_restart_hardware(hardware: &HardwareClient) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: vec![disk_id], + name: String::new(), }, ) .await diff --git a/lib/crowdb-console-shared/src/bootstrap_intent.rs b/lib/crowdb-console-shared/src/bootstrap_intent.rs new file mode 100644 index 000000000..7405468fc --- /dev/null +++ b/lib/crowdb-console-shared/src/bootstrap_intent.rs @@ -0,0 +1,292 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Immutable pre-Group-0 topology intent for interrupted bootstrap recovery. + +use std::collections::HashSet; +use std::fs::{self, OpenOptions}; +use std::io::Write; +use std::os::unix::fs::OpenOptionsExt; +use std::path::Path; +use std::sync::atomic::{AtomicU64, Ordering}; + +use serde::{Deserialize, Serialize}; + +use crate::config::{ConsoleConfig, NodeEntry, RackEntry, ServerEntry, ServiceType}; +use crate::error::{Error, Result}; + +const VERSION: u32 = 1; +static NEXT_TEMP: AtomicU64 = AtomicU64::new(0); + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct BootstrapIntent { + version: u32, + members: Vec, + racks: Vec, + nodes: Vec, + servers: Vec, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct IntentRack { + id: u64, + name: String, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct IntentNode { + id: u64, + rack_id: u64, + host: String, + ssh_port: u16, + ssh_user: String, + ssh_credential_ref: Option, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct IntentServer { + id: String, + node_id: u64, + management_url: String, + rpc_url: Option, +} + +impl BootstrapIntent { + /// Capture immutable hardware and KV endpoint identities before bootstrap. + /// + /// # Errors + /// Rejects missing members, missing management endpoints and inline SSH secrets. + pub fn capture(config: &ConsoleConfig, members: &[u64]) -> Result { + if members.is_empty() + || members + .iter() + .enumerate() + .any(|(index, node)| members[..index].contains(node)) + { + return Err(Error::Config( + "bootstrap members must be nonempty and distinct".into(), + )); + } + let nodes: Vec<_> = config + .nodes + .iter() + .map(|node| { + if node.ssh_key.is_some() || node.ssh_password.is_some() { + return Err(Error::Config( + "bootstrap intent cannot store inline SSH secrets".into(), + )); + } + Ok(IntentNode { + id: node.id, + rack_id: node.rack_id, + host: node.host.clone(), + ssh_port: node.ssh_port, + ssh_user: node.ssh_user.clone(), + ssh_credential_ref: node.ssh_credential_ref.clone(), + }) + }) + .collect::>()?; + let servers: Vec<_> = config + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| { + let node_id = server + .node_id + .ok_or_else(|| Error::Config("bootstrap KV server has no node id".into()))?; + Ok(IntentServer { + id: server.id.clone(), + node_id, + management_url: server.url.clone(), + rpc_url: server.rpc_url.clone(), + }) + }) + .collect::>()?; + let mut rack_ids = HashSet::new(); + if config.racks.iter().any(|rack| !rack_ids.insert(rack.id)) { + return Err(Error::Config("duplicate bootstrap rack id".into())); + } + let mut node_ids = HashSet::new(); + if nodes + .iter() + .any(|node| !node_ids.insert(node.id) || !rack_ids.contains(&node.rack_id)) + { + return Err(Error::Config("duplicate or orphan bootstrap node".into())); + } + let mut endpoint_nodes = HashSet::new(); + if servers.iter().any(|server| { + !endpoint_nodes.insert(server.node_id) + || !node_ids.contains(&server.node_id) + || server.management_url.is_empty() + }) { + return Err(Error::Config("duplicate or invalid bootstrap KV endpoint".into())); + } + for member in members { + if !nodes.iter().any(|node| node.id == *member) + || !servers.iter().any(|server| server.node_id == *member) + { + return Err(Error::Config(format!( + "bootstrap member {member} lacks node or KV endpoint" + ))); + } + } + Ok(Self { + version: VERSION, + members: members.to_vec(), + racks: config + .racks + .iter() + .map(|rack| IntentRack { + id: rack.id, + name: rack.name.clone(), + }) + .collect(), + nodes, + servers, + }) + } + + #[must_use] + pub fn members(&self) -> &[u64] { + &self.members + } + + /// Create the intent once, accepting an identical interrupted attempt. + /// + /// # Errors + /// Rejects changed intent, symlinks, invalid content or I/O errors. + pub fn seal(&self, path: &Path) -> Result<()> { + if path.exists() || fs::symlink_metadata(path).is_ok() { + return if Self::load(path)? == *self { + Ok(()) + } else { + Err(Error::Conflict { + kind: "bootstrap intent".into(), + id: path.display().to_string(), + }) + }; + } + let parent = path + .parent() + .ok_or_else(|| Error::Config("bootstrap intent has no parent".into()))?; + fs::create_dir_all(parent)?; + let data = toml::to_string(self).map_err(|error| Error::Config(error.to_string()))?; + let temporary = path.with_extension(format!( + "bootstrap-tmp-{}-{}", + std::process::id(), + NEXT_TEMP.fetch_add(1, Ordering::Relaxed) + )); + let result = (|| { + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&temporary)?; + file.write_all(data.as_bytes())?; + file.sync_all()?; + match fs::hard_link(&temporary, path) { + Ok(()) => { + fs::File::open(parent)?.sync_all()?; + Ok(()) + } + Err(error) if error.kind() == std::io::ErrorKind::AlreadyExists => { + if Self::load(path)? == *self { + Ok(()) + } else { + Err(Error::Conflict { + kind: "bootstrap intent".into(), + id: path.display().to_string(), + }) + } + } + Err(error) => Err(Error::Io(error)), + } + })(); + let _ = fs::remove_file(&temporary); + result + } + + /// Read a previously sealed intent without accepting a legacy console file. + /// + /// # Errors + /// Rejects symlinks, unknown fields, invalid version or I/O errors. + pub fn load(path: &Path) -> Result { + if !fs::symlink_metadata(path)?.file_type().is_file() { + return Err(Error::Config("bootstrap intent is not a regular file".into())); + } + let intent: Self = + toml::from_str(&fs::read_to_string(path)?).map_err(|error| Error::Config(error.to_string()))?; + if intent.version != VERSION { + return Err(Error::Config("unsupported bootstrap intent version".into())); + } + if Self::capture(&intent.to_config(), &intent.members)? != intent { + return Err(Error::Config("bootstrap intent contents are inconsistent".into())); + } + Ok(intent) + } + + /// Restore only the bootstrap inputs into a fresh in-memory console context. + #[must_use] + pub fn to_config(&self) -> ConsoleConfig { + ConsoleConfig { + racks: self + .racks + .iter() + .map(|rack| RackEntry { + id: rack.id, + name: rack.name.clone(), + }) + .collect(), + nodes: self + .nodes + .iter() + .map(|node| NodeEntry { + id: node.id, + rack_id: node.rack_id, + host: node.host.clone(), + ssh_port: node.ssh_port, + ssh_user: node.ssh_user.clone(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: node.ssh_credential_ref.clone(), + }) + .collect(), + servers: self + .servers + .iter() + .map(|server| ServerEntry { + id: server.id.clone(), + url: server.management_url.clone(), + node_id: Some(server.node_id), + rpc_url: server.rpc_url.clone(), + rest_port: None, + rpc_port: None, + auto_start: false, + binary: None, + election_profile: None, + pid: None, + service_type: ServiceType::Kv, + rpc_workers: None, + no_fsync: false, + }) + .collect(), + ..Default::default() + } + } + + /// Remove the intent only after Group 0 contents were verified. + /// + /// # Errors + /// Returns an I/O error if removal fails. + pub(crate) fn clear_verified(path: &Path) -> Result<()> { + fs::remove_file(path)?; + if let Some(parent) = path.parent() { + fs::File::open(parent)?.sync_all()?; + } + Ok(()) + } +} diff --git a/lib/crowdb-console-shared/src/cluster_deployer.rs b/lib/crowdb-console-shared/src/cluster_deployer.rs index da93aac7e..eb53344b5 100644 --- a/lib/crowdb-console-shared/src/cluster_deployer.rs +++ b/lib/crowdb-console-shared/src/cluster_deployer.rs @@ -336,6 +336,7 @@ impl CrowdbClusterDeployer { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }, ) .await?; diff --git a/lib/crowdb-console-shared/src/config.rs b/lib/crowdb-console-shared/src/config.rs index 2d0237aa9..b2b500376 100644 --- a/lib/crowdb-console-shared/src/config.rs +++ b/lib/crowdb-console-shared/src/config.rs @@ -1,167 +1,24 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Console configuration: persisted registry of `crowdb-kv-server` instances. -//! -//! C2 status: file-backed `[[server]]` list; later phases extend with -//! racks, nodes, ssh creds. The struct is the single source of truth so -//! the storage format can evolve without touching call sites. +//! In-memory console operation inputs and topology snapshots. Durable cluster +//! records live in Group 0; process launch policy uses `config::web`. -use std::path::{Path, PathBuf}; -use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::atomic::AtomicU64; use crowdb_protocol::{NodeId, RackId}; use serde::{Deserialize, Serialize}; use std::collections::BTreeMap; -static TMP_FILE_COUNTER: AtomicU64 = AtomicU64::new(0); - use crate::cluster::DiskGroupId; use crate::error::{Error, Result}; -use std::fmt; - pub mod web; -/// Serde helper: serialize a `BTreeMap` with string keys (TOML -/// requires string keys) and deserialize back to `u64` keys. -mod int_key { - use serde::de::{Deserialize, Deserializer, MapAccess, Visitor}; - use serde::ser::{Serialize, Serializer}; - use std::collections::BTreeMap; - use std::fmt; - use std::marker::PhantomData; - use std::str::FromStr; - - pub fn serialize(map: &BTreeMap, serializer: S) -> Result - where - K: ToString + Ord, - V: Serialize, - S: Serializer, - { - let string_map: BTreeMap = map.iter().map(|(k, v)| (k.to_string(), v)).collect(); - string_map.serialize(serializer) - } - - pub fn deserialize<'de, K, V, D>(deserializer: D) -> Result, D::Error> - where - K: FromStr + Ord, - K::Err: fmt::Display, - V: Deserialize<'de>, - D: Deserializer<'de>, - { - struct IntKeyVisitor(PhantomData<(K, V)>); - - impl<'de, K, V> Visitor<'de> for IntKeyVisitor - where - K: FromStr + Ord, - K::Err: fmt::Display, - V: Deserialize<'de>, - { - type Value = BTreeMap; - - fn expecting(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.write_str("a map with string-encoded integer keys") - } - - fn visit_map(self, mut access: A) -> Result - where - A: MapAccess<'de>, - { - let mut map = BTreeMap::new(); - while let Some((key, value)) = access.next_entry::()? { - let k = K::from_str(&key).map_err(serde::de::Error::custom)?; - map.insert(k, value); - } - Ok(map) - } - } - - deserializer.deserialize_map(IntKeyVisitor::(PhantomData)) - } -} - -pub trait ConsoleConfigEngine: Send + Sync { - /// Load the console configuration from the engine's storage. - /// - /// # Errors - /// Returns an error if loading fails (e.g., file not found, parse error). - fn load(&self) -> Result; - - /// Save the console configuration to the engine's storage. - /// - /// # Errors - /// Returns an error if saving fails (e.g., permission denied, write error). - fn save(&self, config: &ConsoleConfig) -> Result<()>; -} - -#[derive(Clone, PartialEq, Eq)] -pub struct TomlFileEngine { - path: PathBuf, -} - -impl fmt::Debug for TomlFileEngine { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.debug_struct("TomlFileEngine") - .field("path", &self.path) - .finish() - } -} - -impl TomlFileEngine { - #[must_use] - pub fn new(path: impl Into) -> Self { - Self { path: path.into() } - } - - #[must_use] - #[allow(dead_code)] - pub(crate) fn path(&self) -> &Path { - &self.path - } - - #[must_use] - pub fn default_path() -> Option { - Some( - crowdb_protocol::port::namespace::runtime_root() - .join("persistent") - .join("console") - .join("crowdb-kv.db.toml"), - ) - } - - #[must_use] - #[allow(dead_code)] - pub(crate) fn from_default_path() -> Option { - Self::default_path().map(Self::new) - } -} - -impl ConsoleConfigEngine for TomlFileEngine { - fn load(&self) -> Result { - match std::fs::read_to_string(&self.path) { - Ok(body) => ConsoleConfig::from_toml_str(&body, &self.path), - Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(ConsoleConfig::default()), - Err(e) => Err(Error::Io(e)), - } - } - - fn save(&self, config: &ConsoleConfig) -> Result<()> { - if let Some(parent) = self.path.parent() { - std::fs::create_dir_all(parent).map_err(Error::Io)?; - } - let body = config.to_toml_string()?; - let seq = TMP_FILE_COUNTER.fetch_add(1, Ordering::Relaxed); - let tmp = self.path.with_extension(format!("toml.tmp.{seq}")); - std::fs::write(&tmp, body).map_err(Error::Io)?; - std::fs::rename(&tmp, &self.path).map_err(Error::Io)?; - Ok(()) - } -} +static TMP_FILE_COUNTER: AtomicU64 = AtomicU64::new(0); -/// On-disk console config. New top-level fields land in later phases -/// (ssh defaults, etc.). Unknown fields are ignored on load and dropped -/// on save (`serde(default)` everywhere) to keep migrations easy. +/// Ephemeral inputs for cluster operations and bootstrap. This struct is not +/// a durable topology store. #[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] pub struct ConsoleConfig { #[serde(default, rename = "rack")] @@ -181,9 +38,6 @@ pub struct ConsoleConfig { /// Reproducible commands for locally deployed benchmark services. #[serde(default)] pub local_launches: BTreeMap, - /// Optional `[bench]` section. Reserved for future use. - #[serde(default, skip_serializing_if = "BenchConfig::is_empty")] - pub(crate) bench: BenchConfig, } /// Retained local process state used to restart a benchmark service without @@ -200,19 +54,6 @@ pub struct LocalLaunchSpec { pub readiness_url: Option, } -/// `[bench]` section. Reserved for future knobs (default reporting -/// dir, max threads, etc.); currently empty. -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -pub(crate) struct BenchConfig {} - -impl BenchConfig { - #[must_use] - #[allow(clippy::unused_self)] - pub(crate) fn is_empty(&self) -> bool { - true - } -} - #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct RackEntry { pub id: RackId, @@ -233,6 +74,9 @@ pub struct NodeEntry { /// back to local-fork lifecycle (C3 path) for tests. #[serde(default)] pub ssh_user: String, + /// Local secret-store lookup key shared across consoles, never secret material. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub ssh_credential_ref: Option, /// Optional explicit private-key path. `None` falls back to /// `~/.ssh/id_ed25519` then `~/.ssh/id_rsa`. #[serde(default, skip_serializing_if = "Option::is_none")] @@ -286,8 +130,7 @@ pub struct DiskEntry { pub device_path: String, } -/// Discriminator for console-deployed server entries. `Kv` is the -/// default for backward compatibility with existing persisted configs. +/// Discriminator for ephemeral console deployment inputs. #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)] pub enum ServiceType { #[default] @@ -298,14 +141,13 @@ pub enum ServiceType { ChunkKv, AccessServer, /// Standalone crowdb-rpc-fb-server (C++ echo server for RPC bench). - /// Not a full KV server — no management port, no sysdata. Tracked - /// in config only for PID/port lifecycle via `cluster destroy`. + /// Not a full KV server — no management port or sysdata. Rpc, } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct ServerEntry { - /// Console-side identifier; must be unique within the file. + /// Console-side identifier; must be unique within an operation context. pub id: String, /// Service URL. For KV this is the `crowdb-kv-server` management base /// URL; for `DiskDB` this is its public crowdb-rpc endpoint. @@ -330,13 +172,11 @@ pub struct ServerEntry { pub election_profile: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub pid: Option, - /// Service type discriminator (R77). Defaults to `Kv` for - /// backward compatibility with pre-R77 persisted configs. + /// Service type discriminator. KV is the default for local fixtures. #[serde(default, skip_serializing_if = "is_default_service_type")] pub service_type: ServiceType, /// `--rpc-workers` value passed to the spawned `crowdb-kv-server`. - /// `None` means the server's default (2) is used. Persisted so - /// restart reuses the same value. + /// `None` means the server's default (2) is used. #[serde(default, skip_serializing_if = "Option::is_none")] pub rpc_workers: Option, /// `--no-fsync` flag passed to the spawned `crowdb-kv-server`. @@ -371,112 +211,6 @@ pub struct ReplicaEntry { pub node_id: NodeId, } -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedConsoleConfig { - #[serde(default, with = "int_key", skip_serializing_if = "BTreeMap::is_empty")] - rack: BTreeMap, - #[serde(default, with = "int_key", skip_serializing_if = "BTreeMap::is_empty")] - node: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - crowdb_kv_server: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - store: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - group: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - disk_group: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - disk: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - local_launch: BTreeMap, - #[serde(default, skip_serializing_if = "BenchConfig::is_empty")] - bench: BenchConfig, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedDiskGroupEntry { - id: DiskGroupId, - rack_id: RackId, - node_id: NodeId, - #[serde(default)] - name: String, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedDiskEntry { - disk_id: String, - disk_group_id: DiskGroupId, - rack_id: RackId, - node_id: NodeId, - disk_type: String, - capacity_bytes: u64, - zone_size_bytes: u64, - unit_size_bytes: u32, - #[serde(default, skip_serializing_if = "String::is_empty")] - device_path: String, -} - -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedRackEntry { - #[serde(default)] - name: String, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedNodeEntry { - rack_id: RackId, - host: String, - #[serde(default = "default_ssh_port")] - ssh_port: u16, - #[serde(default)] - ssh_user: String, - #[serde(default, skip_serializing_if = "Option::is_none")] - ssh_key: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - ssh_password: Option, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedServerEntry { - node_id: Option, - url: String, - #[serde(default, skip_serializing_if = "Option::is_none")] - rpc_url: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - rest_port: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - rpc_port: Option, - #[serde(default)] - auto_start: bool, - #[serde(default, skip_serializing_if = "Option::is_none")] - binary: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - election_profile: Option, - #[serde(default, skip_serializing_if = "is_default_service_type")] - service_type: ServiceType, - #[serde(default, skip_serializing_if = "Option::is_none")] - rpc_workers: Option, - #[serde(default, skip_serializing_if = "std::ops::Not::not")] - no_fsync: bool, - #[serde(default, skip_serializing_if = "Option::is_none")] - pid: Option, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedStoreEntry { - store_id: u64, - #[serde(default)] - nodes: Vec, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedGroupEntry { - store_id: u64, - group_id: u64, - #[serde(default)] - replicas: Vec, -} - impl ServerEntry { /// Convenience constructor for a plain registered server (C2 style). #[must_use] @@ -500,49 +234,6 @@ impl ServerEntry { } impl ConsoleConfig { - /// Default config file path. - /// - /// Config is persisted below the workspace persistent runtime namespace. - /// This file stores registered crowdb-kv-server instances for the console. - #[must_use] - pub(crate) fn default_path() -> Option { - TomlFileEngine::default_path() - } - - /// Load the config from `path`. A missing file yields a default - /// (empty) config so first-run is friendly. - /// - /// # Errors - /// Returns `Error::Io` for non-`NotFound` filesystem errors and - /// `Error::Config` for TOML parse failures. - pub fn load(path: &Path) -> Result { - TomlFileEngine::new(path).load() - } - - /// Save the config atomically (write to a tempfile, then rename). - /// - /// # Errors - /// Filesystem and TOML serialization errors are propagated. - pub(crate) fn save(&self, path: &Path) -> Result<()> { - TomlFileEngine::new(path).save(self) - } - - /// Load configuration using the provided engine. - /// - /// # Errors - /// Returns an error if the engine's load fails. - pub fn load_with_engine(engine: &dyn ConsoleConfigEngine) -> Result { - engine.load() - } - - /// Save configuration using the provided engine. - /// - /// # Errors - /// Returns an error if the engine's save fails. - pub fn save_with_engine(&self, engine: &dyn ConsoleConfigEngine) -> Result<()> { - engine.save(self) - } - /// Add a server entry. Rejects duplicate `id` and duplicate `url`. /// /// # Errors @@ -945,354 +636,11 @@ impl ConsoleConfig { pub fn disks_in_group(&self, dg_id: DiskGroupId) -> Vec<&DiskEntry> { self.disks.iter().filter(|d| d.disk_group_id == dg_id).collect() } - - #[allow(clippy::too_many_lines)] - fn to_persisted(&self) -> PersistedConsoleConfig { - let rack = self - .racks - .iter() - .map(|entry| { - ( - entry.id, - PersistedRackEntry { - name: entry.name.clone(), - }, - ) - }) - .collect(); - let node = self - .nodes - .iter() - .map(|entry| { - ( - entry.id, - PersistedNodeEntry { - rack_id: entry.rack_id, - host: entry.host.clone(), - ssh_port: entry.ssh_port, - ssh_user: entry.ssh_user.clone(), - ssh_key: entry.ssh_key.clone(), - ssh_password: entry.ssh_password.clone(), - }, - ) - }) - .collect(); - let crowdb_kv_server = self - .servers - .iter() - .map(|entry| { - ( - entry.id.clone(), - PersistedServerEntry { - node_id: entry.node_id, - url: entry.url.clone(), - rpc_url: entry.rpc_url.clone(), - rest_port: entry.rest_port, - rpc_port: entry.rpc_port, - auto_start: entry.auto_start, - binary: entry.binary.clone(), - election_profile: entry.election_profile.clone(), - service_type: entry.service_type, - rpc_workers: entry.rpc_workers, - no_fsync: entry.no_fsync, - pid: entry.pid, - }, - ) - }) - .collect(); - let store = self - .stores - .iter() - .map(|entry| { - ( - entry.store_id.to_string(), - PersistedStoreEntry { - store_id: entry.store_id, - nodes: entry.nodes.clone(), - }, - ) - }) - .collect(); - let group = self - .groups - .iter() - .map(|entry| { - ( - format!("{}:{}", entry.store_id, entry.group_id), - PersistedGroupEntry { - store_id: entry.store_id, - group_id: entry.group_id, - replicas: entry.replicas.clone(), - }, - ) - }) - .collect(); - let disk_group = self - .disk_groups - .iter() - .map(|entry| { - ( - entry.id.to_string(), - PersistedDiskGroupEntry { - id: entry.id, - rack_id: entry.rack_id, - node_id: entry.node_id, - name: entry.name.clone(), - }, - ) - }) - .collect(); - let disk = self - .disks - .iter() - .map(|entry| { - ( - entry.disk_id.clone(), - PersistedDiskEntry { - disk_id: entry.disk_id.clone(), - disk_group_id: entry.disk_group_id, - rack_id: entry.rack_id, - node_id: entry.node_id, - disk_type: entry.disk_type.clone(), - capacity_bytes: entry.capacity_bytes, - zone_size_bytes: entry.zone_size_bytes, - unit_size_bytes: entry.unit_size_bytes, - device_path: entry.device_path.clone(), - }, - ) - }) - .collect(); - PersistedConsoleConfig { - rack, - node, - crowdb_kv_server, - store, - group, - disk_group, - disk, - local_launch: self.local_launches.clone(), - bench: self.bench.clone(), - } - } - - fn from_persisted(persisted: PersistedConsoleConfig) -> Self { - let mut racks: Vec = persisted - .rack - .into_iter() - .map(|(id, entry)| RackEntry { id, name: entry.name }) - .collect(); - racks.sort_by_key(|r| r.id); - let mut nodes: Vec = persisted - .node - .into_iter() - .map(|(id, entry)| NodeEntry { - id, - rack_id: entry.rack_id, - host: entry.host, - ssh_port: entry.ssh_port, - ssh_user: entry.ssh_user, - ssh_key: entry.ssh_key, - ssh_password: entry.ssh_password, - }) - .collect(); - nodes.sort_by_key(|n| n.id); - let mut servers: Vec = persisted - .crowdb_kv_server - .into_iter() - .map(|(id, entry)| ServerEntry { - id, - url: entry.url, - node_id: entry.node_id, - rpc_url: entry.rpc_url, - rest_port: entry.rest_port, - rpc_port: entry.rpc_port, - auto_start: entry.auto_start, - binary: entry.binary, - election_profile: entry.election_profile, - pid: entry.pid, - service_type: entry.service_type, - rpc_workers: entry.rpc_workers, - no_fsync: entry.no_fsync, - }) - .collect(); - servers.sort_by(|a, b| a.id.cmp(&b.id)); - let mut stores: Vec = persisted - .store - .into_values() - .map(|entry| StoreEntry { - store_id: entry.store_id, - nodes: entry.nodes, - }) - .collect(); - stores.sort_by_key(|s| s.store_id); - let mut groups: Vec = persisted - .group - .into_values() - .map(|entry| GroupEntry { - store_id: entry.store_id, - group_id: entry.group_id, - replicas: entry.replicas, - }) - .collect(); - groups.sort_by_key(|g| (g.store_id, g.group_id)); - let mut disk_groups: Vec = persisted - .disk_group - .into_values() - .map(|entry| DiskGroupEntry { - id: entry.id, - rack_id: entry.rack_id, - node_id: entry.node_id, - name: entry.name, - }) - .collect(); - disk_groups.sort_by_key(|dg| dg.id); - let mut disks: Vec = persisted - .disk - .into_values() - .map(|entry| DiskEntry { - disk_id: entry.disk_id, - disk_group_id: entry.disk_group_id, - rack_id: entry.rack_id, - node_id: entry.node_id, - disk_type: entry.disk_type, - capacity_bytes: entry.capacity_bytes, - zone_size_bytes: entry.zone_size_bytes, - unit_size_bytes: entry.unit_size_bytes, - device_path: entry.device_path, - }) - .collect(); - disks.sort_by(|a, b| a.disk_id.cmp(&b.disk_id)); - Self { - racks, - nodes, - servers, - stores, - groups, - disk_groups, - disks, - local_launches: persisted.local_launch, - bench: persisted.bench, - } - } - - fn from_toml_str(body: &str, path: &Path) -> Result { - let persisted: PersistedConsoleConfig = - toml::from_str(body).map_err(|e| Error::Config(format!("{}: {e}", path.display())))?; - Ok(Self::from_persisted(persisted)) - } - - fn to_toml_string(&self) -> Result { - toml::to_string_pretty(&self.to_persisted()).map_err(|e| Error::Config(format!("serialize: {e}"))) - } } #[cfg(test)] mod tests { - use super::{ - ConsoleConfig, GroupEntry, LocalLaunchSpec, ReplicaEntry, ServerEntry, StoreEntry, TomlFileEngine, - }; - use crowdb_test_harness::test_dirs; - - #[test] - fn round_trip_load_save() { - let dir = tempdir(); - let path = dir.join("console.toml"); - - let mut cfg = ConsoleConfig::default(); - let mut a = ServerEntry::new("a", "http://127.0.0.1:10000"); - a.node_id = Some(1); - a.rpc_url = Some("http://127.0.0.1:9921".into()); - a.rest_port = Some(10000); - a.rpc_port = Some(9921); - a.auto_start = true; - a.election_profile = Some("test".into()); - a.pid = Some(12345); - cfg.add_server(a).unwrap(); - cfg.add_server(ServerEntry::new("b", "http://127.0.0.1:10001")) - .unwrap(); - cfg.local_launches.insert( - "b".into(), - LocalLaunchSpec { - program: "/example/deploy/bin/crowdb-diskdb".into(), - args: vec!["--config".into(), "conf/server.toml".into()], - workdir: "/example/deploy".into(), - env: std::collections::BTreeMap::from([("LD_LIBRARY_PATH".into(), "/example/lib".into())]), - readiness_url: Some("http://127.0.0.1:10002".into()), - }, - ); - cfg.stores.push(StoreEntry { - store_id: 7, - nodes: vec![1, 2], - }); - cfg.groups.push(GroupEntry { - store_id: 7, - group_id: 70, - replicas: vec![ - ReplicaEntry { - replica_id: 700, - node_id: 1, - }, - ReplicaEntry { - replica_id: 701, - node_id: 2, - }, - ], - }); - - cfg.save(&path).unwrap(); - let loaded = ConsoleConfig::load(&path).unwrap(); - let expected = cfg.clone(); - assert_eq!(expected, loaded); - } - - #[test] - fn pid_is_persisted_to_disk() { - let dir = tempdir(); - let path = dir.join("console.toml"); - - let mut cfg = ConsoleConfig::default(); - let mut entry = ServerEntry::new("a", "http://127.0.0.1:10000"); - entry.pid = Some(4242); - cfg.add_server(entry).unwrap(); - - cfg.save(&path).unwrap(); - let raw = std::fs::read_to_string(&path).unwrap(); - assert!(raw.contains("pid = 4242"), "runtime pid must be persisted: {raw}"); - } - - #[test] - fn missing_file_yields_default() { - let dir = tempdir(); - let path = dir.join("nope.toml"); - let cfg = ConsoleConfig::load(&path).unwrap(); - assert!(cfg.servers.is_empty()); - } - - #[test] - fn toml_engine_round_trip() { - let dir = tempdir(); - let path = dir.join("engine.toml"); - let engine = TomlFileEngine::new(path.clone()); - - let mut cfg = ConsoleConfig::default(); - cfg.add_server(ServerEntry::new("a", "http://127.0.0.1:10000")) - .unwrap(); - - cfg.save_with_engine(&engine).unwrap(); - let loaded = ConsoleConfig::load_with_engine(&engine).unwrap(); - - assert_eq!(cfg, loaded); - } - - #[test] - fn default_path_points_to_persistent_runtime_namespace() { - let expected = crowdb_protocol::port::namespace::runtime_root() - .join("persistent") - .join("console") - .join("crowdb-kv.db.toml"); - assert_eq!(TomlFileEngine::default_path().unwrap(), expected); - assert_eq!(ConsoleConfig::default_path().unwrap(), expected); - } + use super::{ConsoleConfig, ServerEntry}; #[test] fn duplicate_id_rejected() { @@ -1316,22 +664,4 @@ mod tests { let err = cfg.remove_server("ghost").unwrap_err(); assert!(matches!(err, crate::error::Error::NotFound { .. })); } - - fn tempdir() -> std::path::PathBuf { - use std::sync::atomic::{AtomicU64, Ordering}; - static COUNTER: AtomicU64 = AtomicU64::new(0); - let base = test_dirs::test_data_dir(); - let unique = format!( - "crowdb-console-cfg-{}-{}-{}", - std::process::id(), - COUNTER.fetch_add(1, Ordering::Relaxed), - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_nanos() - ); - let dir = base.join(unique); - std::fs::create_dir_all(&dir).unwrap(); - dir - } } diff --git a/lib/crowdb-console-shared/src/launch/remote.rs b/lib/crowdb-console-shared/src/launch/remote.rs index 1d2ac73a6..1c155bd0c 100644 --- a/lib/crowdb-console-shared/src/launch/remote.rs +++ b/lib/crowdb-console-shared/src/launch/remote.rs @@ -24,6 +24,7 @@ async fn connect(launch: &LaunchRecord, credential_root: &Path) -> Result // Licensed under the Apache License, Version 2.0. -//! Cluster-level operations: status, topology, init, reset, clean. +//! Cluster-level operations: status, topology, init, destroy, clean. //! //! `init` bootstraps group 0 (store 0, group 0) on the selected nodes, //! wires remotes, and writes the hardware + KV-cluster topology into -//! group-0 sysdata. `reset` tears down the cluster in dependency order. -//! `clean` removes orphaned sysdata entries without touching running -//! servers. +//! group-0 sysdata. `destroy` tears down confirmed membership in +//! dependency order. `clean` wipes data on confirmed replicas. use std::collections::{HashMap, HashSet}; use std::sync::Arc; @@ -27,14 +26,13 @@ use crate::ops::hardware::{self, AddDiskInput}; use crate::ops::OpContext; mod bootstrap; -pub use bootstrap::{init, InitSummary}; +pub use bootstrap::{init, init_with_intent, InitSummary}; +// Bootstrap runs before Group 0 registration exists, so its sealed intent +// supplies the initial management endpoint for each selected node. fn server_client(ctx: &OpContext, node_id: u64) -> Result { let url = ctx.node_mgmt_url(node_id)?; - ServerClient::new(&url).map_err(|e| Error::UpstreamRpc { - node_id: url, - status: format!("client build: {e}"), - }) + ServerClient::new(&url) } /// Get cluster status: list all stores from group-0 sysdata. @@ -50,115 +48,51 @@ pub async fn status(ctx: &OpContext) -> Result Result> { - let client = server_client(ctx, node_id)?; + let url = ctx.live_node_mgmt_url(node_id).await?; + let client = ServerClient::new(&url)?; client.topology().await } -/// Reset the cluster: tear down all groups, stores, and sysdata in -/// dependency order. Stops all running servers first. +/// Destroy the confirmed cluster in dependency order. Process shutdown is +/// handled by the caller's local launch runtime after metadata teardown. /// /// # Errors -/// Returns an error if any teardown step fails (best-effort: continues -/// on partial failures and returns the first error). +/// Returns an error on the first failed or unconfirmed teardown step. pub async fn destroy(ctx: &OpContext) -> Result<()> { - let cfg = ctx.config().clone(); - - // Phase 1: remove all non-system groups while the KV management APIs are - // still reachable. - for server in &cfg.servers { - if server.service_type != crate::config::ServiceType::Kv { - continue; - } - if let Some(node_id) = server.node_id { - if let Ok(client) = server_client(ctx, node_id) { - if let Ok(stores) = client.topology().await { - for s in &stores { - if s.store_id == 0 { - continue; - } - let _ = client.remove_store(s.store_id).await; - } - } - } - } - } - - // Phase 2: clear sysdata, then remove group 0 last (best-effort). - let sysmd = ctx.sysmd(); - let stores = sysmd.list_stores().await.unwrap_or_default(); - for s in &stores { - let _ = sysmd.remove_store(s.store_id).await; - } - for server in &cfg.servers { - if server.service_type != crate::config::ServiceType::Kv { - continue; - } - if let Some(node_id) = server.node_id { - if let Ok(client) = server_client(ctx, node_id) { - let _ = client.remove_group(0, 0).await; - } - } - } - - // Phase 3: stop all running services concurrently. A graceful stop may - // consume the full per-process timeout, so serial waits can exceed the - // CLI lifecycle bound and leave the persisted config pointing at dead - // processes. - let mut stop_handles = Vec::with_capacity(cfg.servers.len()); - for pid in cfg.servers.iter().filter_map(|server| server.pid) { - stop_handles.push(tokio::task::spawn_blocking(move || { - let _ = crate::lifecycle::stop_pid(pid); - })); + let stores = ctx.sysmd().list_stores().await?; + let system = stores + .iter() + .find(|store| store.store_id == 0) + .ok_or_else(|| Error::NotFound { + kind: "system store".into(), + id: "0".into(), + })?; + let system_nodes = system.node_ids.clone(); + let mut system_clients = Vec::with_capacity(system_nodes.len()); + for node_id in system_nodes { + let url = ctx.live_node_mgmt_url(node_id).await?; + system_clients.push(ServerClient::new(&url)?); } - for handle in stop_handles { - let _ = handle.await; + if system_clients.is_empty() { + return Err(Error::Validation { + field: "system store".into(), + message: "Group 0 has no live hosts".into(), + }); } - - // Phase 4: clear local config. - { - let mut cfg = ctx.config_mut(); - cfg.stores.clear(); - cfg.groups.clear(); - cfg.servers.clear(); - cfg.local_launches.clear(); - cfg.disks.clear(); - cfg.disk_groups.clear(); - cfg.nodes.clear(); - cfg.racks.clear(); + for store in stores.iter().filter(|store| store.store_id != 0) { + super::kv_logical::remove_store(ctx, store.store_id).await?; } - - Ok(()) -} - -/// Remove orphaned sysdata entries (stores/groups/replicas that have -/// no corresponding running server). Does not stop any running -/// servers. -/// -/// # Errors -/// Returns an error if the sysdata scan fails. -pub async fn reset(ctx: &OpContext) -> Result<()> { - let sysmd = ctx.sysmd(); - let stores = sysmd.list_stores().await?; - - // For each store, check if any hosting node has a running server. - let cfg = ctx.config().clone(); - for store in &stores { - let mut any_alive = false; - for node_id in &store.node_ids { - if cfg.server_for_node(*node_id).is_some() { - if let Ok(client) = server_client(ctx, *node_id) { - if client.health().await.is_ok() { - any_alive = true; - break; - } - } - } - } - if !any_alive { - let _ = sysmd.remove_store(store.store_id).await; + for group in ctx.sysmd().list_groups_in_store(0).await? { + if group.group_id != 0 { + super::kv_logical::remove_group(ctx, 0, group.group_id).await?; } } - + // Keep the system group available until all other metadata is gone. + // Resolve every endpoint before removing any member. + for client in &system_clients { + client.remove_group(0, 0).await?; + client.remove_store(0).await?; + } Ok(()) } @@ -179,13 +113,20 @@ pub struct CleanResult { /// # Errors /// Returns an error if no servers are configured. pub async fn clean(ctx: &OpContext, store_id: u64, group_id: u64) -> Result { - let cfg = ctx.config().clone(); - let mut mgmt_urls: Vec = cfg - .servers - .iter() - .filter(|server| server.service_type == ServiceType::Kv) - .map(|server| server.url.clone()) - .collect(); + // The confirmed replica set determines which nodes must be wiped. A local + // launch registry may contain stopped or unrelated processes, and cannot + // substitute for Group 0 membership after bootstrap. + let replicas = ctx.sysmd().list_replicas_in_group(store_id, group_id).await?; + if replicas.is_empty() { + return Err(Error::NotFound { + kind: "group replicas".into(), + id: format!("{store_id}/{group_id}"), + }); + } + let mut mgmt_urls = Vec::with_capacity(replicas.len()); + for replica in replicas { + mgmt_urls.push(ctx.live_node_mgmt_url(replica.node_id).await?); + } mgmt_urls.sort(); mgmt_urls.dedup(); if mgmt_urls.is_empty() { @@ -477,8 +418,22 @@ pub async fn local_deploy_combined( diskio_dummy_disk_type: &str, ) -> Result { local_deploy(ctx, 3, Some(workspace), tunables).await?; + local_deploy_combined_after_kv(ctx, workspace, disk, chunk, diskio_dummy_disk_type).await +} + +/// Complete the local storage stack after a verified KV bootstrap. +/// +/// # Errors +/// Returns a provisioning or readiness error. +pub async fn local_deploy_combined_after_kv( + ctx: &OpContext, + workspace: &std::path::Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, + diskio_dummy_disk_type: &str, +) -> Result { for group_id in &disk.data_groups { - crate::ops::kv_logical::add_group(ctx, 0, *group_id, 100 + *group_id, &[1, 2, 3]).await?; + ensure_local_data_group(ctx, *group_id).await?; } let diskdb = local_deploy_diskdb(ctx, workspace, disk).await?; let diskio = local_deploy_diskio( @@ -514,8 +469,21 @@ pub async fn local_deploy_combined_file_backed( chunk: &LocalChunkdbDeployConfig, ) -> Result { local_deploy(ctx, 3, Some(workspace), tunables).await?; + local_deploy_combined_file_backed_after_kv(ctx, workspace, disk, chunk).await +} + +/// Complete the file-backed storage stack after a verified KV bootstrap. +/// +/// # Errors +/// Returns a provisioning or readiness error. +pub async fn local_deploy_combined_file_backed_after_kv( + ctx: &OpContext, + workspace: &std::path::Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, +) -> Result { for group_id in &disk.data_groups { - crate::ops::kv_logical::add_group(ctx, 0, *group_id, 100 + *group_id, &[1, 2, 3]).await?; + ensure_local_data_group(ctx, *group_id).await?; } let diskdb = local_deploy_diskdb(ctx, workspace, disk).await?; let diskio = local_deploy_diskio( @@ -537,6 +505,28 @@ pub async fn local_deploy_combined_file_backed( }) } +async fn ensure_local_data_group(ctx: &OpContext, group_id: u64) -> Result<()> { + let group = ctx.sysmd().get_group(0, group_id).await?; + let replicas = ctx.sysmd().list_replicas_in_group(0, group_id).await?; + if group.is_none() && replicas.is_empty() { + return crate::ops::kv_logical::add_group(ctx, 0, group_id, 100 + group_id, &[1, 2, 3]).await; + } + let mut actual: Vec<_> = replicas + .iter() + .map(|replica| (replica.replica_id, replica.node_id)) + .collect(); + actual.sort_unstable(); + let expected = vec![(100 + group_id, 1), (101 + group_id, 2), (102 + group_id, 3)]; + if group.is_some() && actual == expected { + Ok(()) + } else { + Err(Error::Conflict { + kind: "local data group".into(), + id: format!("0/{group_id}"), + }) + } +} + async fn local_deploy_diskio( ctx: &OpContext, workspace: &std::path::Path, @@ -874,7 +864,7 @@ pub async fn local_deploy_diskdb( workspace: &std::path::Path, cfg: &LocalDiskdbDeployConfig, ) -> Result { - let nodes = validate_diskdb_deploy(ctx, cfg)?; + let nodes = validate_diskdb_deploy(ctx, cfg).await?; ensure_diskdb_hardware(ctx, &nodes).await?; let (disk_group_count, disk_count) = provision_diskdb_topology(ctx, &nodes, cfg).await?; let ports = alloc_diskdb_ports(workspace, nodes.len())?; @@ -891,12 +881,20 @@ pub async fn local_deploy_diskdb( async fn ensure_diskdb_hardware(ctx: &OpContext, nodes: &[NodeEntry]) -> Result<()> { for node in nodes { if ctx.sysmd().get_rack(node.rack_id).await?.is_none() { + let name = ctx + .config() + .racks + .iter() + .find(|rack| rack.id == node.rack_id) + .map(|rack| rack.name.clone()) + .unwrap_or_default(); ctx.sysmd() .add_rack( node.rack_id, &RackValue { status: HwStatus::Up as i32, node_ids: Vec::new(), + name, }, ) .await?; @@ -912,6 +910,10 @@ async fn ensure_diskdb_hardware(ctx: &OpContext, nodes: &[NodeEntry]) -> Result< disk_group_ids: Vec::new(), status_changed_at_ms: 0, temp_failure_since_ms: None, + management_host: node.host.clone(), + ssh_port: node.ssh_port, + ssh_user: node.ssh_user.clone(), + ssh_credential_ref: node.ssh_credential_ref.clone(), }, ) .await?; @@ -920,7 +922,7 @@ async fn ensure_diskdb_hardware(ctx: &OpContext, nodes: &[NodeEntry]) -> Result< Ok(()) } -fn validate_diskdb_deploy(ctx: &OpContext, cfg: &LocalDiskdbDeployConfig) -> Result> { +async fn validate_diskdb_deploy(ctx: &OpContext, cfg: &LocalDiskdbDeployConfig) -> Result> { if cfg.disk_groups_per_node == 0 || cfg.disks_per_group == 0 || cfg.data_groups.is_empty() { return Err(Error::Validation { field: "diskdb_topology".into(), @@ -935,12 +937,9 @@ fn validate_diskdb_deploy(ctx: &OpContext, cfg: &LocalDiskdbDeployConfig) -> Res message: "deploy the KV cluster before DiskDB".into(), }); } - let configured_groups = ctx.config().groups.clone(); + ctx.kv().refresh_topology().await?; for group_id in &cfg.data_groups { - if !configured_groups - .iter() - .any(|group| group.store_id == 0 && group.group_id == *group_id) - { + if ctx.sysmd().get_group(0, *group_id).await?.is_none() { return Err(Error::NotFound { kind: "kv_group".into(), id: format!("0:{group_id}"), @@ -966,8 +965,13 @@ async fn provision_diskdb_topology( for node in nodes { for local_group in 0..cfg.disk_groups_per_node { let disk_group_id = node.id * 100 + u64::try_from(local_group).unwrap_or(u64::MAX) + 1; - hardware::add_disk_group(ctx, node.id, disk_group_id, &format!("bench-dg-{disk_group_id}")) - .await?; + hardware::add_disk_group_to_group0( + ctx, + node.id, + disk_group_id, + &format!("bench-dg-{disk_group_id}"), + ) + .await?; let disks = (0..cfg.disks_per_group) .map(|disk| AddDiskInput { disk_id: format!("{:016x}{:016x}", disk_group_id, disk + 1), @@ -978,15 +982,26 @@ async fn provision_diskdb_topology( device_path: String::new(), }) .collect::>(); - hardware::add_disks_batch(ctx, node.id, disk_group_id, &disks).await?; + for disk in &disks { + hardware::add_disk_to_group0(ctx, node.id, disk_group_id, disk).await?; + } let instance_id = 10_000 + node.id; ctx.sysmd() .set_owner(node.rack_id, node.id, disk_group_id, instance_id, lease_expiry_ms) .await?; let data_group = cfg.data_groups[disk_group_count % cfg.data_groups.len()]; - ctx.sysmd() - .set_bind(node.rack_id, node.id, disk_group_id, 0, data_group) - .await?; + if let Some(binding) = ctx.sysmd().get_bind(node.rack_id, node.id, disk_group_id).await? { + if binding.store_id != 0 || binding.group_id != data_group { + return Err(Error::Conflict { + kind: "disk group binding".into(), + id: disk_group_id.to_string(), + }); + } + } else { + ctx.sysmd() + .set_bind(node.rack_id, node.id, disk_group_id, 0, data_group) + .await?; + } disk_group_count += 1; disk_count += disks.len(); } @@ -1153,6 +1168,26 @@ pub async fn local_deploy( workspace_dir: Option<&std::path::Path>, tunables: Option<&KvDeployTunables>, ) -> Result { + let (rack_id, node_ids) = prepare_local_deploy(ctx, node_count, workspace_dir, tunables).await?; + let init_summary = init(ctx, &node_ids).await?; + Ok(LocalDeploySummary { + node_count, + rack_id, + node_ids, + init_summary, + }) +} + +/// Start local KV processes and retain their bootstrap inputs without writing Group 0. +/// +/// # Errors +/// Returns a validation, binary, spawn, or readiness error. +pub async fn prepare_local_deploy( + ctx: &OpContext, + node_count: usize, + workspace_dir: Option<&std::path::Path>, + tunables: Option<&KvDeployTunables>, +) -> Result<(u64, Vec)> { if node_count == 0 { return Err(Error::Validation { field: "node_count".into(), @@ -1193,14 +1228,7 @@ pub async fn local_deploy( } } - let init_summary = init(ctx, &node_ids).await?; - - Ok(LocalDeploySummary { - node_count, - rack_id, - node_ids, - init_summary, - }) + Ok((rack_id, node_ids)) } /// Default workspace path for `local_deploy` when no explicit @@ -1269,6 +1297,7 @@ fn write_rack_and_nodes(ctx: &OpContext, rack_id: u64, node_ids: &[u64]) { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); } } @@ -1325,6 +1354,7 @@ async fn deploy_servers( ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; // Every process owns a stable server directory. WAL and btree data // remain direct children of that directory as waldata/ and ctdata/. diff --git a/lib/crowdb-console-shared/src/ops/cluster/bootstrap.rs b/lib/crowdb-console-shared/src/ops/cluster/bootstrap.rs index 2ecd3e293..7356991ca 100644 --- a/lib/crowdb-console-shared/src/ops/cluster/bootstrap.rs +++ b/lib/crowdb-console-shared/src/ops/cluster/bootstrap.rs @@ -4,10 +4,12 @@ //! Initialize system-group processes and publish bootstrap metadata. use super::server_client; +use crate::bootstrap_intent::BootstrapIntent; use crate::config::{ReplicaEntry, ServiceType}; use crate::error::{Error, Result}; use crate::ops::OpContext; use std::collections::{HashMap, HashSet}; +use std::path::Path; mod leader; mod nodes; @@ -21,6 +23,40 @@ pub struct InitSummary { pub nodes: Vec<(u64, u64)>, } +/// Initialize from a sealed pre-Group-0 intent, then delete it after verified +/// publication. A retry can restore an empty in-memory console context. +/// +/// # Errors +/// Rejects changed member or topology identity and propagates bootstrap errors. +pub async fn init_with_intent(ctx: &OpContext, nodes: &[u64], path: &Path) -> Result { + let intent = if std::fs::symlink_metadata(path).is_ok() { + let sealed = BootstrapIntent::load(path)?; + if sealed.members() != nodes { + return Err(Error::Conflict { + kind: "bootstrap members".into(), + id: path.display().to_string(), + }); + } + let current = ctx.config().clone(); + if current.racks.is_empty() && current.nodes.is_empty() && current.servers.is_empty() { + *ctx.config_mut() = sealed.to_config(); + } else if BootstrapIntent::capture(¤t, nodes)? != sealed { + return Err(Error::Conflict { + kind: "bootstrap topology".into(), + id: path.display().to_string(), + }); + } + sealed + } else { + let captured = BootstrapIntent::capture(&ctx.config(), nodes)?; + captured.seal(path)?; + captured + }; + let result = init(ctx, intent.members()).await?; + BootstrapIntent::clear_verified(path)?; + Ok(result) +} + /// Initialize the cluster by bootstrapping group 0 on the listed nodes. /// /// # Errors diff --git a/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs b/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs index 86e926aeb..48ea52246 100644 --- a/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs +++ b/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs @@ -32,7 +32,7 @@ impl Record { GetOutcome::Found { value, .. } => { let actual: serde_json::Value = serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?; - if actual == self.value { + if self.same_identity(ctx, &actual)? { Ok(true) } else { Err(Error::Conflict { @@ -44,6 +44,38 @@ impl Record { } } + fn same_identity(&self, ctx: &OpContext, actual: &serde_json::Value) -> Result { + if self.key.starts_with("/hw/rack/") { + let intended: RackValue = serde_json::from_value(self.value.clone()) + .map_err(|error| Error::Config(error.to_string()))?; + let actual: RackValue = + serde_json::from_value(actual.clone()).map_err(|error| Error::Config(error.to_string()))?; + let rack_id = RackKey::from_path(&self.key) + .map_err(|error| Error::Config(error.to_string()))? + .rack_id; + let expected_nodes: Vec<_> = ctx + .config() + .nodes + .iter() + .filter(|node| node.rack_id == rack_id) + .map(|node| node.id) + .collect(); + return Ok(actual.name == intended.name + && actual.node_ids.iter().all(|node| expected_nodes.contains(node))); + } + if self.key.starts_with("/hw/node/") { + let intended: NodeValue = serde_json::from_value(self.value.clone()) + .map_err(|error| Error::Config(error.to_string()))?; + let actual: NodeValue = + serde_json::from_value(actual.clone()).map_err(|error| Error::Config(error.to_string()))?; + return Ok(actual.management_host == intended.management_host + && actual.ssh_port == intended.ssh_port + && actual.ssh_user == intended.ssh_user + && actual.ssh_credential_ref == intended.ssh_credential_ref); + } + Ok(actual == &self.value) + } + async fn create(&self, ctx: &OpContext) -> Result<()> { let payload = serde_json::to_vec(&self.value).map_err(|error| Error::Config(error.to_string()))?; match ctx.kv().put_cas(0, 0, self.key.as_bytes(), &payload, 0).await { @@ -97,6 +129,7 @@ fn intended_records(ctx: &OpContext, store_nodes: &[u64], members: &[(u64, u64)] RackValue { status: HwStatus::Up as i32, node_ids: Vec::new(), + name: rack.name.clone(), }, )?); } @@ -108,6 +141,10 @@ fn intended_records(ctx: &OpContext, store_nodes: &[u64], members: &[(u64, u64)] }, NodeValue { status: HwStatus::Up as i32, + management_host: node.host.clone(), + ssh_port: node.ssh_port, + ssh_user: node.ssh_user.clone(), + ssh_credential_ref: node.ssh_credential_ref.clone(), ..Default::default() }, )?); diff --git a/lib/crowdb-console-shared/src/ops/context.rs b/lib/crowdb-console-shared/src/ops/context.rs index 2757508e2..e0b13dd8c 100644 --- a/lib/crowdb-console-shared/src/ops/context.rs +++ b/lib/crowdb-console-shared/src/ops/context.rs @@ -17,9 +17,8 @@ use crate::error::{Error, Result}; /// (hardware hierarchy, KV-cluster topology, service registry). /// - **`kv`** — a [`CrowdbKvClient`] for the KV data-plane (put/get/ /// delete/scan on user stores/groups). -/// - **`config`** — the local TOML [`ConsoleConfig`] (rack/node/server -/// entries, bootstrap state). Mutated under an `RwLock` and persisted -/// by the caller via the engine. +/// - **`config`** — ephemeral [`ConsoleConfig`] inputs for bootstrap and +/// local benchmark deployment. Group 0 remains the topology authority. /// - **`discovery`** — an optional [`ServiceDiscoveryClient`] for /// discovering living service instances (diskdb, chunkdb, etc.) via /// the group-0 service registry. `None` when the caller (e.g. a unit diff --git a/lib/crowdb-console-shared/src/ops/hardware.rs b/lib/crowdb-console-shared/src/ops/hardware.rs index 09c6e1f98..10f843d22 100644 --- a/lib/crowdb-console-shared/src/ops/hardware.rs +++ b/lib/crowdb-console-shared/src/ops/hardware.rs @@ -15,6 +15,125 @@ use crate::config::{DiskEntry, DiskGroupEntry, NodeEntry, RackEntry}; use crate::error::{Error, Result}; use crate::ops::OpContext; +mod authority; +mod authority_storage; + +pub use authority_storage::{ + add_disk_group_to_group0, add_disk_to_group0, list_disk_groups_from_group0, list_disks_from_group0, + remove_disk_from_group0, remove_disk_group_from_group0, +}; + +/// Create a rack only after Group 0 confirms the exact record. +/// +/// # Errors +/// Returns a conflict for a different existing record, or the authority error. +pub async fn add_rack_to_group0(ctx: &OpContext, rack_id: u64, name: &str) -> Result { + let entry = RackEntry { + id: rack_id, + name: name.to_owned(), + }; + let created = authority::create( + ctx, + crowdb_protocol::key::RackKey { rack_id }, + &RackValue { + status: HwStatus::Up as i32, + node_ids: Vec::new(), + name: name.to_owned(), + }, + ) + .await; + if let Err(Error::Conflict { .. }) = &created { + if ctx + .sysmd() + .get_rack(rack_id) + .await? + .is_some_and(|rack| rack.name == name) + { + return Ok(entry); + } + } + created?; + Ok(entry) +} + +/// Create a node only after its rack and exact record are confirmed in Group 0. +/// +/// # Errors +/// Returns a missing rack, conflicting node, or authority error. +pub async fn add_node_to_group0(ctx: &OpContext, entry: NodeEntry) -> Result { + let value = NodeValue { + status: HwStatus::Up as i32, + management_host: entry.host.clone(), + ssh_port: entry.ssh_port, + ssh_user: entry.ssh_user.clone(), + ssh_credential_ref: entry.ssh_credential_ref.clone(), + ..Default::default() + }; + authority::create_node(ctx, entry.rack_id, entry.id, &value).await?; + Ok(entry) +} + +/// Read rack names from confirmed Group 0 state. +/// +/// # Errors +/// Returns an authority error; a local topology file is never consulted. +pub async fn list_racks_from_group0(ctx: &OpContext) -> Result> { + authority::ready(ctx).await?; + let mut racks: Vec<_> = ctx + .sysmd() + .list_racks() + .await? + .into_iter() + .map(|(id, value)| RackEntry { id, name: value.name }) + .collect(); + racks.sort_unstable_by_key(|rack| rack.id); + Ok(racks) +} + +/// Read node connection identity from confirmed Group 0 state. +/// +/// # Errors +/// Returns an authority error; private SSH material is never returned. +pub async fn list_nodes_from_group0(ctx: &OpContext, rack_id: Option) -> Result> { + authority::ready(ctx).await?; + let mut nodes: Vec<_> = ctx + .sysmd() + .list_nodes() + .await? + .into_iter() + .filter(|(rack, _, _)| rack_id.is_none() || rack_id == Some(*rack)) + .map(|(rack_id, id, value)| NodeEntry { + id, + rack_id, + host: value.management_host, + ssh_port: value.ssh_port, + ssh_user: value.ssh_user, + ssh_key: None, + ssh_password: None, + ssh_credential_ref: value.ssh_credential_ref, + }) + .collect(); + nodes.sort_unstable_by_key(|node| node.id); + Ok(nodes) +} + +/// Remove an empty rack after Group 0 confirms no node belongs to it. +/// +/// # Errors +/// Returns a conflict if children remain, or the authority error. +pub async fn remove_rack_from_group0(ctx: &OpContext, rack_id: u64) -> Result<()> { + authority::remove_empty_rack(ctx, rack_id).await +} + +/// Remove an unused node and its rack membership in one confirmed Group 0 write. +/// +/// # Errors +/// Rejects a node with disk groups or KV replicas, a missing node, or an +/// uncertain authority result. +pub async fn remove_node_from_group0(ctx: &OpContext, node_id: u64) -> Result<()> { + authority::remove_empty_node(ctx, node_id).await +} + // ── rack ──────────────────────────────────────────────────────── /// Add a rack to the local config and group-0 sysdata. @@ -35,6 +154,7 @@ pub async fn add_rack(ctx: &OpContext, rack_id: u64, name: &str) -> Result Result { disk_group_ids: Vec::new(), status_changed_at_ms: 0, temp_failure_since_ms: None, + management_host: entry.host.clone(), + ssh_port: entry.ssh_port, + ssh_user: entry.ssh_user.clone(), + ssh_credential_ref: entry.ssh_credential_ref.clone(), }; let _ = ctx.sysmd().add_node(entry.rack_id, entry.id, &value).await; } @@ -153,6 +277,7 @@ pub async fn add_disk_group(ctx: &OpContext, node_id: u64, dg_id: u64, name: &st let value = crowdb_protocol::diskdb::rpc::DiskGroupValue { status: HwStatus::Up as i32, disk_ids: Vec::new(), + name: String::new(), }; let _ = ctx.sysmd().add_disk_group(rack_id, node_id, dg_id, &value).await; } @@ -219,7 +344,7 @@ pub fn list_disk_groups(ctx: &OpContext, node_id: u64) -> Vec { // ── disk ──────────────────────────────────────────────────────── /// Input for adding a disk. Mirrors the web handler's `AddDiskBody`. -#[derive(Debug, Clone)] +#[derive(Debug, Clone, serde::Deserialize)] pub struct AddDiskInput { pub disk_id: String, pub disk_type: String, @@ -565,6 +690,7 @@ mod tests { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); config diff --git a/lib/crowdb-console-shared/src/ops/hardware/authority.rs b/lib/crowdb-console-shared/src/ops/hardware/authority.rs new file mode 100644 index 000000000..6cedce19e --- /dev/null +++ b/lib/crowdb-console-shared/src/ops/hardware/authority.rs @@ -0,0 +1,348 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Conditional Group 0 hardware publication with confirmed outcomes. + +use crowdb_kv_client::{BatchOp, Error as KvError, GetOutcome, ReadMode}; +use crowdb_protocol::key::TextKey; +use crowdb_protocol::{ + common::{NodeValue, RackValue}, + key::{NodeKey, RackKey}, +}; + +use crate::error::{Error, Result}; +use crate::ops::OpContext; + +pub(super) async fn ready(ctx: &OpContext) -> Result<()> { + ctx.kv().refresh_topology().await?; + Ok(()) +} + +pub(super) async fn create(ctx: &OpContext, key: impl TextKey, value: &T) -> Result<()> { + ready(ctx).await?; + let path = key.to_path(); + let intended = serde_json::to_value(value).map_err(|error| Error::Config(error.to_string()))?; + let payload = serde_json::to_vec(value).map_err(|error| Error::Config(error.to_string()))?; + match ctx.kv().put_cas(0, 0, path.as_bytes(), &payload, 0).await { + Ok(_) => Ok(()), + Err(error @ (KvError::CasFailed { .. } | KvError::OutcomeUnknown)) => { + match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, .. } => { + let actual: serde_json::Value = + serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?; + if actual == intended { + Ok(()) + } else { + Err(Error::Conflict { + kind: "hardware".into(), + id: path, + }) + } + } + GetOutcome::NotFound => Err(error.into()), + } + } + Err(error) => Err(error.into()), + } +} + +/// Update the rack membership and create its node in one conditional Group 0 write. +pub(super) async fn create_node( + ctx: &OpContext, + rack_id: u64, + node_id: u64, + value: &NodeValue, +) -> Result<()> { + ready(ctx).await?; + let rack_path = RackKey { rack_id }.to_path(); + let node_path = NodeKey { rack_id, node_id }.to_path(); + for attempt in 0..10u64 { + let (mut rack, revision) = match ctx + .kv() + .get(0, 0, rack_path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, revision, .. } => ( + serde_json::from_slice::(&value) + .map_err(|error| Error::Config(error.to_string()))?, + revision, + ), + GetOutcome::NotFound => { + return Err(Error::NotFound { + kind: "rack".into(), + id: rack_id.to_string(), + }) + } + }; + let existing = ctx + .kv() + .get(0, 0, node_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + let node_exists = matches!(existing, GetOutcome::Found { .. }); + if let GetOutcome::Found { value: stored, .. } = existing { + let actual: NodeValue = + serde_json::from_slice(&stored).map_err(|error| Error::Config(error.to_string()))?; + if !same_node_connection(&actual, value) { + return Err(Error::Conflict { + kind: "node".into(), + id: node_id.to_string(), + }); + } + if rack.node_ids.contains(&node_id) { + return Ok(()); + } + } + if !rack.node_ids.contains(&node_id) { + rack.node_ids.push(node_id); + rack.node_ids.sort_unstable(); + } + let rack_bytes = serde_json::to_vec(&rack).map_err(|error| Error::Config(error.to_string()))?; + let mut ops = vec![BatchOp::Put { + key: rack_path.as_bytes().to_vec().into(), + value: rack_bytes.into(), + }]; + if !node_exists { + let node_bytes = serde_json::to_vec(value).map_err(|error| Error::Config(error.to_string()))?; + ops.push(BatchOp::Put { + key: node_path.as_bytes().to_vec().into(), + value: node_bytes.into(), + }); + } + match ctx + .kv() + .batch_write_cas(0, 0, &ops, rack_path.as_bytes(), revision) + .await + { + Ok(_) => return Ok(()), + Err(KvError::CasFailed { .. } | KvError::CasBusy) => { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1) * 5)).await; + } + Err(KvError::OutcomeUnknown) => { + let confirmed = ctx + .kv() + .get(0, 0, node_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + if let GetOutcome::Found { value: stored, .. } = confirmed { + let actual: NodeValue = + serde_json::from_slice(&stored).map_err(|error| Error::Config(error.to_string()))?; + if same_node_connection(&actual, value) { + let rack = ctx + .kv() + .get(0, 0, rack_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + if let GetOutcome::Found { value, .. } = rack { + let rack: RackValue = serde_json::from_slice(&value) + .map_err(|error| Error::Config(error.to_string()))?; + if rack.node_ids.contains(&node_id) { + return Ok(()); + } + } + return Err(KvError::OutcomeUnknown.into()); + } + return Err(Error::Conflict { + kind: "node".into(), + id: node_id.to_string(), + }); + } + return Err(KvError::OutcomeUnknown.into()); + } + Err(error) => return Err(error.into()), + } + } + Err(KvError::CasBusy.into()) +} + +fn same_node_connection(actual: &NodeValue, intended: &NodeValue) -> bool { + actual.management_host == intended.management_host + && actual.ssh_port == intended.ssh_port + && actual.ssh_user == intended.ssh_user + && actual.ssh_credential_ref == intended.ssh_credential_ref +} + +/// Delete an empty rack only while its confirmed revision still matches. +pub(super) async fn remove_empty_rack(ctx: &OpContext, rack_id: u64) -> Result<()> { + ready(ctx).await?; + let path = RackKey { rack_id }.to_path(); + for attempt in 0..10u64 { + let (rack, revision) = match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, revision } => ( + serde_json::from_slice::(&value) + .map_err(|error| Error::Config(error.to_string()))?, + revision, + ), + GetOutcome::NotFound => { + return Err(Error::NotFound { + kind: "rack".into(), + id: rack_id.to_string(), + }) + } + }; + if !rack.node_ids.is_empty() || !ctx.sysmd().list_nodes_in_rack(rack_id).await?.is_empty() { + return Err(Error::Conflict { + kind: "rack with nodes".into(), + id: rack_id.to_string(), + }); + } + let ops = [BatchOp::Delete { + key: path.as_bytes().to_vec().into(), + }]; + match ctx + .kv() + .batch_write_cas(0, 0, &ops, path.as_bytes(), revision) + .await + { + Ok(_) => return Ok(()), + Err(KvError::CasFailed { .. } | KvError::CasBusy) => { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1) * 5)).await; + } + Err(KvError::OutcomeUnknown) => { + return match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::NotFound => Ok(()), + GetOutcome::Found { .. } => Err(KvError::OutcomeUnknown.into()), + }; + } + Err(error) => return Err(error.into()), + } + } + Err(KvError::CasBusy.into()) +} + +/// Remove a node only after confirmed authority shows no children or KV replicas. +pub(super) async fn remove_empty_node(ctx: &OpContext, node_id: u64) -> Result<()> { + ready(ctx).await?; + for attempt in 0..10u64 { + let (rack_id, node) = ctx + .sysmd() + .list_nodes() + .await? + .into_iter() + .find_map(|(rack_id, id, value)| (id == node_id).then_some((rack_id, value))) + .ok_or_else(|| Error::NotFound { + kind: "node".into(), + id: node_id.to_string(), + })?; + if node_is_used(ctx, rack_id, node_id, &node).await? { + return Err(Error::Conflict { + kind: "node with children".into(), + id: node_id.to_string(), + }); + } + let rack_path = RackKey { rack_id }.to_path(); + let node_path = NodeKey { rack_id, node_id }.to_path(); + let (mut rack, revision) = match ctx + .kv() + .get(0, 0, rack_path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, revision } => ( + serde_json::from_slice::(&value) + .map_err(|error| Error::Config(error.to_string()))?, + revision, + ), + GetOutcome::NotFound => { + return Err(Error::Conflict { + kind: "node without rack".into(), + id: node_id.to_string(), + }) + } + }; + if !rack.node_ids.contains(&node_id) { + return Err(Error::Conflict { + kind: "node missing from rack membership".into(), + id: node_id.to_string(), + }); + } + rack.node_ids.retain(|id| *id != node_id); + let rack_bytes = serde_json::to_vec(&rack).map_err(|error| Error::Config(error.to_string()))?; + let ops = [ + BatchOp::Put { + key: rack_path.as_bytes().to_vec().into(), + value: rack_bytes.into(), + }, + BatchOp::Delete { + key: node_path.as_bytes().to_vec().into(), + }, + ]; + match ctx + .kv() + .batch_write_cas(0, 0, &ops, rack_path.as_bytes(), revision) + .await + { + Ok(_) => return Ok(()), + Err(KvError::CasFailed { .. } | KvError::CasBusy) => { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1) * 5)).await; + } + Err(KvError::OutcomeUnknown) => { + if node_removal_confirmed(ctx, &rack_path, &node_path, node_id).await? { + return Ok(()); + } + return Err(KvError::OutcomeUnknown.into()); + } + Err(error) => return Err(error.into()), + } + } + Err(KvError::CasBusy.into()) +} + +async fn node_is_used(ctx: &OpContext, rack_id: u64, node_id: u64, node: &NodeValue) -> Result { + if !node.disk_group_ids.is_empty() + || !ctx + .sysmd() + .list_disk_groups_on_node(rack_id, node_id) + .await? + .is_empty() + || ctx + .sysmd() + .list_all_replicas() + .await? + .iter() + .any(|replica| replica.node_id == node_id) + { + return Ok(true); + } + Ok(ctx + .sysmd() + .read_all_kv_server_instances() + .await? + .into_iter() + .any(|(_, instance)| { + instance + .extra + .and_then(|extra| extra.kv_server) + .and_then(|server| server.node_id) + == Some(node_id) + })) +} + +async fn node_removal_confirmed( + ctx: &OpContext, + rack_path: &str, + node_path: &str, + node_id: u64, +) -> Result { + let rack = ctx + .kv() + .get(0, 0, rack_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + let node = ctx + .kv() + .get(0, 0, node_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + let (GetOutcome::Found { value, .. }, GetOutcome::NotFound) = (rack, node) else { + return Ok(false); + }; + let rack: RackValue = serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?; + Ok(!rack.node_ids.contains(&node_id)) +} diff --git a/lib/crowdb-console-shared/src/ops/hardware/authority_storage.rs b/lib/crowdb-console-shared/src/ops/hardware/authority_storage.rs new file mode 100644 index 000000000..3bd5d0434 --- /dev/null +++ b/lib/crowdb-console-shared/src/ops/hardware/authority_storage.rs @@ -0,0 +1,429 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Rack-fenced disk-group and disk mutations in Group 0. + +use crowdb_kv_client::{BatchOp, Error as KvError, GetOutcome, ReadMode}; +use crowdb_protocol::common::{DiskId, HwStatus, NodeValue, RackValue}; +use crowdb_protocol::diskdb::rpc::{DiskGroupValue, DiskValue}; +use crowdb_protocol::key::{DiskGroupKey, DiskKey, NodeKey, RackKey, TextKey}; +use crowdb_protocol::DiskIdExt; + +use crate::config::{DiskEntry, DiskGroupEntry}; +use crate::error::{Error, Result}; +use crate::ops::hardware::{authority, validate_disk_input, AddDiskInput}; +use crate::ops::OpContext; + +const RETRIES: u64 = 10; + +async fn rack_revision(ctx: &OpContext, rack_id: u64) -> Result<(String, RackValue, u64)> { + let path = RackKey { rack_id }.to_path(); + let (value, revision) = required::(ctx, &path, "rack", rack_id.to_string()).await?; + Ok((path, value, revision)) +} + +async fn required( + ctx: &OpContext, + path: &str, + kind: &str, + id: String, +) -> Result<(T, u64)> { + match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, revision, .. } => Ok(( + serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?, + revision, + )), + GetOutcome::NotFound => Err(Error::NotFound { + kind: kind.into(), + id, + }), + } +} + +async fn optional(ctx: &OpContext, path: &str) -> Result> { + match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, .. } => Ok(Some( + serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?, + )), + GetOutcome::NotFound => Ok(None), + } +} + +fn put(path: &str, value: &T) -> Result { + let encoded = serde_json::to_vec(value).map_err(|error| Error::Config(error.to_string()))?; + Ok(BatchOp::Put { + key: path.as_bytes().to_vec().into(), + value: encoded.into(), + }) +} + +fn delete(path: &str) -> BatchOp { + BatchOp::Delete { + key: path.as_bytes().to_vec().into(), + } +} + +async fn fenced( + ctx: &OpContext, + rack_path: &str, + rack: &RackValue, + revision: u64, + mut ops: Vec, +) -> Result { + ops.insert(0, put(rack_path, rack)?); + match ctx + .kv() + .batch_write_cas(0, 0, &ops, rack_path.as_bytes(), revision) + .await + { + Ok(_) => Ok(true), + Err(KvError::CasFailed { .. } | KvError::CasBusy | KvError::OutcomeUnknown) => Ok(false), + Err(error) => Err(error.into()), + } +} + +async fn pause(attempt: u64) { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1) * 5)).await; +} + +fn node_location(nodes: &[(u64, u64, NodeValue)], node_id: u64) -> Result { + nodes + .iter() + .find_map(|(rack, id, _)| (*id == node_id).then_some(*rack)) + .ok_or_else(|| Error::NotFound { + kind: "node".into(), + id: node_id.to_string(), + }) +} + +/// Create a disk group and update its node membership in one confirmed write. +/// +/// # Errors +/// Returns an authority error, a missing node, or a conflicting group. +pub async fn add_disk_group_to_group0( + ctx: &OpContext, + node_id: u64, + dg_id: u64, + name: &str, +) -> Result { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let entry = DiskGroupEntry { + id: dg_id, + rack_id, + node_id, + name: name.into(), + }; + let node_path = NodeKey { rack_id, node_id }.to_path(); + let group_path = DiskGroupKey { + rack_id, + node_id, + disk_group_id: dg_id, + } + .to_path(); + for attempt in 0..RETRIES { + let (rack_path, rack, revision) = rack_revision(ctx, rack_id).await?; + let (mut node, _) = required::(ctx, &node_path, "node", node_id.to_string()).await?; + let current = optional::(ctx, &group_path).await?; + if let Some(group) = current { + if group.name != name || !node.disk_group_ids.contains(&dg_id) { + return Err(Error::Conflict { + kind: "disk_group".into(), + id: dg_id.to_string(), + }); + } + return Ok(entry); + } + if node.disk_group_ids.contains(&dg_id) { + return Err(Error::Conflict { + kind: "disk_group membership".into(), + id: dg_id.to_string(), + }); + } + node.disk_group_ids.push(dg_id); + node.disk_group_ids.sort_unstable(); + node.last_used_dg_id = node.last_used_dg_id.max(dg_id); + let group = DiskGroupValue { + status: HwStatus::Up as i32, + disk_ids: Vec::new(), + name: name.into(), + }; + if fenced( + ctx, + &rack_path, + &rack, + revision, + vec![put(&node_path, &node)?, put(&group_path, &group)?], + ) + .await? + { + return Ok(entry); + } + pause(attempt).await; + } + Err(KvError::OutcomeUnknown.into()) +} + +/// Return confirmed disk groups on one node. +/// +/// # Errors +/// Returns an authority error or a missing node. +pub async fn list_disk_groups_from_group0(ctx: &OpContext, node_id: u64) -> Result> { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let mut groups: Vec<_> = ctx + .sysmd() + .list_disk_groups_on_node(rack_id, node_id) + .await? + .into_iter() + .map(|group| DiskGroupEntry { + id: group.dg_id, + rack_id, + node_id, + name: group.value.name, + }) + .collect(); + groups.sort_unstable_by_key(|group| group.id); + Ok(groups) +} + +/// Delete an empty, unowned disk group and remove node membership atomically. +/// +/// # Errors +/// Returns an authority error, missing record, or child/assignment conflict. +pub async fn remove_disk_group_from_group0(ctx: &OpContext, node_id: u64, dg_id: u64) -> Result<()> { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let node_path = NodeKey { rack_id, node_id }.to_path(); + let group_path = DiskGroupKey { + rack_id, + node_id, + disk_group_id: dg_id, + } + .to_path(); + let mut uncertain = false; + for attempt in 0..RETRIES { + let (rack_path, rack, revision) = rack_revision(ctx, rack_id).await?; + let (mut node, _) = required::(ctx, &node_path, "node", node_id.to_string()).await?; + let group = optional::(ctx, &group_path).await?; + if group.is_none() && uncertain && !node.disk_group_ids.contains(&dg_id) { + return Ok(()); + } + let group = group.ok_or_else(|| Error::NotFound { + kind: "disk_group".into(), + id: dg_id.to_string(), + })?; + if !group.disk_ids.is_empty() + || !ctx + .sysmd() + .list_disks_in_group(rack_id, node_id, dg_id) + .await? + .is_empty() + || ctx.sysmd().get_owner(rack_id, node_id, dg_id).await?.is_some() + || ctx.sysmd().get_bind(rack_id, node_id, dg_id).await?.is_some() + { + return Err(Error::Conflict { + kind: "disk_group with children or assignment".into(), + id: dg_id.to_string(), + }); + } + if !node.disk_group_ids.contains(&dg_id) { + return Err(Error::Conflict { + kind: "disk_group membership".into(), + id: dg_id.to_string(), + }); + } + node.disk_group_ids.retain(|id| *id != dg_id); + if fenced( + ctx, + &rack_path, + &rack, + revision, + vec![put(&node_path, &node)?, delete(&group_path)], + ) + .await? + { + return Ok(()); + } + uncertain = true; + pause(attempt).await; + } + Err(KvError::OutcomeUnknown.into()) +} + +/// Add a validated disk and update its group membership in one confirmed write. +/// +/// # Errors +/// Returns a validation, authority, missing group, or conflicting disk error. +pub async fn add_disk_to_group0( + ctx: &OpContext, + node_id: u64, + dg_id: u64, + input: &AddDiskInput, +) -> Result { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let (entry, disk_id, value) = validate_disk_input(input, dg_id, rack_id, node_id)?; + let group_path = DiskGroupKey { + rack_id, + node_id, + disk_group_id: dg_id, + } + .to_path(); + let disk_path = DiskKey { + rack_id, + node_id, + disk_group_id: dg_id, + disk_id, + } + .to_path(); + for attempt in 0..RETRIES { + let (rack_path, rack, revision) = rack_revision(ctx, rack_id).await?; + let (mut group, _) = + required::(ctx, &group_path, "disk_group", dg_id.to_string()).await?; + if let Some(actual) = optional::(ctx, &disk_path).await? { + if actual == value && group.disk_ids.contains(&disk_id) { + return Ok(entry); + } + return Err(Error::Conflict { + kind: "disk".into(), + id: input.disk_id.clone(), + }); + } + if group.disk_ids.contains(&disk_id) { + return Err(Error::Conflict { + kind: "disk membership".into(), + id: input.disk_id.clone(), + }); + } + group.disk_ids.push(disk_id); + group.disk_ids.sort_unstable_by_key(|id| (id.high, id.low)); + if fenced( + ctx, + &rack_path, + &rack, + revision, + vec![put(&group_path, &group)?, put(&disk_path, &value)?], + ) + .await? + { + return Ok(entry); + } + pause(attempt).await; + } + Err(KvError::OutcomeUnknown.into()) +} + +/// Return confirmed disks on one node and disk group. +/// +/// # Errors +/// Returns an authority error or a missing node. +pub async fn list_disks_from_group0(ctx: &OpContext, node_id: u64, dg_id: u64) -> Result> { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let mut disks = Vec::new(); + for (disk_id, value) in ctx.sysmd().list_disks_in_group(rack_id, node_id, dg_id).await? { + disks.push(disk_entry(rack_id, node_id, dg_id, disk_id, &value)); + } + disks.sort_unstable_by(|a, b| a.disk_id.cmp(&b.disk_id)); + Ok(disks) +} + +/// Remove a disk and its group membership in one confirmed write. +/// +/// # Errors +/// Returns an authority error, missing disk, or membership conflict. +pub async fn remove_disk_from_group0( + ctx: &OpContext, + node_id: u64, + dg_id: u64, + disk_id: &str, +) -> Result { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let id = DiskId::from_display_string(disk_id).map_err(|message| Error::Validation { + field: "disk_id".into(), + message, + })?; + let group_path = DiskGroupKey { + rack_id, + node_id, + disk_group_id: dg_id, + } + .to_path(); + let disk_path = DiskKey { + rack_id, + node_id, + disk_group_id: dg_id, + disk_id: id, + } + .to_path(); + let mut removed = None; + for attempt in 0..RETRIES { + let (rack_path, rack, revision) = rack_revision(ctx, rack_id).await?; + let (mut group, _) = + required::(ctx, &group_path, "disk_group", dg_id.to_string()).await?; + let current = optional::(ctx, &disk_path).await?; + if current.is_none() && !group.disk_ids.contains(&id) { + return removed.ok_or_else(|| Error::NotFound { + kind: "disk".into(), + id: disk_id.into(), + }); + } + let value = current.ok_or_else(|| Error::Conflict { + kind: "disk membership".into(), + id: disk_id.into(), + })?; + if !group.disk_ids.contains(&id) { + return Err(Error::Conflict { + kind: "disk membership".into(), + id: disk_id.into(), + }); + } + let entry = disk_entry(rack_id, node_id, dg_id, id, &value); + group.disk_ids.retain(|candidate| *candidate != id); + if fenced( + ctx, + &rack_path, + &rack, + revision, + vec![put(&group_path, &group)?, delete(&disk_path)], + ) + .await? + { + return Ok(entry); + } + removed = Some(entry); + pause(attempt).await; + } + Err(KvError::OutcomeUnknown.into()) +} + +fn disk_entry(rack_id: u64, node_id: u64, dg_id: u64, disk_id: DiskId, value: &DiskValue) -> DiskEntry { + DiskEntry { + disk_id: disk_id.to_display_string(), + rack_id, + node_id, + disk_group_id: dg_id, + disk_type: match crowdb_protocol::diskdb::rpc::DiskType::try_from(value.disk_type) { + Ok(kind) => format!("{kind:?}"), + Err(()) => value.disk_type.to_string(), + }, + capacity_bytes: value + .capacity_units + .saturating_mul(u64::from(value.unit_size_bytes)), + zone_size_bytes: value + .zone_size_units + .saturating_mul(u64::from(value.unit_size_bytes)), + unit_size_bytes: value.unit_size_bytes, + device_path: value.device_path.clone(), + } +} diff --git a/lib/crowdb-console-shared/src/ops/s3.rs b/lib/crowdb-console-shared/src/ops/s3.rs index 3763d64ba..424727dc8 100644 --- a/lib/crowdb-console-shared/src/ops/s3.rs +++ b/lib/crowdb-console-shared/src/ops/s3.rs @@ -12,15 +12,17 @@ use crowdb_protocol::port::namespace::{assign_process_ports, RuntimeNamespace}; use crowdb_protocol::ServicePort; use serde::Serialize; -use crate::config::{ConsoleConfig, LocalLaunchSpec, ServerEntry, ServiceType}; +use crate::config::{ConsoleConfig, LocalLaunchSpec, NodeEntry, RackEntry, ServerEntry, ServiceType}; use crate::error::{Error, Result}; use crate::lifecycle; use crate::ops::cluster::{self, KvDeployTunables, LocalChunkdbDeployConfig, LocalDiskdbDeployConfig}; use crate::ops::OpContext; -const CONFIG_FILE: &str = "console.toml"; +mod local_state; + const MARKER_FILE: &str = "s3-mini-cluster.json"; const INITIALIZING_FILE: &str = "s3-mini-cluster.initializing.json"; +const BOOTSTRAP_INTENT_FILE: &str = "bootstrap-intent.toml"; const MASTER_KEY: &str = "1111111111111111111111111111111111111111111111111111111111111111"; const NAMESPACE_ID: &str = "s3-mini-cluster"; const BODY_PREVIEW_LIMIT: usize = 64 * 1024; @@ -58,10 +60,10 @@ pub struct MiniClusterStatus { #[must_use] pub fn config_path(data_dir: &Path) -> PathBuf { - data_dir.join(CONFIG_FILE) + local_state::path(data_dir) } -/// Load a persisted mini-cluster record and console configuration. +/// Load a persisted mini-cluster record and local process state. /// /// # Errors /// Returns an error for an unrecognized directory or invalid persisted data. @@ -71,7 +73,8 @@ pub fn load(data_dir: &Path) -> Result<(ConsoleConfig, MiniClusterRecord)> { message: format!("{} is not a CROWDB S3 mini-cluster: {error}", data_dir.display()), })?; let record = serde_json::from_slice(&marker).map_err(|error| Error::Config(error.to_string()))?; - Ok((ConsoleConfig::load(&config_path(data_dir))?, record)) + let (config, _) = local_state::load(data_dir)?; + Ok((config, record)) } /// Create or restart a persistent local S3 mini-cluster. @@ -105,6 +108,9 @@ async fn start_with_profile( ) -> Result { archive_incomplete_attempt(data_dir)?; validate_location(data_dir)?; + std::fs::create_dir_all(data_dir)?; + let canonical_root = std::fs::canonicalize(data_dir)?; + let data_dir = canonical_root.as_path(); let marker_path = data_dir.join(MARKER_FILE); if marker_path.exists() { let (_, record) = load(data_dir)?; @@ -126,7 +132,6 @@ async fn start_with_profile( return restart(data_dir).await; } - std::fs::create_dir_all(data_dir)?; if storage_profile == StorageProfile::Persistent { RuntimeNamespace::persistent(data_dir, NAMESPACE_ID).map_err(namespace_error)?; } @@ -173,6 +178,9 @@ async fn start_with_profile( diskdb_client_rpc_workers: None, metrics_interval: None, }; + if let Some(status) = resume_if_interrupted(data_dir, &disk, &chunk, storage_profile).await? { + return Ok(status); + } let mut record = MiniClusterRecord { version: 1, endpoint: String::new(), @@ -187,7 +195,7 @@ async fn start_with_profile( Ok(endpoints) => endpoints, Err(error) => { stop_config_processes(&mut ctx.config_mut()); - let _ = ctx.config().save(&config_path(data_dir)); + let _ = local_state::save(data_dir, &ctx.config()); return Err(error); } }; @@ -200,10 +208,33 @@ async fn start_with_profile( Ok(status) } +async fn resume_if_interrupted( + data_dir: &Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, + storage_profile: StorageProfile, +) -> Result> { + if !data_dir.join(INITIALIZING_FILE).exists() || !local_state::path(data_dir).exists() { + return Ok(None); + } + let record: MiniClusterRecord = serde_json::from_slice(&std::fs::read(data_dir.join(INITIALIZING_FILE))?) + .map_err(|error| Error::Config(error.to_string()))?; + if record.version != 1 || record.storage_profile != storage_profile { + return Err(Error::Conflict { + kind: "S3 bootstrap profile".into(), + id: data_dir.display().to_string(), + }); + } + resume_incomplete(data_dir, disk, chunk, record).await.map(Some) +} + fn archive_incomplete_attempt(data_dir: &Path) -> Result<()> { if !data_dir.join(INITIALIZING_FILE).exists() || data_dir.join(MARKER_FILE).exists() { return Ok(()); } + if local_state::path(data_dir).exists() || data_dir.join(BOOTSTRAP_INTENT_FILE).exists() { + return Ok(()); + } let name = data_dir .file_name() .and_then(|value| value.to_str()) @@ -238,23 +269,69 @@ async fn initialize_new( no_fsync: (storage_profile == StorageProfile::Memory).then_some(true), ..KvDeployTunables::default() }; - match storage_profile { - StorageProfile::Persistent => { - cluster::local_deploy_combined_file_backed(ctx, data_dir, Some(&tunables), disk, chunk).await?; + let (_, nodes) = cluster::prepare_local_deploy(ctx, 3, Some(data_dir), Some(&tunables)).await?; + local_state::save(data_dir, &ctx.config())?; + cluster::init_with_intent(ctx, &nodes, &data_dir.join(BOOTSTRAP_INTENT_FILE)).await?; + initialize_after_kv(ctx, data_dir, disk, chunk, storage_profile).await +} + +async fn initialize_after_kv( + ctx: &OpContext, + data_dir: &Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, + storage_profile: StorageProfile, +) -> Result { + let storage_services = ctx + .config() + .servers + .iter() + .filter(|server| { + matches!( + server.service_type, + ServiceType::Diskdb | ServiceType::Diskio | ServiceType::Chunkdb + ) + }) + .count(); + if storage_services == 9 { + for group in &disk.data_groups { + if ctx.sysmd().get_group(0, *group).await?.is_none() { + return Err(Error::NotFound { + kind: "S3 data group".into(), + id: group.to_string(), + }); + } + } + cluster::restart_storage_services(ctx).await?; + } else if storage_services == 0 { + match storage_profile { + StorageProfile::Persistent => { + cluster::local_deploy_combined_file_backed_after_kv(ctx, data_dir, disk, chunk).await?; + } + StorageProfile::Memory => { + cluster::local_deploy_combined_after_kv(ctx, data_dir, disk, chunk, "mem").await?; + } } - StorageProfile::Memory => { - cluster::local_deploy_combined(ctx, data_dir, Some(&tunables), disk, chunk, "mem").await?; + } else { + clear_partial_storage_launches(ctx, data_dir)?; + match storage_profile { + StorageProfile::Persistent => { + cluster::local_deploy_combined_file_backed_after_kv(ctx, data_dir, disk, chunk).await?; + } + StorageProfile::Memory => { + cluster::local_deploy_combined_after_kv(ctx, data_dir, disk, chunk, "mem").await?; + } } } let seeds = management_seeds(&ctx.config()); - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; let chunk_kv = spawn_chunk_kv(data_dir, &seeds).await?; add_service(ctx, chunk_kv)?; - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; let access = spawn_access(data_dir, &seeds).await?; let endpoint = access.entry.url.clone(); add_service(ctx, access)?; - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; let (web_endpoint, web_pid) = spawn_web(data_dir).await?; Ok(StartedEndpoints { s3: endpoint, @@ -263,8 +340,91 @@ async fn initialize_new( }) } +fn clear_partial_storage_launches(ctx: &OpContext, data_dir: &Path) -> Result<()> { + let partial = ctx + .config() + .servers + .iter() + .filter(|server| { + matches!( + server.service_type, + ServiceType::Diskdb | ServiceType::Diskio | ServiceType::Chunkdb + ) + }) + .cloned() + .collect::>(); + for server in &partial { + if let Some(pid) = server.pid.filter(|pid| lifecycle::process_is_alive(*pid)) { + let launch = ctx + .config() + .local_launches + .get(&server.id) + .cloned() + .ok_or_else(|| Error::Config(format!("{} has no launch specification", server.id)))?; + let actual_cwd = std::fs::read_link(format!("/proc/{pid}/cwd"))?; + if actual_cwd != Path::new(&launch.workdir) { + return Err(Error::Conflict { + kind: "S3 process identity".into(), + id: server.id.clone(), + }); + } + lifecycle::stop_pid_with_timeout(pid, Duration::from_secs(5))?; + if lifecycle::process_is_alive(pid) { + return Err(Error::Conflict { + kind: "S3 process still running".into(), + id: server.id.clone(), + }); + } + } + } + let ids = partial.into_iter().map(|server| server.id).collect::>(); + let mut config = ctx.config_mut(); + config.servers.retain(|server| !ids.contains(&server.id)); + for id in ids { + config.local_launches.remove(&id); + } + local_state::save(data_dir, &config) +} + +async fn resume_incomplete( + data_dir: &Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, + mut record: MiniClusterRecord, +) -> Result { + let (mut config, seeds) = local_state::load(data_dir)?; + restore_launch_nodes(&mut config)?; + let group0 = config + .servers + .iter() + .find(|server| server.service_type == ServiceType::Kv) + .and_then(|server| server.rpc_url.as_deref()) + .ok_or_else(|| Error::Config("S3 bootstrap has no KV RPC seed".into()))? + .trim_start_matches("http://") + .to_owned(); + let ctx = OpContext::new(group0, seeds.clone(), config); + for node_id in 1..=3 { + let server_dir = data_dir + .join("rack1") + .join(format!("node{node_id}")) + .join(format!("kv-server-{node_id}")); + crate::ops::kv_server::restart(&ctx, node_id, Some(&server_dir), None, &seeds).await?; + } + local_state::save(data_dir, &ctx.config())?; + cluster::init_with_intent(&ctx, &[1, 2, 3], &data_dir.join(BOOTSTRAP_INTENT_FILE)).await?; + let endpoints = initialize_after_kv(&ctx, data_dir, disk, chunk, record.storage_profile).await?; + record.endpoint = endpoints.s3; + record.web_endpoint = endpoints.web; + record.web_pid = Some(endpoints.web_pid); + save_record(&data_dir.join(MARKER_FILE), &record)?; + std::fs::remove_file(data_dir.join(INITIALIZING_FILE))?; + let status = status_from(data_dir, false, &ctx.config(), &record); + Ok(status) +} + async fn restart(data_dir: &Path) -> Result { - let (config, mut record) = load(data_dir)?; + let (mut config, mut record) = load(data_dir)?; + restore_launch_nodes(&mut config)?; if let Some(pid) = record.web_pid.take() { let _ = lifecycle::stop_pid_with_timeout(pid, Duration::from_secs(5)); save_record(&data_dir.join(MARKER_FILE), &record)?; @@ -288,7 +448,7 @@ async fn restart(data_dir: &Path) -> Result { crate::ops::kv_server::restart(&ctx, node_id, Some(&server_dir), None, &seeds).await?; } cluster::restart_storage_services(&ctx).await?; - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; for kind in [ServiceType::ChunkKv, ServiceType::AccessServer] { let server = ctx .config() @@ -306,7 +466,7 @@ async fn restart(data_dir: &Path) -> Result { record.endpoint.clone_from(&spawned.entry.url); } add_service(&ctx, spawned)?; - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; continue; }; let mut launch = ctx @@ -329,9 +489,9 @@ async fn restart(data_dir: &Path) -> Result { { entry.pid = Some(pid); } - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; } - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; let (web_endpoint, web_pid) = spawn_web(data_dir).await?; record.web_endpoint = web_endpoint; record.web_pid = Some(web_pid); @@ -367,7 +527,7 @@ pub fn stop(data_dir: &Path) -> Result { let _ = lifecycle::stop_pid_with_timeout(pid, Duration::from_secs(5)); } stop_config_processes(&mut config); - config.save(&config_path(data_dir))?; + local_state::save(data_dir, &config)?; save_record(&data_dir.join(MARKER_FILE), &record)?; Ok(status_from(data_dir, false, &config, &record)) } @@ -392,7 +552,10 @@ fn validate_location(data_dir: &Path) -> Result<()> { return Ok(()); } let mut entries = std::fs::read_dir(data_dir)?; - if entries.next().transpose()?.is_none() || data_dir.join(MARKER_FILE).exists() { + if entries.next().transpose()?.is_none() + || data_dir.join(MARKER_FILE).exists() + || (data_dir.join(INITIALIZING_FILE).exists() && local_state::path(data_dir).exists()) + { return Ok(()); } Err(Error::Validation { @@ -421,6 +584,36 @@ fn management_seeds(config: &ConsoleConfig) -> Vec { .collect() } +fn restore_launch_nodes(config: &mut ConsoleConfig) -> Result<()> { + config.add_rack(RackEntry { + id: 1, + name: "rack-1".into(), + })?; + let node_ids: Vec<_> = config + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| { + server + .node_id + .ok_or_else(|| Error::Config("S3 KV launch has no node id".into())) + }) + .collect::>()?; + for id in node_ids { + config.add_node(NodeEntry { + id, + rack_id: 1, + host: "127.0.0.1".into(), + ssh_port: 22, + ssh_user: String::new(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: None, + })?; + } + Ok(()) +} + struct SpawnedService { entry: ServerEntry, launch: LocalLaunchSpec, @@ -508,33 +701,85 @@ async fn spawn_access(data_dir: &Path, seeds: &[String]) -> Result Result<(String, u32)> { + use crate::config::web::{WebMode, WebProcessConfig}; + let binary = find_binary("CROWDB_WEB_BIN", "crowdb-web")?; let port = assign_cluster_port(data_dir, ServicePort::Web, "web-1")?; - let workdir = data_dir.join("services/web-1"); + let root = std::fs::canonicalize(data_dir)?; + let workdir = root.join("services/web-1"); let log_dir = workdir.join("log"); std::fs::create_dir_all(&log_dir)?; + let ui_root = root.join("ui"); + std::fs::create_dir_all(&ui_root)?; let endpoint = format!("http://127.0.0.1:{port}"); + let (_, seeds) = local_state::load(data_dir)?; + let config = WebProcessConfig { + version: 1, + mode: WebMode::BareMetal, + bind: "127.0.0.1".into(), + port, + group0_management_seeds: seeds, + ui_root, + monitor_status: None, + log_dir, + log_max_file_mb: 30, + log_max_files: 5, + request_timeout_ms: Some(5_000), + }; + config.validate()?; + let config_path = root.join("s3-web.toml"); + std::fs::write( + &config_path, + toml::to_string_pretty(&config).map_err(|error| Error::Config(error.to_string()))?, + )?; + let mut env = BTreeMap::new(); + env.insert("CROWDB_ICEBERG_MANAGE_TOKEN".into(), web_token(&root)?); let launch = LocalLaunchSpec { program: binary.to_string_lossy().into_owned(), - args: vec![ - "--bind".into(), - "127.0.0.1".into(), - "--port".into(), - port.to_string(), - "--config".into(), - config_path(data_dir).to_string_lossy().into_owned(), - "--skip-startup-restore".into(), - "--log-dir".into(), - log_dir.to_string_lossy().into_owned(), - ], + args: vec!["--config".into(), config_path.to_string_lossy().into_owned()], workdir: workdir.to_string_lossy().into_owned(), - env: BTreeMap::new(), + env, readiness_url: Some(format!("{endpoint}/healthz")), }; let pid = spawn(&launch, "web-1").await?; Ok((endpoint, pid)) } +fn web_token(root: &Path) -> Result { + use std::io::{Read, Write}; + use std::os::unix::fs::{OpenOptionsExt, PermissionsExt}; + + let path = root.join("s3-web-manage.token"); + if let Ok(metadata) = std::fs::symlink_metadata(&path) { + if !metadata.file_type().is_file() || metadata.permissions().mode() & 0o077 != 0 { + return Err(Error::Config( + "S3 Web management token file is not private".into(), + )); + } + let token = std::fs::read_to_string(path)?; + if token.len() != 64 || !token.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(Error::Config("S3 Web management token is invalid".into())); + } + return Ok(token); + } + let mut bytes = [0u8; 32]; + std::fs::File::open("/dev/urandom")?.read_exact(&mut bytes)?; + let mut token = String::with_capacity(64); + for byte in bytes { + const HEX: &[u8; 16] = b"0123456789abcdef"; + token.push(char::from(HEX[usize::from(byte >> 4)])); + token.push(char::from(HEX[usize::from(byte & 0x0f)])); + } + let mut file = std::fs::OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(path)?; + file.write_all(token.as_bytes())?; + file.sync_all()?; + Ok(token) +} + fn assign_cluster_port(data_dir: &Path, service: ServicePort, identity: &str) -> Result { if data_dir.join("namespace.json").is_file() { return RuntimeNamespace::persistent(data_dir, NAMESPACE_ID) diff --git a/lib/crowdb-console-shared/src/ops/s3/local_state.rs b/lib/crowdb-console-shared/src/ops/s3/local_state.rs new file mode 100644 index 000000000..2bdc97c65 --- /dev/null +++ b/lib/crowdb-console-shared/src/ops/s3/local_state.rs @@ -0,0 +1,123 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Local S3 mini-cluster process inputs and runtime identity, without topology. + +use std::collections::{BTreeMap, HashSet}; +use std::fs::{self, OpenOptions}; +use std::io::Write; +use std::os::unix::fs::OpenOptionsExt; +use std::path::{Path, PathBuf}; +use std::sync::atomic::{AtomicU64, Ordering}; + +use serde::{Deserialize, Serialize}; + +use crate::config::{ConsoleConfig, LocalLaunchSpec, ServerEntry, ServiceType}; +use crate::error::{Error, Result}; + +const FILE: &str = "s3-local-state.toml"; +const VERSION: u32 = 1; +static NEXT_TEMP: AtomicU64 = AtomicU64::new(0); + +#[derive(Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct LocalState { + version: u32, + group0_seeds: Vec, + #[serde(rename = "service")] + services: Vec, + #[serde(default)] + local_launches: BTreeMap, +} + +pub(super) fn path(data_dir: &Path) -> PathBuf { + data_dir.join(FILE) +} + +pub(super) fn load(data_dir: &Path) -> Result<(ConsoleConfig, Vec)> { + let body = fs::read_to_string(path(data_dir))?; + let state: LocalState = toml::from_str(&body).map_err(|error| Error::Config(error.to_string()))?; + state.validate()?; + Ok(( + ConsoleConfig { + servers: state.services, + local_launches: state.local_launches, + ..ConsoleConfig::default() + }, + state.group0_seeds, + )) +} + +pub(super) fn save(data_dir: &Path, config: &ConsoleConfig) -> Result<()> { + let group0_seeds: Vec<_> = config + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| server.url.clone()) + .collect(); + let state = LocalState { + version: VERSION, + group0_seeds, + services: config.servers.clone(), + local_launches: config.local_launches.clone(), + }; + state.validate()?; + let body = toml::to_string_pretty(&state).map_err(|error| Error::Config(error.to_string()))?; + let destination = path(data_dir); + let temporary = destination.with_extension(format!( + "tmp.{}.{}", + std::process::id(), + NEXT_TEMP.fetch_add(1, Ordering::Relaxed) + )); + let result = (|| { + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&temporary)?; + file.write_all(body.as_bytes())?; + file.sync_all()?; + fs::rename(&temporary, &destination)?; + fs::File::open(data_dir)?.sync_all()?; + Ok(()) + })(); + if result.is_err() { + let _ = fs::remove_file(temporary); + } + result +} + +impl LocalState { + fn validate(&self) -> Result<()> { + let expected_seeds: Vec<_> = self + .services + .iter() + .filter(|service| service.service_type == ServiceType::Kv) + .map(|service| service.url.clone()) + .collect(); + if self.version != VERSION + || expected_seeds.is_empty() + || expected_seeds != self.group0_seeds + || self.group0_seeds.iter().any(String::is_empty) + { + return Err(Error::Config( + "S3 local state version or Group 0 seeds are invalid".into(), + )); + } + let mut identities = HashSet::new(); + if self + .services + .iter() + .any(|service| !identities.insert(&service.id)) + || self + .local_launches + .values() + .any(|launch| launch.env.contains_key("CROWDB_S3_MASTER_KEY")) + { + return Err(Error::Config( + "S3 local state has duplicate services or inline secrets".into(), + )); + } + Ok(()) + } +} diff --git a/lib/crowdb-console-shared/src/ssh.rs b/lib/crowdb-console-shared/src/ssh.rs index 4d87c2a9e..274709c38 100644 --- a/lib/crowdb-console-shared/src/ssh.rs +++ b/lib/crowdb-console-shared/src/ssh.rs @@ -423,6 +423,7 @@ mod tests { ssh_user: user.into(), ssh_key: key.map(Into::into), ssh_password: password.map(Into::into), + ssh_credential_ref: None, } } diff --git a/lib/crowdb-console-shared/tests/bootstrap_intent_test.rs b/lib/crowdb-console-shared/tests/bootstrap_intent_test.rs new file mode 100644 index 000000000..50247e4f3 --- /dev/null +++ b/lib/crowdb-console-shared/tests/bootstrap_intent_test.rs @@ -0,0 +1,184 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::os::unix::fs::{symlink, PermissionsExt}; + +use crowdb_console_shared::bootstrap_intent::BootstrapIntent; +use crowdb_console_shared::config::{ConsoleConfig, NodeEntry, RackEntry, ServerEntry, ServiceType}; +use crowdb_console_shared::error::Error; +use crowdb_console_shared::ops::{cluster as cluster_ops, OpContext}; +use crowdb_test_harness::cluster::KvCluster; +use crowdb_test_harness::test_dirs::tempdir_in_test_data; + +#[path = "common/bootstrap_authority.rs"] +mod bootstrap_authority; + +fn config() -> ConsoleConfig { + let mut config = ConsoleConfig::default(); + config.racks.push(RackEntry { + id: 1, + name: "rack-one".into(), + }); + config.nodes.push(NodeEntry { + id: 1, + rack_id: 1, + host: "127.0.0.1".into(), + ssh_port: 22, + ssh_user: "operator".into(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: Some("ops-key".into()), + }); + config.servers.push(ServerEntry { + id: "kv-1".into(), + url: "http://127.0.0.1:10000".into(), + node_id: Some(1), + rpc_url: Some("127.0.0.1:10100".into()), + rest_port: Some(10000), + rpc_port: Some(10100), + auto_start: true, + binary: Some("/private/bin/crowdb-kv-server".into()), + election_profile: None, + pid: Some(42), + service_type: ServiceType::Kv, + rpc_workers: None, + no_fsync: false, + }); + config +} + +#[test] +fn sealed_intent_is_private_immutable_and_restores_only_bootstrap_inputs() { + let dir = tempdir_in_test_data("bootstrap-intent"); + let path = dir.path().join("intent.toml"); + let intent = BootstrapIntent::capture(&config(), &[1]).unwrap(); + intent.seal(&path).unwrap(); + intent.seal(&path).unwrap(); + assert_eq!( + std::fs::metadata(&path).unwrap().permissions().mode() & 0o777, + 0o600 + ); + assert_eq!(BootstrapIntent::load(&path).unwrap(), intent); + let body = std::fs::read_to_string(&path).unwrap(); + assert!(!body.contains("/private/bin")); + assert!(!body.contains("pid")); + let restored = intent.to_config(); + assert_eq!(restored.racks[0].name, "rack-one"); + assert_eq!(restored.nodes[0].ssh_credential_ref.as_deref(), Some("ops-key")); + assert!(restored.servers[0].pid.is_none()); + assert!(restored.servers[0].binary.is_none()); + + let mut changed = config(); + changed.racks[0].name = "other".into(); + let error = BootstrapIntent::capture(&changed, &[1]) + .unwrap() + .seal(&path) + .unwrap_err(); + assert!(matches!(error, Error::Conflict { .. })); + assert_eq!(BootstrapIntent::load(&path).unwrap(), intent); +} + +#[test] +fn intent_rejects_inline_secrets_symlinks_and_legacy_fields() { + let mut config = config(); + config.nodes[0].ssh_key = Some("/private/id_ed25519".into()); + assert!(BootstrapIntent::capture(&config, &[1]).is_err()); + config.nodes[0].ssh_key = None; + assert!(BootstrapIntent::capture(&config, &[1, 1]).is_err()); + + let dir = tempdir_in_test_data("bootstrap-intent-invalid"); + let target = dir.path().join("target.toml"); + std::fs::write(&target, "version = 1\nlegacy = true\n").unwrap(); + let link = dir.path().join("intent.toml"); + symlink(&target, &link).unwrap(); + assert!(BootstrapIntent::load(&link).is_err()); + assert!(BootstrapIntent::capture(&config, &[1]) + .unwrap() + .seal(&link) + .is_err()); + assert!(BootstrapIntent::load(&target).is_err()); + + let strict = dir.path().join("strict.toml"); + BootstrapIntent::capture(&config, &[1]) + .unwrap() + .seal(&strict) + .unwrap(); + let body = std::fs::read_to_string(&strict).unwrap(); + let changed = body.replace("name = \"rack-one\"", "name = \"rack-one\"\nlegacy = true"); + assert_ne!(changed, body); + std::fs::write(&strict, changed).unwrap(); + assert!(BootstrapIntent::load(&strict).is_err()); +} + +#[tokio::test] +async fn interrupted_bootstrap_restores_identity_then_clears_verified_intent() { + let cluster = KvCluster::start().await; + let source = bootstrap_authority::context(&cluster).await; + let dir = tempdir_in_test_data("bootstrap-intent-resume"); + let path = dir.path().join("intent.toml"); + BootstrapIntent::capture(&source.config(), &[1]) + .unwrap() + .seal(&path) + .unwrap(); + + let resumed = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + cluster_ops::init_with_intent(&resumed, &[1], &path) + .await + .unwrap(); + assert!(!path.exists()); + assert_eq!( + resumed.sysmd().get_store(0).await.unwrap().unwrap().node_ids, + vec![1] + ); + assert_eq!(resumed.config().racks.len(), 1); + assert_eq!(resumed.config().nodes.len(), 1); +} + +#[tokio::test] +async fn committed_group_zero_is_verified_before_stale_intent_is_removed() { + let cluster = KvCluster::start().await; + let source = bootstrap_authority::context(&cluster).await; + let dir = tempdir_in_test_data("bootstrap-intent-committed"); + let path = dir.path().join("intent.toml"); + BootstrapIntent::capture(&source.config(), &[1]) + .unwrap() + .seal(&path) + .unwrap(); + cluster_ops::init(&source, &[1]).await.unwrap(); + assert!(path.exists()); + + let resumed = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + cluster_ops::init_with_intent(&resumed, &[1], &path) + .await + .unwrap(); + assert!(!path.exists()); + assert_eq!( + resumed.sysmd().list_replicas_in_group(0, 0).await.unwrap().len(), + 1 + ); +} + +#[tokio::test] +async fn changed_bootstrap_topology_is_rejected_before_group_zero_mutation() { + let cluster = KvCluster::start().await; + let ctx = bootstrap_authority::context(&cluster).await; + let dir = tempdir_in_test_data("bootstrap-intent-conflict"); + let path = dir.path().join("intent.toml"); + BootstrapIntent::capture(&ctx.config(), &[1]) + .unwrap() + .seal(&path) + .unwrap(); + ctx.config_mut().racks[0].name = "changed".into(); + let result = cluster_ops::init_with_intent(&ctx, &[1], &path).await; + assert!(matches!(result, Err(Error::Conflict { .. })), "{result:?}"); + assert!(ctx.sysmd().get_store(0).await.unwrap().is_none()); + assert!(path.exists()); +} diff --git a/lib/crowdb-console-shared/tests/common/bootstrap_node.rs b/lib/crowdb-console-shared/tests/common/bootstrap_node.rs index 1cf610881..82e406aac 100644 --- a/lib/crowdb-console-shared/tests/common/bootstrap_node.rs +++ b/lib/crowdb-console-shared/tests/common/bootstrap_node.rs @@ -72,7 +72,9 @@ pub fn context(first: &TestNode, second: &TestNode) -> OpContext { serde_json::from_value(json!({"id": (i + 1).to_string(), "node_id": i + 1, "url": url})).unwrap() }) .collect(); - let mut config = ConsoleConfig::default(); - config.servers = servers; + let config = ConsoleConfig { + servers, + ..Default::default() + }; OpContext::new("127.0.0.1:9".into(), Vec::new(), config) } diff --git a/lib/crowdb-console-shared/tests/kv_e2e_test.rs b/lib/crowdb-console-shared/tests/kv_e2e_test.rs index 25fa07fde..ba37a5796 100644 --- a/lib/crowdb-console-shared/tests/kv_e2e_test.rs +++ b/lib/crowdb-console-shared/tests/kv_e2e_test.rs @@ -36,6 +36,7 @@ async fn spawn_server() -> Option<(u32, String)> { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "s1".into(), diff --git a/lib/crowdb-console-shared/tests/lifecycle_e2e_test.rs b/lib/crowdb-console-shared/tests/lifecycle_e2e_test.rs index acbde233a..6ca496bbd 100644 --- a/lib/crowdb-console-shared/tests/lifecycle_e2e_test.rs +++ b/lib/crowdb-console-shared/tests/lifecycle_e2e_test.rs @@ -79,6 +79,7 @@ async fn deploy_local_and_observe_topology() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); diff --git a/lib/crowdb-console-shared/tests/mgmt_e2e_test.rs b/lib/crowdb-console-shared/tests/mgmt_e2e_test.rs index 0e1cbcea2..c488fc2f6 100644 --- a/lib/crowdb-console-shared/tests/mgmt_e2e_test.rs +++ b/lib/crowdb-console-shared/tests/mgmt_e2e_test.rs @@ -35,6 +35,7 @@ async fn spawn_server() -> Option<(u32, ServerClient)> { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let _ = RackEntry { id: 1, diff --git a/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs b/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs index cc539e249..f3802124b 100644 --- a/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs +++ b/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs @@ -3,7 +3,7 @@ use crowdb_console_shared::error::Error; use crowdb_console_shared::ops::cluster; -use crowdb_protocol::common::{HwStatus, RackValue}; +use crowdb_protocol::common::{HwStatus, NodeValue, RackValue}; use crowdb_protocol::TextKey; use crowdb_test_harness::cluster::KvCluster; #[path = "common/bootstrap_authority.rs"] @@ -17,6 +17,7 @@ async fn bootstrap_rejects_conflicting_hardware_without_overwriting_authority() let existing = RackValue { status: HwStatus::Up as i32, node_ids: vec![99], + ..Default::default() }; ctx.sysmd().add_rack(1, &existing).await.unwrap(); let result = cluster::init(&ctx, &[1]).await; @@ -26,6 +27,42 @@ async fn bootstrap_rejects_conflicting_hardware_without_overwriting_authority() assert!(ctx.config().stores.is_empty()); } +#[tokio::test] +async fn bootstrap_retry_preserves_known_rack_membership_and_node_runtime_fields() { + let cluster = KvCluster::start().await; + let ctx = context(&cluster).await; + ctx.sysmd() + .add_rack( + 1, + &RackValue { + status: HwStatus::Maintenance as i32, + node_ids: vec![1], + name: String::new(), + }, + ) + .await + .unwrap(); + ctx.sysmd() + .add_node( + 1, + 1, + &NodeValue { + status: HwStatus::Maintenance as i32, + management_host: "127.0.0.1".into(), + ssh_port: 22, + disk_group_ids: vec![42], + ..Default::default() + }, + ) + .await + .unwrap(); + cluster::init(&ctx, &[1]).await.unwrap(); + assert_eq!(ctx.sysmd().get_rack(1).await.unwrap().unwrap().node_ids, vec![1]); + let node = ctx.sysmd().get_node(1, 1).await.unwrap().unwrap(); + assert_eq!(node.status, HwStatus::Maintenance as i32); + assert_eq!(node.disk_group_ids, vec![42]); +} + #[tokio::test] async fn bootstrap_preflights_logical_conflicts_before_publishing_missing_hardware() { let cluster = KvCluster::start().await; @@ -40,6 +77,38 @@ async fn bootstrap_preflights_logical_conflicts_before_publishing_missing_hardwa assert!(ctx.sysmd().get_rack(1).await.unwrap().is_none()); } +#[tokio::test] +async fn bootstrap_rejects_a_different_rack_name_even_when_membership_matches() { + let cluster = KvCluster::start().await; + let ctx = context(&cluster).await; + let existing = RackValue { + status: HwStatus::Up as i32, + name: "another-rack".into(), + ..Default::default() + }; + ctx.sysmd().add_rack(1, &existing).await.unwrap(); + let result = cluster::init(&ctx, &[1]).await; + assert!(matches!(result, Err(Error::Conflict { .. })), "{result:?}"); + assert_eq!(ctx.sysmd().get_rack(1).await.unwrap(), Some(existing)); +} + +#[tokio::test] +async fn bootstrap_publishes_shared_hardware_labels_and_credential_reference() { + let cluster = KvCluster::start().await; + let ctx = context(&cluster).await; + { + let mut config = ctx.config_mut(); + config.racks[0].name = "rack-a".into(); + config.nodes[0].ssh_credential_ref = Some("ops-key".into()); + } + cluster::init(&ctx, &[1]).await.unwrap(); + assert_eq!(ctx.sysmd().get_rack(1).await.unwrap().unwrap().name, "rack-a"); + let node = ctx.sysmd().get_node(1, 1).await.unwrap().unwrap(); + assert_eq!(node.management_host, "127.0.0.1"); + assert_eq!(node.ssh_port, 22); + assert_eq!(node.ssh_credential_ref.as_deref(), Some("ops-key")); +} + #[tokio::test] async fn bootstrap_resumes_missing_records_and_preserves_committed_revisions() { let cluster = KvCluster::start().await; @@ -50,6 +119,7 @@ async fn bootstrap_resumes_missing_records_and_preserves_committed_revisions() { &RackValue { status: HwStatus::Up as i32, node_ids: Vec::new(), + ..Default::default() }, ) .await @@ -88,4 +158,9 @@ async fn bootstrap_resumes_missing_records_and_preserves_committed_revisions() { assert_eq!(before, after, "matching committed content must not be rewritten"); assert_eq!(ctx.sysmd().get_store(0).await.unwrap().unwrap().node_ids, vec![1]); assert_eq!(ctx.sysmd().list_replicas_in_group(0, 0).await.unwrap().len(), 1); + let node = ctx.sysmd().get_node(1, 1).await.unwrap().unwrap(); + assert_eq!(node.management_host, "127.0.0.1"); + assert_eq!(node.ssh_port, 22); + assert!(node.ssh_user.is_empty()); + assert!(node.ssh_credential_ref.is_none()); } diff --git a/lib/crowdb-console-shared/tests/ops_cluster_test.rs b/lib/crowdb-console-shared/tests/ops_cluster_test.rs index fd0238cc8..774d5538e 100644 --- a/lib/crowdb-console-shared/tests/ops_cluster_test.rs +++ b/lib/crowdb-console-shared/tests/ops_cluster_test.rs @@ -1,13 +1,15 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Tests for [`ops::cluster`] validation paths. The `init` / `reset` / -//! `clean` logic requires a running cluster and is covered by E2E tests -//! in Phase 4; here we verify the guard clauses. +//! Tests for [`ops::cluster`] validation and authority boundaries. use crowdb_console_shared::config::ConsoleConfig; use crowdb_console_shared::error::Error; use crowdb_console_shared::ops::{self, OpContext}; +use crowdb_test_harness::cluster::KvCluster; + +#[path = "common/bootstrap_authority.rs"] +mod bootstrap_authority; fn ctx() -> OpContext { OpContext::new("127.0.0.1:1".into(), vec![], ConsoleConfig::default()) @@ -33,3 +35,15 @@ async fn init_dedup_nodes() { Error::NodeUnreachable { .. } | Error::NotFound { .. } )); } + +#[tokio::test] +async fn clean_requires_confirmed_group_replicas() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + ops::cluster::init(&bootstrap, &[1]).await.unwrap(); + + // The bootstrap context still has a local server entry, but no local + // entry may justify wiping a group absent from Group 0. + let err = ops::cluster::clean(&bootstrap, 17, 2).await.unwrap_err(); + assert!(matches!(err, Error::NotFound { .. }), "{err:?}"); +} diff --git a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs new file mode 100644 index 000000000..f8d8d80be --- /dev/null +++ b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs @@ -0,0 +1,297 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_console_shared::config::{ConsoleConfig, NodeEntry}; +use crowdb_console_shared::error::Error; +use crowdb_console_shared::ops::{cluster as cluster_ops, hardware, OpContext}; +use crowdb_test_harness::cluster::KvCluster; +use std::sync::atomic::Ordering; + +#[path = "common/bootstrap_authority.rs"] +mod bootstrap_authority; +#[path = "common/rpc_response_proxy.rs"] +mod rpc_response_proxy; + +#[tokio::test] +async fn separate_consoles_confirm_matching_hardware_and_reject_conflicts() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + cluster_ops::init(&bootstrap, &[1]).await.unwrap(); + + let first = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + let second = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + + hardware::add_rack_to_group0(&first, 2, "rack-two").await.unwrap(); + hardware::add_rack_to_group0(&second, 2, "rack-two") + .await + .unwrap(); + let conflict = hardware::add_rack_to_group0(&second, 2, "other") + .await + .unwrap_err(); + assert!(matches!(conflict, Error::Conflict { .. }), "{conflict:?}"); + assert_eq!( + second.sysmd().get_rack(2).await.unwrap().unwrap().name, + "rack-two" + ); + assert_eq!( + hardware::list_racks_from_group0(&second).await.unwrap()[1].name, + "rack-two" + ); + assert!(first.config().racks.is_empty()); + assert!(second.config().racks.is_empty()); + + let node = NodeEntry { + id: 2, + rack_id: 2, + host: "10.0.0.2".into(), + ssh_port: 2222, + ssh_user: "operator".into(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: Some("ops-key".into()), + }; + hardware::add_node_to_group0(&first, node.clone()).await.unwrap(); + hardware::add_node_to_group0(&second, node.clone()).await.unwrap(); + second + .sysmd() + .set_node_status(2, 2, crowdb_protocol::common::HwStatus::Maintenance) + .await + .unwrap(); + hardware::add_node_to_group0(&first, node.clone()).await.unwrap(); + assert_eq!( + first.sysmd().get_node(2, 2).await.unwrap().unwrap().status, + crowdb_protocol::common::HwStatus::Maintenance as i32 + ); + assert_eq!( + second.sysmd().get_rack(2).await.unwrap().unwrap().node_ids, + vec![2] + ); + let changed = NodeEntry { + host: "10.0.0.3".into(), + ..node + }; + let conflict = hardware::add_node_to_group0(&second, changed).await.unwrap_err(); + assert!(matches!(conflict, Error::Conflict { .. }), "{conflict:?}"); + let actual = second.sysmd().get_node(2, 2).await.unwrap().unwrap(); + assert_eq!(actual.management_host, "10.0.0.2"); + assert_eq!(actual.ssh_credential_ref.as_deref(), Some("ops-key")); + let listed = hardware::list_nodes_from_group0(&second, Some(2)).await.unwrap(); + assert_eq!(listed.len(), 1); + assert_eq!(listed[0].host, "10.0.0.2"); + assert!(listed[0].ssh_key.is_none()); + assert!(listed[0].ssh_password.is_none()); + assert!(second.config().nodes.is_empty()); + + let occupied = hardware::remove_rack_from_group0(&second, 2).await.unwrap_err(); + assert!(matches!(occupied, Error::Conflict { .. }), "{occupied:?}"); + assert!(second.sysmd().get_rack(2).await.unwrap().is_some()); + + hardware::add_rack_to_group0(&first, 3, "empty").await.unwrap(); + hardware::remove_rack_from_group0(&second, 3).await.unwrap(); + assert!(first.sysmd().get_rack(3).await.unwrap().is_none()); + + // A repeated rack create preserves its confirmed child membership. + hardware::add_rack_to_group0(&first, 2, "rack-two").await.unwrap(); + assert_eq!( + first.sysmd().get_rack(2).await.unwrap().unwrap().node_ids, + vec![2] + ); + + let occupied = hardware::remove_node_from_group0(&second, 1).await.unwrap_err(); + assert!(matches!(occupied, Error::Conflict { .. }), "{occupied:?}"); + hardware::remove_node_from_group0(&second, 2).await.unwrap(); + assert!(first.sysmd().get_node(2, 2).await.unwrap().is_none()); + assert!(first + .sysmd() + .get_rack(2) + .await + .unwrap() + .unwrap() + .node_ids + .is_empty()); + hardware::remove_rack_from_group0(&first, 2).await.unwrap(); + assert!(second.sysmd().get_rack(2).await.unwrap().is_none()); +} + +#[tokio::test] +async fn concurrent_node_creates_preserve_both_rack_memberships() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + cluster_ops::init(&bootstrap, &[1]).await.unwrap(); + let first = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + let second = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + hardware::add_rack_to_group0(&first, 7, "rack-seven") + .await + .unwrap(); + let node = |id| NodeEntry { + id, + rack_id: 7, + host: format!("10.0.0.{id}"), + ssh_port: 22, + ssh_user: String::new(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: None, + }; + let (a, b) = tokio::join!( + hardware::add_node_to_group0(&first, node(8)), + hardware::add_node_to_group0(&second, node(9)) + ); + a.unwrap(); + b.unwrap(); + assert_eq!( + first.sysmd().get_rack(7).await.unwrap().unwrap().node_ids, + vec![8, 9] + ); + assert_eq!(first.sysmd().list_nodes_in_rack(7).await.unwrap().len(), 2); +} + +#[tokio::test] +async fn committed_rack_survives_a_lost_conditional_write_response() { + let cluster = KvCluster::start().await; + let proxy = rpc_response_proxy::TestResponseProxy::start(cluster.group0_leader_endpoint.clone()).await; + let ctx = OpContext::new( + proxy.endpoint.clone(), + vec![proxy.management_endpoint.clone()], + ConsoleConfig::default(), + ); + proxy.armed.store(true, Ordering::SeqCst); + hardware::add_rack_to_group0(&ctx, 9, "rack-nine").await.unwrap(); + assert_eq!(proxy.dropped.load(Ordering::SeqCst), 1); + assert_eq!(ctx.sysmd().get_rack(9).await.unwrap().unwrap().name, "rack-nine"); +} + +#[tokio::test] +async fn two_consoles_share_disk_groups_and_disks_without_local_topology() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + cluster_ops::init(&bootstrap, &[1]).await.unwrap(); + let first = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + let second = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + hardware::add_rack_to_group0(&first, 12, "storage").await.unwrap(); + hardware::add_node_to_group0( + &first, + NodeEntry { + id: 12, + rack_id: 12, + host: "127.0.0.1".into(), + ssh_port: 22, + ssh_user: String::new(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: None, + }, + ) + .await + .unwrap(); + hardware::add_disk_group_to_group0(&first, 12, 5, "hot") + .await + .unwrap(); + hardware::add_disk_group_to_group0(&second, 12, 5, "hot") + .await + .unwrap(); + let conflict = hardware::add_disk_group_to_group0(&second, 12, 5, "cold") + .await + .unwrap_err(); + assert!(matches!(conflict, Error::Conflict { .. })); + assert_eq!( + hardware::list_disk_groups_from_group0(&second, 12).await.unwrap()[0].name, + "hot" + ); + let disk = hardware::AddDiskInput { + disk_id: "0000000000000000-000000000000000c".into(), + disk_type: "Ssd".into(), + capacity_bytes: 4096, + zone_size_bytes: 4096, + unit_size_bytes: 4096, + device_path: "/dev/test".into(), + }; + hardware::add_disk_to_group0(&first, 12, 5, &disk).await.unwrap(); + hardware::add_disk_to_group0(&second, 12, 5, &disk).await.unwrap(); + assert_eq!( + hardware::list_disks_from_group0(&second, 12, 5) + .await + .unwrap() + .len(), + 1 + ); + assert!(matches!( + hardware::remove_disk_group_from_group0(&second, 12, 5).await, + Err(Error::Conflict { .. }) + )); + assert!(matches!( + hardware::remove_node_from_group0(&second, 12).await, + Err(Error::Conflict { .. }) + )); + hardware::remove_disk_from_group0(&second, 12, 5, &disk.disk_id) + .await + .unwrap(); + hardware::remove_disk_group_from_group0(&first, 12, 5) + .await + .unwrap(); + hardware::remove_node_from_group0(&second, 12).await.unwrap(); + assert!(first.config().disk_groups.is_empty()); + assert!(second.config().disks.is_empty()); +} + +#[tokio::test] +async fn committed_disk_group_survives_a_lost_response() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + cluster_ops::init(&bootstrap, &[1]).await.unwrap(); + let proxy = rpc_response_proxy::TestResponseProxy::start(cluster.group0_leader_endpoint.clone()).await; + let ctx = OpContext::new( + proxy.endpoint.clone(), + vec![proxy.management_endpoint.clone()], + ConsoleConfig::default(), + ); + hardware::add_rack_to_group0(&ctx, 13, "storage").await.unwrap(); + hardware::add_node_to_group0( + &ctx, + NodeEntry { + id: 13, + rack_id: 13, + host: "127.0.0.1".into(), + ssh_port: 22, + ssh_user: String::new(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: None, + }, + ) + .await + .unwrap(); + proxy.armed.store(true, Ordering::SeqCst); + hardware::add_disk_group_to_group0(&ctx, 13, 1, "recovered") + .await + .unwrap(); + assert_eq!(proxy.dropped.load(Ordering::SeqCst), 1); + assert_eq!( + hardware::list_disk_groups_from_group0(&ctx, 13).await.unwrap()[0].name, + "recovered" + ); +} diff --git a/lib/crowdb-console-shared/tests/ops_hardware_test.rs b/lib/crowdb-console-shared/tests/ops_hardware_test.rs index a6cb91ffe..159a45ddf 100644 --- a/lib/crowdb-console-shared/tests/ops_hardware_test.rs +++ b/lib/crowdb-console-shared/tests/ops_hardware_test.rs @@ -53,6 +53,7 @@ async fn remove_rack_with_nodes_conflict() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }, ) .await @@ -75,6 +76,7 @@ async fn add_node_and_list() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }, ) .await @@ -101,6 +103,7 @@ async fn add_node_unknown_rack_validation() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }, ) .await @@ -129,6 +132,7 @@ async fn remove_node_with_server_conflict() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }, ) .await diff --git a/lib/crowdb-console-shared/tests/ops_kv_server_test.rs b/lib/crowdb-console-shared/tests/ops_kv_server_test.rs index de065a40c..a8be3111f 100644 --- a/lib/crowdb-console-shared/tests/ops_kv_server_test.rs +++ b/lib/crowdb-console-shared/tests/ops_kv_server_test.rs @@ -25,6 +25,7 @@ fn ctx_with_node() -> OpContext { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); OpContext::new_for_test("127.0.0.1:59999".into(), vec![], cfg) diff --git a/lib/crowdb-console-shared/tests/s3_mini_cluster_test.rs b/lib/crowdb-console-shared/tests/s3_mini_cluster_test.rs index 9a79d5121..a695d99e3 100644 --- a/lib/crowdb-console-shared/tests/s3_mini_cluster_test.rs +++ b/lib/crowdb-console-shared/tests/s3_mini_cluster_test.rs @@ -26,6 +26,28 @@ fn incomplete_marker_fails_closed() { assert!(error.to_string().contains("config error")); } +#[test] +fn local_launch_state_rejects_topology_and_legacy_console_file() { + let dir = TestDir::new("s3-mini-local-only").expect("create test directory"); + std::fs::write( + dir.path().join("s3-mini-cluster.json"), + r#"{"version":1,"endpoint":"http://127.0.0.1:16000","tenant":"local"}"#, + ) + .unwrap(); + std::fs::write(dir.path().join("console.toml"), "[[rack]]\nid = 1\n").unwrap(); + assert!( + s3::status(dir.path()).is_err(), + "legacy topology must not be loaded" + ); + std::fs::write( + dir.path().join("s3-local-state.toml"), + "version = 1\ngroup0_seeds = ['http://127.0.0.1:10000']\n[[rack]]\nid = 1\n", + ) + .unwrap(); + let error = s3::status(dir.path()).expect_err("local state cannot contain topology"); + assert!(error.to_string().contains("unknown field"), "{error}"); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] #[ignore = "starts the complete local storage and S3 process stack"] async fn persistent_cluster_survives_stop_restart_and_range_read() { @@ -36,13 +58,13 @@ async fn persistent_cluster_survives_stop_restart_and_range_read() { .await .expect("web health request"); assert!(health.status().is_success()); - let servers = reqwest::get(format!("{}/api/servers", started.web_endpoint)) + let preview = reqwest::get(format!("{}/api/preview", started.web_endpoint)) .await - .expect("web server-list request") + .expect("web preview request") .text() .await - .expect("web server-list body"); - assert!(servers.contains("access-server-1")); + .expect("web preview body"); + assert!(preview.contains("group0")); let client = s3::S3HttpClient::from_data_dir(dir.path()).expect("S3 client"); client .request(Method::PUT, Some("durable-bucket"), None, &[], None, None) @@ -60,9 +82,13 @@ async fn persistent_cluster_survives_stop_restart_and_range_read() { .await .expect("put object"); let marker = std::fs::read_to_string(dir.path().join("s3-mini-cluster.json")).expect("marker"); - let config = std::fs::read_to_string(dir.path().join("console.toml")).expect("config"); + let config = std::fs::read_to_string(dir.path().join("s3-local-state.toml")).expect("local state"); assert!(!marker.contains("1111111111111111")); assert!(!config.contains("1111111111111111")); + assert!(!config.contains("[[rack]]")); + assert!(!config.contains("[[node]]")); + assert!(!config.contains("[[store]]")); + assert!(!dir.path().join("console.toml").exists()); let stopped = s3::stop(dir.path()).expect("stop cluster"); assert_eq!(stopped.running_services, 0); diff --git a/lib/crowdb-diskio-client/tests/disk_io_group0_sync_test.rs b/lib/crowdb-diskio-client/tests/disk_io_group0_sync_test.rs index f7c7a348f..4a10d7c88 100644 --- a/lib/crowdb-diskio-client/tests/disk_io_group0_sync_test.rs +++ b/lib/crowdb-diskio-client/tests/disk_io_group0_sync_test.rs @@ -149,6 +149,7 @@ async fn disk_io_e2e_group0_sync() { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: all_disk_ids, + name: String::new(), }, ) .await diff --git a/lib/crowdb-kv-client/src/hardware/hierarchy.rs b/lib/crowdb-kv-client/src/hardware/hierarchy.rs index 74d383c95..0178ffdb8 100644 --- a/lib/crowdb-kv-client/src/hardware/hierarchy.rs +++ b/lib/crowdb-kv-client/src/hardware/hierarchy.rs @@ -18,8 +18,6 @@ use std::sync::Arc; use std::time::{SystemTime, UNIX_EPOCH}; -use tracing::warn; - use crowdb_protocol::common::{HwStatus, NodeValue, RackValue}; use crowdb_protocol::common_type::{DiskGroupId, NodeId, RackId}; use crowdb_protocol::diskdb::rpc::{DiskGroupValue, DiskValue}; @@ -951,22 +949,14 @@ impl HardwareClient { // Remove child disks. let disks = self.list_disks_in_group(rack_id, node_id, dg_id).await?; for (disk_id, _) in &disks { - if let Err(e) = self.remove_disk(rack_id, node_id, dg_id, disk_id).await { - warn!(error = %e, dg_id, "cascade: remove_disk failed; continuing"); - } + self.remove_disk(rack_id, node_id, dg_id, disk_id).await?; } // Remove owner map entry. - if let Err(e) = self.remove_owner(rack_id, node_id, dg_id).await { - warn!(error = %e, dg_id, "cascade: remove_owner failed; continuing"); - } + self.remove_owner(rack_id, node_id, dg_id).await?; // Remove bind map entry. - if let Err(e) = self.remove_bind(rack_id, node_id, dg_id).await { - warn!(error = %e, dg_id, "cascade: remove_bind failed; continuing"); - } + self.remove_bind(rack_id, node_id, dg_id).await?; // Remove usage summary. - if let Err(e) = self.remove_disk_group_usage(dg_id).await { - warn!(error = %e, dg_id, "cascade: remove_disk_group_usage failed; continuing"); - } + self.remove_disk_group_usage(dg_id).await?; // Remove the disk-group record itself. self.remove_disk_group(rack_id, node_id, dg_id).await } @@ -976,9 +966,7 @@ impl HardwareClient { pub async fn remove_node_cascade(&self, rack_id: RackId, node_id: NodeId) -> Result<()> { let dgs = self.list_disk_groups_on_node(rack_id, node_id).await?; for dg in &dgs { - if let Err(e) = self.remove_disk_group_cascade(rack_id, node_id, dg.dg_id).await { - warn!(error = %e, node_id, dg_id = dg.dg_id, "cascade: remove_disk_group_cascade failed; continuing"); - } + self.remove_disk_group_cascade(rack_id, node_id, dg.dg_id).await?; } self.remove_node(rack_id, node_id).await } @@ -988,9 +976,7 @@ impl HardwareClient { pub async fn remove_rack_cascade(&self, rack_id: RackId) -> Result<()> { let nodes = self.list_nodes_in_rack(rack_id).await?; for (node_id, _) in &nodes { - if let Err(e) = self.remove_node_cascade(rack_id, *node_id).await { - warn!(error = %e, rack_id, node_id, "cascade: remove_node_cascade failed; continuing"); - } + self.remove_node_cascade(rack_id, *node_id).await?; } self.remove_rack(rack_id).await } diff --git a/lib/crowdb-protocol/src/types/common.rs b/lib/crowdb-protocol/src/types/common.rs index 2f7c941f2..b957450d5 100644 --- a/lib/crowdb-protocol/src/types/common.rs +++ b/lib/crowdb-protocol/src/types/common.rs @@ -177,6 +177,8 @@ pub struct ErrorInfo { pub struct RackValue { pub status: i32, pub node_ids: Vec, + #[serde(default, skip_serializing_if = "String::is_empty")] + pub name: String, } #[derive(Clone, Debug, PartialEq, Default, Serialize, Deserialize)] @@ -186,6 +188,18 @@ pub struct NodeValue { pub disk_group_ids: Vec, pub status_changed_at_ms: u64, pub temp_failure_since_ms: Option, + #[serde(default, skip_serializing_if = "String::is_empty")] + pub management_host: String, + #[serde(default, skip_serializing_if = "is_zero_u16")] + pub ssh_port: u16, + #[serde(default, skip_serializing_if = "String::is_empty")] + pub ssh_user: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub ssh_credential_ref: Option, +} + +fn is_zero_u16(value: &u16) -> bool { + *value == 0 } #[derive(Clone, Debug, PartialEq, Default, Serialize, Deserialize)] diff --git a/lib/crowdb-protocol/src/types/diskdb.rs b/lib/crowdb-protocol/src/types/diskdb.rs index fa8aa3c57..6f8d4844d 100644 --- a/lib/crowdb-protocol/src/types/diskdb.rs +++ b/lib/crowdb-protocol/src/types/diskdb.rs @@ -150,6 +150,8 @@ pub struct DiskValue { pub struct DiskGroupValue { pub status: i32, pub disk_ids: Vec, + #[serde(default, skip_serializing_if = "String::is_empty")] + pub name: String, } #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Default, Serialize, Deserialize)] diff --git a/lib/crowdb-protocol/tests/hardware_identity_test.rs b/lib/crowdb-protocol/tests/hardware_identity_test.rs new file mode 100644 index 000000000..723183a74 --- /dev/null +++ b/lib/crowdb-protocol/tests/hardware_identity_test.rs @@ -0,0 +1,27 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_protocol::common::{NodeValue, RackValue}; + +#[test] +fn hardware_values_preserve_shared_identity_and_read_older_records() { + let rack: RackValue = serde_json::from_str(r#"{"status":1,"node_ids":[7]}"#).unwrap(); + assert!(rack.name.is_empty()); + let node: NodeValue = serde_json::from_str( + r#"{"status":1,"last_used_dg_id":0,"disk_group_ids":[],"status_changed_at_ms":0,"temp_failure_since_ms":null}"#, + ) + .unwrap(); + assert!(node.management_host.is_empty()); + assert_eq!(node.ssh_port, 0); + assert!(node.ssh_credential_ref.is_none()); + + let current = NodeValue { + management_host: "node.example".into(), + ssh_port: 2222, + ssh_user: "operator".into(), + ssh_credential_ref: Some("node-7".into()), + ..node + }; + let round_trip: NodeValue = serde_json::from_slice(&serde_json::to_vec(¤t).unwrap()).unwrap(); + assert_eq!(round_trip, current); +} diff --git a/lib/crowdb-test-harness/src/hardware.rs b/lib/crowdb-test-harness/src/hardware.rs index 1f785f48b..cb6355264 100644 --- a/lib/crowdb-test-harness/src/hardware.rs +++ b/lib/crowdb-test-harness/src/hardware.rs @@ -33,6 +33,7 @@ pub async fn seed_hardware(hw: &HardwareClient, disk_ids: &[DiskId]) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![NODE_ID], + ..Default::default() }, ) .await @@ -47,6 +48,7 @@ pub async fn seed_hardware(hw: &HardwareClient, disk_ids: &[DiskId]) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await @@ -59,6 +61,7 @@ pub async fn seed_hardware(hw: &HardwareClient, disk_ids: &[DiskId]) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.to_vec(), + name: String::new(), }, ) .await diff --git a/pixi.toml b/pixi.toml index c6b6194d5..826d9fcf1 100644 --- a/pixi.toml +++ b/pixi.toml @@ -158,8 +158,8 @@ clean-env = "bash tools/runtime/clean-runtime.sh env" # Optional full-workspace compile check. CI test groups use their component # tasks to avoid building unrelated test targets on each runner. build-tests = "cargo test --workspace --all-targets --no-run" -# Verify every workspace package with tests is assigned to a CI test task. -test-task-coverage = "python tools/ci-checks/check-test-task-coverage.py" +# Verify every workspace package is assigned to a test task reachable from CI. +check-ci-test-tasks = "python tools/ci-checks/check-ci-test-tasks.py" # ── Lint job ── # (uses rs-fmt / rs-lint / tree-lint defined above, no separate test-* tasks) @@ -184,6 +184,7 @@ test-chunk-kv = { cmd = "cargo test -p crowdb-chunk-kv --tests" } test-chunk-stream = { cmd = "cargo test -p crowdb-chunk-stream --tests" } test-chunk-kv-client = { cmd = "cargo test -p crowdb-chunk-kv-client --tests" } test-chunk-kv-server = { cmd = "cargo test -p crowdb-chunk-kv-server --tests" } +test-access-multipart = { cmd = "cargo test -p crowdb-access-multipart --tests" } test-unit = { cmd = "bash tools/pixi-tasks/test-unit.sh" } # ── ServerTests job: spawns crowdb-kv-server / crowdb-diskdb / crowdb-diskio ── diff --git a/tools/README.md b/tools/README.md index 600abd3b2..a7c3e4734 100644 --- a/tools/README.md +++ b/tools/README.md @@ -27,7 +27,7 @@ Run commands through `pixi run`. Read only the directory relevant to the task. Common entry points: ```sh -pixi run test-task-coverage +pixi run check-ci-test-tasks pixi run check-version pixi run clean-env pixi run bash tools/test-metrics/measure.sh test-access-iceberg test-monitor @@ -35,7 +35,7 @@ pixi run bash tools/test-metrics/measure.sh test-access-iceberg test-monitor For test ownership and timings, read [`doc/working/test.md`](../doc/working/test.md). CI calls Pixi group tasks; -update the group script when adding a component, then run `test-task-coverage`. +update the group script when adding a component, then run `check-ci-test-tasks`. Feature-gated or ignored client tests need explicit task selection. Container packaging and its acceptance scripts live under diff --git a/tools/ci-checks/check-test-task-coverage.py b/tools/ci-checks/check-ci-test-tasks.py similarity index 74% rename from tools/ci-checks/check-test-task-coverage.py rename to tools/ci-checks/check-ci-test-tasks.py index 190fd6ac6..14b931501 100644 --- a/tools/ci-checks/check-test-task-coverage.py +++ b/tools/ci-checks/check-ci-test-tasks.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Check that every tested Rust workspace package has a CI task assignment.""" +"""Check workspace package assignments and CI test-task reachability.""" import json import subprocess @@ -26,6 +26,7 @@ "test-chunkdb": {"crowdb-chunkdb"}, "test-chunk-client": {"crowdb-chunk-client"}, "test-diskio-client": {"crowdb-diskio-client"}, + "test-access-multipart": {"crowdb-access-multipart"}, "test-access-s3": {"crowdb-access-s3"}, "test-access-iceberg": {"crowdb-access-iceberg"}, "test-access-server": {"crowdb-access-server"}, @@ -62,10 +63,10 @@ def main() -> int: missing = sorted(packages - set(assignments) - set(SUPPORT_PACKAGES)) unknown_support = sorted(set(SUPPORT_PACKAGES) - packages) if missing: - print("Rust workspace packages missing from CI test-task coverage:") + print("Rust workspace packages missing from CI test tasks:") for package in missing: print(f" {package}") - print("Add the package to TASK_PACKAGES in tools/ci-checks/check-test-task-coverage.py") + print("Add the package to TASK_PACKAGES in tools/ci-checks/check-ci-test-tasks.py") return 1 if unknown_support: print("Support-package allowlist contains packages not in the workspace:") @@ -84,7 +85,7 @@ def main() -> int: } missing_tasks = [task for task in TASK_PACKAGES if task not in pixi_tasks] if missing_tasks: - print("Coverage map references missing Pixi tasks:") + print("CI test-task map references missing Pixi tasks:") for task in missing_tasks: print(f" {task}") return 1 @@ -95,32 +96,50 @@ def main() -> int: if f"-p {package}" not in pixi_tasks[task] ] if missing_commands: - print("Coverage map packages are not targeted by their Pixi tasks:") + print("CI test-task map packages are not targeted by their Pixi tasks:") for task, package in missing_commands: print(f" {task}: {package}") return 1 - reachable = graph.reachable((root / ".github/workflows/ci.yml").read_text()) + workflows = root / ".github/workflows" + reachable = graph.reachable((workflows / "ci.yml").read_text()) required = {("default", task) for task in TASK_PACKAGES} required.update({ - ("default", "test-single-node-container"), ("default", "test-console-ui"), ("s3-e2e", "test-boto3-e2e"), ("iceberg-e2e", "test-pyiceberg-e2e"), ("iceberg-e2e", "test-iceberg-native"), ("iceberg-e2e", "test-java-iceberg-e2e"), ("iceberg-e2e", "test-java-iceberg-fileio-e2e"), - ("iceberg-e2e", "test-rust-iceberg-e2e"), ("iceberg-e2e", "test-iceberg-rck"), }) unreachable = sorted(required - reachable) if unreachable: - print("Test tasks not reachable from CI:") + print("Test tasks not reachable from regular CI:") for environment, task in unreachable: print(f" {environment}: {task}") return 1 - print(f"Test-task coverage verified for {len(packages)} workspace packages") + manual_tasks = { + "docker-preview.yml": {("default", "test-single-node-container")}, + "iceberg-rust-sdk.yml": {("iceberg-e2e", "test-rust-iceberg-e2e")}, + } + for workflow, tasks in manual_tasks.items(): + manual_reachable = graph.reachable((workflows / workflow).read_text()) + missing = sorted(tasks - manual_reachable) + if missing: + print(f"Manual workflow {workflow} does not reach its test tasks:") + for environment, task in missing: + print(f" {environment}: {task}") + return 1 + unexpected = sorted(tasks & reachable) + if unexpected: + print("Manual-only test tasks also reachable from regular CI:") + for environment, task in unexpected: + print(f" {environment}: {task}") + return 1 + + print(f"CI test tasks verified for {len(packages)} workspace packages") for package in sorted(assignments): print(f" {package}: {assignments[package]}") for package, reason in sorted(SUPPORT_PACKAGES.items()): diff --git a/tools/ci-checks/check-container-symbols.py b/tools/ci-checks/check-container-symbols.py new file mode 100644 index 000000000..c9e28ecf4 --- /dev/null +++ b/tools/ci-checks/check-container-symbols.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +# Copyright 2026-present Gian +# Licensed under the Apache License, Version 2.0. +"""Verify the symbol archive matches the staged release runtime exactly.""" + +import re +import struct +import subprocess +import tempfile +import zlib +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +RUNTIME = ROOT / "target/container-runtime" +SYMBOLS = ROOT / "target/container-symbols" + + +def sections(path: Path) -> str: + return subprocess.run( + ["readelf", "-W", "-S", str(path)], check=True, text=True, + capture_output=True, + ).stdout + + +def check_debuglink(binary: Path, symbol: Path) -> None: + with tempfile.TemporaryDirectory() as temp_dir: + section = Path(temp_dir) / "debuglink" + subprocess.run( + ["objcopy", f"--dump-section=.gnu_debuglink={section}", str(binary)], + check=True, + ) + data = section.read_bytes() + name, _, _ = data.partition(b"\0") + offset = (len(name) + 4) & ~3 + if name.decode() != symbol.name or len(data) < offset + 4: + raise ValueError(f"Wrong debuglink name or size: {binary}") + expected_crc = struct.unpack_from(" None: + for metadata in ("VERSION", "SOURCE_REVISION"): + if (RUNTIME / metadata).read_bytes() != (SYMBOLS / metadata).read_bytes(): + raise ValueError(f"Runtime and symbol {metadata} differ") + subprocess.run( + ["sha256sum", "--check", str(SYMBOLS / "RUNTIME_SHA256SUMS")], + cwd=RUNTIME, check=True, + ) + binaries = sorted((RUNTIME / "bin").iterdir()) + libraries = sorted((RUNTIME / "lib").glob("libcrowdb*.so")) + for binary in binaries + libraries: + symbol = SYMBOLS / binary.parent.name / f"{binary.name}.debug" + if not symbol.is_file(): + raise ValueError(f"Missing debug symbols: {symbol}") + if not re.search(r"\s\.debug_line\s", sections(symbol)): + raise ValueError(f"Missing source lines: {symbol}") + if re.search(r"\s\.debug_line\s", sections(binary)): + raise ValueError(f"Runtime still has source lines: {binary}") + check_debuglink(binary, symbol) + print(f"Verified symbols for {binary.relative_to(RUNTIME)}") + + +if __name__ == "__main__": + main() diff --git a/tools/pixi-tasks/test-iceberg-rck.sh b/tools/pixi-tasks/test-iceberg-rck.sh index eb94ed901..7ad9a7a7e 100644 --- a/tools/pixi-tasks/test-iceberg-rck.sh +++ b/tools/pixi-tasks/test-iceberg-rck.sh @@ -6,15 +6,20 @@ cd "${PIXI_PROJECT_ROOT:?}" source tools/pixi-tasks/prepare-iceberg.sh revision=6976e020b894f6a6777704df2b8c4458cb291ae9 -export CROWDB_ICEBERG_RCK_ROOT="${CROWDB_ICEBERG_RCK_ROOT:-$PIXI_PROJECT_ROOT/target/iceberg-rck}" +export CROWDB_ICEBERG_RCK_ROOT="${CROWDB_ICEBERG_RCK_ROOT:-${RUNNER_TEMP:-$PIXI_PROJECT_ROOT/target}/iceberg-rck}" if [[ ! -e "$CROWDB_ICEBERG_RCK_ROOT" ]]; then mkdir -p "$CROWDB_ICEBERG_RCK_ROOT" git -C "$CROWDB_ICEBERG_RCK_ROOT" init git -C "$CROWDB_ICEBERG_RCK_ROOT" fetch --depth 1 https://github.com/apache/iceberg.git "$revision" git -C "$CROWDB_ICEBERG_RCK_ROOT" checkout --detach FETCH_HEAD fi -[[ "$(git -C "$CROWDB_ICEBERG_RCK_ROOT" rev-parse HEAD)" == "$revision" ]] || { - echo 'Iceberg RCK requires the pinned source revision; existing checkout was preserved' >&2 +[[ -d "$CROWDB_ICEBERG_RCK_ROOT/.git" ]] || { + echo "Iceberg RCK source directory is not a checkout: $CROWDB_ICEBERG_RCK_ROOT" >&2 + exit 1 +} +actual=$(git -C "$CROWDB_ICEBERG_RCK_ROOT" rev-parse HEAD) +[[ "$actual" == "$revision" ]] || { + echo "Iceberg RCK requires $revision; found $actual at $CROWDB_ICEBERG_RCK_ROOT (checkout preserved)" >&2 exit 1 } pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e \ diff --git a/tools/pixi-tasks/test-iceberg-sdk.sh b/tools/pixi-tasks/test-iceberg-sdk.sh index c245215f0..ae9e8efca 100644 --- a/tools/pixi-tasks/test-iceberg-sdk.sh +++ b/tools/pixi-tasks/test-iceberg-sdk.sh @@ -5,5 +5,4 @@ set -euo pipefail cd "${PIXI_PROJECT_ROOT:?}" pixi run -e iceberg-e2e test-java-iceberg-e2e -pixi run -e iceberg-e2e test-rust-iceberg-e2e pixi run -e iceberg-e2e test-iceberg-rck diff --git a/tools/pixi-tasks/test-pyiceberg-e2e.sh b/tools/pixi-tasks/test-pyiceberg-e2e.sh index 20ad924a3..9657e85ba 100644 --- a/tools/pixi-tasks/test-pyiceberg-e2e.sh +++ b/tools/pixi-tasks/test-pyiceberg-e2e.sh @@ -4,14 +4,10 @@ set -euo pipefail cd "${PIXI_PROJECT_ROOT:?}" -pixi run -e default -- cmake -S app/crowdb-diskio -B app/crowdb-diskio/build -DCMAKE_BUILD_TYPE=Release -pixi run -e default -- cmake --build app/crowdb-diskio/build -j 4 --target crowdb-diskio -pixi run -e default -- cargo build -p crowdb-kv-server -p crowdb-diskdb -p crowdb-chunkdb -p crowdb-chunk-kv-server -p crowdb-access-server -pixi run -e default clean-env -pixi run -e default test-access-server -CROWDB_RUNTIME_ROOT="$PIXI_PROJECT_ROOT/.crowdb-runtime/ephemeral/iceberg-e2e" CROWDB_ICEBERG_E2E_PYTHON="$PIXI_PROJECT_ROOT/.pixi/envs/iceberg-e2e/bin/python" pixi run -e default -- cargo test -p crowdb-access-server --features iceberg-e2e --test iceberg_full_stack_test -- --nocapture +source tools/pixi-tasks/prepare-iceberg.sh +CROWDB_RUNTIME_ROOT="$PIXI_PROJECT_ROOT/.crowdb-runtime/ephemeral/iceberg-e2e" CROWDB_ICEBERG_E2E_PYTHON="$PIXI_PROJECT_ROOT/.pixi/envs/iceberg-e2e/bin/python" pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e --test iceberg_full_stack_test -- --nocapture -CROWDB_ICEBERG_E2E_PYTHON="$CONDA_PREFIX/bin/python" pixi run -e default -- cargo test -p crowdb-access-server --features iceberg-e2e \ +CROWDB_ICEBERG_E2E_PYTHON="$CONDA_PREFIX/bin/python" pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e \ --test iceberg_namespace_sdk_test official_complete_listing_rejects_each_spool_limit_and_releases_resources -- --ignored --exact -CROWDB_ICEBERG_E2E_PYTHON="$CONDA_PREFIX/bin/python" pixi run -e default -- cargo test -p crowdb-access-server --features iceberg-e2e \ +CROWDB_ICEBERG_E2E_PYTHON="$CONDA_PREFIX/bin/python" pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e \ --test iceberg_gc_control_test official_sdk_foreground_progresses_under_gc_backlog -- --ignored --exact diff --git a/tools/pixi-tasks/test-suite.sh b/tools/pixi-tasks/test-suite.sh index 30d67f64a..c00831e1f 100644 --- a/tools/pixi-tasks/test-suite.sh +++ b/tools/pixi-tasks/test-suite.sh @@ -10,6 +10,7 @@ pixi run test-server pixi run -e s3-e2e test-boto3-e2e pixi run -e iceberg-e2e test-iceberg-e2e pixi run -e iceberg-e2e test-iceberg-sdk +pixi run -e iceberg-e2e test-rust-iceberg-e2e pixi run test-console pixi run clean-env pixi run test-console-ui diff --git a/tools/pixi-tasks/test-unit.sh b/tools/pixi-tasks/test-unit.sh index 5e1d72057..6b8493d60 100644 --- a/tools/pixi-tasks/test-unit.sh +++ b/tools/pixi-tasks/test-unit.sh @@ -14,4 +14,5 @@ pixi run test-chunk-kv pixi run test-chunk-stream pixi run test-chunk-kv-client pixi run test-chunk-kv-server +pixi run test-access-multipart pixi run test-access-iceberg diff --git a/tools/release.py b/tools/release.py new file mode 100644 index 000000000..0ab079bf4 --- /dev/null +++ b/tools/release.py @@ -0,0 +1,145 @@ +#!/usr/bin/env python3 +# Copyright 2026-present Gian +# Licensed under the Apache License, Version 2.0. +"""Prepare a versioned candidate and dispatch the verified container workflow. + +Run through Pixi: pixi run -- python tools/release.py --dry-run + pixi run -- python tools/release.py --execute +""" + +import argparse +import difflib +import re +import subprocess +import sys +import tomllib +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +VERSION_RE = re.compile(r"^(\d+)\.(\d+)\.(\d+)(-dev)?$") +REPO = "buzzcrow/crowdb" + + +def command(*args: str, capture: bool = False) -> str: + result = subprocess.run( + args, cwd=ROOT, check=True, text=True, + stdout=subprocess.PIPE if capture else None, + timeout=60, + ) + return result.stdout.strip() if capture else "" + + +def next_version(current: str, bump: str) -> str: + match = VERSION_RE.fullmatch(current) + if match is None: + raise ValueError(f"Unsupported VERSION: {current}") + major, minor, patch = (int(part) for part in match.group(1, 2, 3)) + if bump == "major": + return f"{major + 1}.0.0" + if bump == "minor": + return f"{major}.{minor + 1}.0" + if match.group(4) is None: + patch += 1 + return f"{major}.{minor}.{patch}" + + +def changes(current: str, target: str) -> dict[Path, str]: + workspace = tomllib.loads((ROOT / "Cargo.toml").read_text(encoding="utf-8")) + workspace_count = len(workspace["workspace"]["members"]) + files = [ + "Cargo.toml", "Cargo.lock", "pixi.toml", + "app/crowdb-access-server/tests/common/iceberg_rust/Cargo.toml", + "app/crowdb-access-server/tests/common/iceberg_rust/Cargo.lock", + "app/crowdb-web/ui/package.json", "app/crowdb-web/ui/package-lock.json", + ] + updates = {ROOT / "VERSION": target + "\n"} + for name in files: + path = ROOT / name + original = path.read_text(encoding="utf-8") + old = f'version = "{current}"' if path.suffix in (".toml", ".lock") else f'"version": "{current}"' + new = old.replace(current, target) + count = original.count(old) + expected = { + "Cargo.lock": workspace_count, + "app/crowdb-web/ui/package-lock.json": 2, + }.get(name, 1) + if count != expected: + raise ValueError(f"Expected {expected} version entries in {name}, found {count}") + updates[path] = original.replace(old, new) + return updates + + +def preflight(tag: str) -> None: + remote_url = command("git", "remote", "get-url", "origin", capture=True) + if remote_url not in ( + "git@github.com:buzzcrow/crowdb.git", + "https://github.com/buzzcrow/crowdb.git", + ): + raise ValueError(f"Unexpected origin: {remote_url}") + if command("git", "branch", "--show-current", capture=True) != "main": + raise ValueError("Run --execute from the clean main branch") + if command("git", "status", "--porcelain", capture=True): + raise ValueError("Working tree must be clean before release") + head = command("git", "rev-parse", "HEAD", capture=True) + remote_lines = command("git", "ls-remote", "origin", "refs/heads/main", capture=True).split() + if not remote_lines: + raise ValueError("origin/main is unavailable") + remote = remote_lines[0] + if head != remote: + raise ValueError("Local main does not match origin/main") + if command("git", "tag", "--list", tag, capture=True): + raise ValueError(f"Local tag already exists: {tag}") + if command("git", "ls-remote", "--tags", "origin", f"refs/tags/{tag}", capture=True): + raise ValueError(f"Remote tag already exists: {tag}") + command("gh", "auth", "status") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + mode = parser.add_mutually_exclusive_group(required=True) + mode.add_argument("--dry-run", action="store_true", help="Print the plan without writing or contacting GitHub") + mode.add_argument("--execute", action="store_true", help="Update, tag, push, release and dispatch") + parser.add_argument("--bump", choices=("patch", "minor", "major"), default="patch") + parser.add_argument("--symbols", action="store_true", help="Build and upload the large optional symbol archive") + args = parser.parse_args() + + current = (ROOT / "VERSION").read_text(encoding="utf-8").strip() + target = next_version(current, args.bump) + tag = f"v{target}" + updates = changes(current, target) + print(f"Release {current} -> {target} ({tag})", flush=True) + for path in updates: + print(f" update {path.relative_to(ROOT)}", flush=True) + print(" check versions and diff; commit and push the candidate to main", flush=True) + print(f" dispatch release-container.yml (symbols: {args.symbols}); tag after verification", flush=True) + if args.dry_run: + for path, updated in updates.items(): + original = path.read_text(encoding="utf-8") + sys.stdout.writelines(difflib.unified_diff( + original.splitlines(keepends=True), updated.splitlines(keepends=True), + fromfile=str(path.relative_to(ROOT)), + tofile=str(path.relative_to(ROOT)), + )) + return + + preflight(tag) + for path, content in updates.items(): + path.write_text(content, encoding="utf-8") + command("pixi", "run", "--", "python", "tools/ci-checks/check-version.py") + command("git", "diff", "--check") + command("git", "add", *(str(path.relative_to(ROOT)) for path in updates)) + command("git", "commit", "-m", f"Release {target}") + command("git", "push", "origin", "HEAD:refs/heads/main") + revision = command("git", "rev-parse", "HEAD", capture=True) + command("gh", "workflow", "run", "release-container.yml", "--repo", REPO, + "--ref", "main", "-f", f"candidate_sha={revision}", + "-f", f"include_symbols={str(args.symbols).lower()}") + print(f"Started candidate verification for {tag}") + + +if __name__ == "__main__": + try: + main() + except (OSError, ValueError, subprocess.CalledProcessError, subprocess.TimeoutExpired) as error: + print(f"Release stopped: {error}", file=sys.stderr) + raise SystemExit(1) from error diff --git a/tools/symbolize-container-core.py b/tools/symbolize-container-core.py new file mode 100644 index 000000000..07a390508 --- /dev/null +++ b/tools/symbolize-container-core.py @@ -0,0 +1,123 @@ +#!/usr/bin/env python3 +# Copyright 2026-present Gian +# Licensed under the Apache License, Version 2.0. + +"""Show source-line stacks from a private core and exact image symbol asset.""" + +import argparse +import hashlib +import json +from pathlib import Path, PurePosixPath +import shutil +import subprocess +import tarfile +import tempfile + + +def output(*command: str) -> str: + return subprocess.check_output(command, text=True).strip() + + +def extract_symbols(archive: Path, destination: Path) -> None: + process = subprocess.Popen(["zstd", "-dc", str(archive)], stdout=subprocess.PIPE) + try: + assert process.stdout is not None + with tarfile.open(fileobj=process.stdout, mode="r|") as bundle: + for member in bundle: + name = member.name.removeprefix("./") + if name in {"", ".", "bin", "lib"} and member.isdir(): + continue + path = PurePosixPath(name) + valid = ( + name in {"VERSION", "SOURCE_REVISION", "RUNTIME_SHA256SUMS"} + or (len(path.parts) == 2 and path.parts[0] in {"bin", "lib"} and path.name.endswith(".debug")) + ) + if not valid or not member.isfile() or path.is_absolute() or ".." in path.parts: + raise ValueError("unexpected symbol archive entry") + target = destination.joinpath(*path.parts) + target.parent.mkdir(exist_ok=True) + source = bundle.extractfile(member) + if source is None: + raise ValueError("missing symbol archive contents") + with source, target.open("wb") as saved: + shutil.copyfileobj(source, saved) + if process.wait() != 0: + raise ValueError("symbol archive decompression failed") + except BaseException: + process.kill() + process.wait() + raise + finally: + if process.stdout is not None: + process.stdout.close() + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--image", required=True, help="Exact Docker image tag or digest") + parser.add_argument("--symbols", type=Path, required=True, help="Matching release symbol archive") + parser.add_argument("--binary", required=True, help="Crashed binary name, such as crowdb-monitor") + parser.add_argument("--core", type=Path, required=True, help="Privately exported core file") + args = parser.parse_args() + if Path(args.binary).name != args.binary or not args.binary.startswith("crowdb-"): + parser.error("--binary must be a CROWDB binary name") + if not args.core.is_file() or not args.symbols.is_file(): + parser.error("core and symbol archive must be readable files") + + labels = json.loads(output("docker", "image", "inspect", args.image))[0]["Config"]["Labels"] + revision = labels["org.opencontainers.image.revision"] + version = labels["org.opencontainers.image.version"] + with tempfile.TemporaryDirectory(prefix="crowdb-core-") as temporary: + root = Path(temporary) + root.chmod(0o700) + runtime = root / "runtime" + runtime.mkdir() + container = output("docker", "create", "--entrypoint", "/bin/true", args.image) + try: + for directory in ("bin", "lib"): + subprocess.run( + ["docker", "cp", f"{container}:/opt/crowdb/{directory}", str(runtime / directory)], + check=True, + ) + finally: + subprocess.run(["docker", "rm", container], check=True, capture_output=True) + symbols = root / "symbols" + symbols.mkdir() + extract_symbols(args.symbols.resolve(), symbols) + if (symbols / "SOURCE_REVISION").read_text().strip() != revision: + raise ValueError("symbol source revision differs from image") + if (symbols / "VERSION").read_text().strip() != version: + raise ValueError("symbol version differs from image") + for line in (symbols / "RUNTIME_SHA256SUMS").read_text().splitlines(): + expected, relative = line.split(maxsplit=1) + relative = relative.lstrip("*") + path = Path(relative) + if path.is_absolute() or path.parts[0] not in {"bin", "lib"} or ".." in path.parts: + raise ValueError("invalid runtime checksum path") + actual = hashlib.sha256((runtime / path).read_bytes()).hexdigest() + if actual != expected: + raise ValueError(f"image binary differs from symbols: {relative}") + binary = runtime / "bin" / args.binary + if not binary.is_file(): + parser.error("crashed binary is absent from the image") + if not (symbols / "bin" / f"{args.binary}.debug").is_file(): + raise ValueError("exact debug symbols for the crashed binary are absent") + for directory in ("bin", "lib"): + for debug in (symbols / directory).glob("*.debug"): + if debug.is_symlink() or not debug.is_file(): + raise ValueError("invalid symbol archive entry") + shutil.copyfile(debug, runtime / directory / debug.name) + print(f"Verified image version {version}, revision {revision}", flush=True) + subprocess.run( + [ + "gdb", "-q", "-batch", "-ex", "set pagination off", + "-ex", "set print frame-arguments none", + "-ex", f"set solib-search-path {runtime / 'lib'}", + "-ex", "thread apply all bt", str(binary), str(args.core.resolve()), + ], + check=True, + ) + + +if __name__ == "__main__": + main()