From fdb8e65664e53943a3fd789fb9cda62ea017f2fc Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 01:23:14 +0800 Subject: [PATCH 01/74] Extract shared multipart location composition --- Cargo.lock | 10 ++ Cargo.toml | 1 + doc/working/plan-s3-multipart.md | 66 +++++++++ lib/crowdb-access-iceberg/Cargo.toml | 1 + .../file/multipart_repository/publication.rs | 53 ++++--- lib/crowdb-access-multipart/Cargo.toml | 16 +++ lib/crowdb-access-multipart/src/lib.rs | 129 ++++++++++++++++++ .../tests/composition_test.rs | 72 ++++++++++ 8 files changed, 318 insertions(+), 30 deletions(-) create mode 100644 doc/working/plan-s3-multipart.md create mode 100644 lib/crowdb-access-multipart/Cargo.toml create mode 100644 lib/crowdb-access-multipart/src/lib.rs create mode 100644 lib/crowdb-access-multipart/tests/composition_test.rs diff --git a/Cargo.lock b/Cargo.lock index 6e4c7ca9..fe482562 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -628,6 +628,7 @@ dependencies = [ "chrono", "crc32fast", "crowdb-access-iceberg", + "crowdb-access-multipart", "crowdb-chunk-client", "crowdb-chunk-kv-client", "crowdb-common", @@ -650,6 +651,15 @@ dependencies = [ "zstd", ] +[[package]] +name = "crowdb-access-multipart" +version = "0.1.0-dev" +dependencies = [ + "crowdb-protocol", + "md-5", + "thiserror 2.0.18", +] + [[package]] name = "crowdb-access-s3" version = "0.1.0-dev" diff --git a/Cargo.toml b/Cargo.toml index 0040296b..3230db57 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -26,6 +26,7 @@ members = [ "container/crowdb-monitor", "lib/crowdb-access-s3", "lib/crowdb-access-iceberg", + "lib/crowdb-access-multipart", ] exclude = ["third-party/hyper"] # : crowdb-tree/ffi moved from `exclude` into `members` now that diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md new file mode 100644 index 00000000..3d22d4b7 --- /dev/null +++ b/doc/working/plan-s3-multipart.md @@ -0,0 +1,66 @@ + + + +# S3 Multipart Plan + +Upstream: [R167](../backlog/R167-s3-multipart-upload.md). + +Goal: add durable S3 multipart uploads while sharing protocol-neutral part +composition and state transitions with Iceberg. + +Status: active. The basic S3 milestone is complete; the protocol-neutral R190 +core referenced by R167 is absent, so extraction begins from the working +Iceberg multipart path. + +## Shared foundation + +- [x] **Metadata-only composition**: extract offset adjustment and composite + MD5 ETag from Iceberg `MultipartRepository::prepare_stream_publication` into + `lib/crowdb-access-multipart`, preserving the existing selected-part fences + in Iceberg. Include overflow and malformed location tests. Files: new crate, + Iceberg publication, workspace manifests. +- [~] **Durable transition core**: isolate session/part states, replacement + generations, completion selection, abort and recovery transitions from + Iceberg catalog-specific keys and records. Keep store CAS and namespace + adaptation in each protocol. Files: shared multipart crate, Iceberg file + repository, S3 metadata store. + +## S3 adapter and HTTP + +- [ ] **Durable S3 records**: add upload and part keys/records with raw 16-byte + MD5, selected revision and cleanup state. Use bucket identity and object key + as namespace scope; preserve immutable part data after replacement. +- [ ] **S3 routes and wire**: classify create/upload/list/complete/abort/list + uploads, parse bounded completion XML, emit compatible responses and errors. + Preserve SigV4 authentication and existing basic routes. +- [ ] **Part ingestion**: reuse the bounded streaming writer and admission + budget, persist part location/integrity before success, reconcile lost replies. +- [ ] **Atomic completion**: fence selected part generations, validate order, + count, size and checksum, compose locations through the shared core, and + publish one immutable object generation without reading part bytes. +- [ ] **Abort and expiry**: make terminal states idempotent, queue unreachable + private part data for bounded cleanup, and protect active/read-pinned data. + +## Acceptance and cleanup + +- [ ] **Focused and E2E tests**: known MD5 vectors, out-of-order/replaced parts, + invalid completion, response loss and restart, abort/expiry cleanup, and + ordinary single-part compatibility. +- [ ] **Gates and docs**: run both access crate suites, access-server E2E, + Rust fmt and clippy separately; update S3 design and remove R167 plus this + plan only after all acceptance criteria pass. + +## Files + +- `Cargo.toml`, `Cargo.lock`, `lib/crowdb-access-multipart/**` +- `lib/crowdb-access-iceberg/src/file/multipart_repository/**` +- `lib/crowdb-access-s3/src/{metadata,route,integrity,wire}.rs` and children +- `app/crowdb-access-server/src/s3/**` +- `lib/crowdb-access-s3/tests/**`, `app/crowdb-access-server/tests/**` + +## Tests + +- Unit: shared composition MD5, location offsets and bounds. +- Integration: durable S3 session/part/complete/abort transitions and Iceberg + publication regression. +- E2E: official S3 client multipart flows, response loss and restart. diff --git a/lib/crowdb-access-iceberg/Cargo.toml b/lib/crowdb-access-iceberg/Cargo.toml index 252c2779..a029acce 100644 --- a/lib/crowdb-access-iceberg/Cargo.toml +++ b/lib/crowdb-access-iceberg/Cargo.toml @@ -26,6 +26,7 @@ lz4_flex = { version = "0.11", default-features = false, features = ["std", "saf md-5 = "0.10" crowdb-chunk-kv-client = { path = "../crowdb-chunk-kv-client" } crowdb-chunk-client = { path = "../crowdb-chunk-client" } +crowdb-access-multipart = { path = "../crowdb-access-multipart" } crowdb-common = { workspace = true } crowdb-protocol = { path = "../crowdb-protocol" } flatbuffers = { workspace = true } diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs index 943407a6..8b334b6c 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs @@ -6,12 +6,22 @@ use crate::file::{ }; use crate::operation::PayloadStore; use crate::record::StorageRecord; -use crowdb_protocol::chunkdb::rpc::Location; -use md5::{Digest, Md5}; -use std::fmt::Write; +use crowdb_access_multipart::MultipartComposer; use super::{check_live, increment, MultipartRepository}; +fn raw_md5(etag: &str) -> Result<[u8; 16], ValidationError> { + let mut raw = [0_u8; 16]; + if etag.len() != 32 { + return Err(ValidationError::Record); + } + for (byte, pair) in raw.iter_mut().zip(etag.as_bytes().chunks_exact(2)) { + let pair = std::str::from_utf8(pair).map_err(|_| ValidationError::Record)?; + *byte = u8::from_str_radix(pair, 16).map_err(|_| ValidationError::Record)?; + } + Ok(raw) +} + impl MultipartRepository { /// Publishes selected durable part locations without reading or rewriting part bytes. /// # Errors @@ -37,41 +47,24 @@ impl MultipartRepository { .get(&completion.selection) .await?; let selection = MultipartSelection::decode(&bytes)?; - let mut locations = Vec::::new(); - let mut length = 0_u64; - let mut md5 = Md5::new(); + let mut composer = MultipartComposer::new(session.limits.max_file_bytes); for (index, selected) in selection.parts().iter().enumerate() { let snapshot = selection.snapshots().and_then(|snapshots| snapshots.get(index)); let Some(stream) = self.selected_stream(session, selected, snapshot).await? else { return Ok(None); }; let etag = stream.content.etag().ok_or(ValidationError::Record)?; - for pair in etag.as_bytes().chunks_exact(2) { - let pair = std::str::from_utf8(pair).map_err(|_| ValidationError::Record)?; - md5.update([u8::from_str_radix(pair, 16).map_err(|_| ValidationError::Record)?]); - } - for mut location in stream + let raw_md5 = raw_md5(etag)?; + let locations = stream .content .locations(stream.length)? - .ok_or(ValidationError::Record)? - { - location.logical_offset = location - .logical_offset - .checked_add(length) - .ok_or(ValidationError::Record)?; - locations.push(location); - } - length = length - .checked_add(stream.length) - .filter(|length| *length <= session.limits.max_file_bytes) .ok_or(ValidationError::Record)?; + composer + .push(stream.length, raw_md5, &locations) + .map_err(|_| ValidationError::Record)?; } - let mut etag = String::with_capacity(40); - for byte in md5.finalize() { - write!(&mut etag, "{byte:02x}").expect("string write cannot fail"); - } - write!(&mut etag, "-{}", selection.count()).expect("string write cannot fail"); - let content = FileContent::from_locations(&locations, length, etag)?; + let assembled = composer.finish().map_err(|_| ValidationError::Record)?; + let content = FileContent::from_locations(&assembled.locations, assembled.length, assembled.etag)?; let path = session.location.relative_key(); let extension = std::path::Path::new(path).extension(); let has_extension = |wanted: &str| extension.is_some_and(|value| value.eq_ignore_ascii_case(wanted)); @@ -93,7 +86,7 @@ impl MultipartRepository { location: session.location.clone(), kind, format, - length, + length: assembled.length, digest: [0; 32], content, hint: None, @@ -107,7 +100,7 @@ impl MultipartRepository { next.phase = MultipartPhase::Publishing; let completion = next.completion.as_mut().ok_or(ValidationError::Record)?; completion.progress.next_part = selection.count(); - completion.progress.completed_bytes = length; + completion.progress.completed_bytes = assembled.length; completion.publication = Some(publication); Ok(Some(self.exchange(session, &next).await?)) } diff --git a/lib/crowdb-access-multipart/Cargo.toml b/lib/crowdb-access-multipart/Cargo.toml new file mode 100644 index 00000000..5a33c8e9 --- /dev/null +++ b/lib/crowdb-access-multipart/Cargo.toml @@ -0,0 +1,16 @@ +[package] +name = "crowdb-access-multipart" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +description = "Protocol-neutral multipart data composition for CROWDB access services." + +[lints] +workspace = true + +[dependencies] +crowdb-protocol = { path = "../crowdb-protocol" } +md-5 = "0.10" +thiserror = { workspace = true } diff --git a/lib/crowdb-access-multipart/src/lib.rs b/lib/crowdb-access-multipart/src/lib.rs new file mode 100644 index 00000000..eb7b0749 --- /dev/null +++ b/lib/crowdb-access-multipart/src/lib.rs @@ -0,0 +1,129 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Storage-backed multipart composition shared by access protocols. +//! +//! Part bytes remain in their original chunks. A completed object contains +//! the selected parts' locations with adjusted logical offsets, while its +//! composite `ETag` uses only the parts' previously recorded raw MD5 digests. + +use std::fmt::Write as _; + +use crowdb_protocol::chunkdb::rpc::Location; +use md5::{Digest, Md5}; + +#[derive(Debug, thiserror::Error, PartialEq, Eq)] +pub enum ComposeError { + #[error("multipart part count exceeds 10000")] + TooManyParts, + #[error("multipart part locations do not cover its logical length")] + InvalidLocations, + #[error("multipart object length exceeds its limit")] + LengthLimit, + #[error("multipart location offset overflow")] + OffsetOverflow, + #[error("multipart completion has no parts")] + Empty, +} + +/// A completed metadata-only multipart object. +pub struct ComposedObject { + pub locations: Vec, + pub length: u64, + pub etag: String, +} + +/// Incrementally composes selected parts in completion order. +pub struct MultipartComposer { + locations: Vec, + length: u64, + max_length: u64, + part_count: u16, + md5: Md5, +} + +impl MultipartComposer { + #[must_use] + pub fn new(max_length: u64) -> Self { + Self { + locations: Vec::new(), + length: 0, + max_length, + part_count: 0, + md5: Md5::new(), + } + } + + /// Adds one already durable part without reading its data. + /// + /// # Errors + /// Rejects invalid or noncontiguous part locations, offset overflow, + /// too many parts, or an object that exceeds `max_length`. On error the + /// composer remains unchanged. + pub fn push( + &mut self, + length: u64, + raw_md5: [u8; 16], + locations: &[Location], + ) -> Result<(), ComposeError> { + if self.part_count >= 10_000 { + return Err(ComposeError::TooManyParts); + } + let next_length = self + .length + .checked_add(length) + .ok_or(ComposeError::OffsetOverflow)?; + if next_length > self.max_length { + return Err(ComposeError::LengthLimit); + } + let mut cursor = 0_u64; + for location in locations { + if location.chunk_id.is_none() + || location.length == 0 + || location.logical_length == 0 + || location.logical_offset != cursor + || location.offset.checked_add(location.length).is_none() + { + return Err(ComposeError::InvalidLocations); + } + cursor = cursor + .checked_add(location.logical_length) + .ok_or(ComposeError::OffsetOverflow)?; + self.length + .checked_add(location.logical_offset) + .ok_or(ComposeError::OffsetOverflow)?; + } + if cursor != length { + return Err(ComposeError::InvalidLocations); + } + for location in locations { + let mut adjusted = location.clone(); + adjusted.logical_offset += self.length; + self.locations.push(adjusted); + } + self.md5.update(raw_md5); + self.length = next_length; + self.part_count += 1; + Ok(()) + } + + /// Completes the composite `ETag` from the selected raw part MD5 values. + /// + /// # Errors + /// Rejects a completion with no selected parts. + pub fn finish(self) -> Result { + if self.part_count == 0 { + return Err(ComposeError::Empty); + } + let mut etag = String::with_capacity(40); + for byte in self.md5.finalize() { + write!(&mut etag, "{byte:02x}").expect("string write cannot fail"); + } + write!(&mut etag, "-{}", self.part_count).expect("string write cannot fail"); + Ok(ComposedObject { + locations: self.locations, + length: self.length, + etag, + }) + } +} diff --git a/lib/crowdb-access-multipart/tests/composition_test.rs b/lib/crowdb-access-multipart/tests/composition_test.rs new file mode 100644 index 00000000..bdbd9a40 --- /dev/null +++ b/lib/crowdb-access-multipart/tests/composition_test.rs @@ -0,0 +1,72 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_multipart::{ComposeError, MultipartComposer}; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; + +fn location(part: u64, logical_length: u64) -> Location { + Location { + chunk_id: Some(ChunkId { high: 1, low: part }), + offset: 100, + length: logical_length + 34, + logical_offset: 0, + logical_length, + } +} + +#[test] +fn completion_composes_locations_and_saved_md5_in_selected_order() { + let mut composer = MultipartComposer::new(12); + composer.push(7, [0x11; 16], &[location(2, 7)]).unwrap(); + composer.push(5, [0x22; 16], &[location(1, 5)]).unwrap(); + let object = composer.finish().unwrap(); + assert_eq!(object.length, 12); + assert_eq!(object.locations[0].logical_offset, 0); + assert_eq!(object.locations[1].logical_offset, 7); + assert_eq!(object.locations[0].chunk_id.as_ref().unwrap().low, 2); + assert_eq!(object.locations[1].chunk_id.as_ref().unwrap().low, 1); + assert_eq!(object.etag, "b4ab393b73e0e71830bf2bf0e63c4d91-2"); +} + +#[test] +fn malformed_part_does_not_advance_composition() { + let mut composer = MultipartComposer::new(12); + let mut invalid = location(1, 5); + invalid.logical_offset = 1; + assert_eq!( + composer.push(5, [0x11; 16], &[invalid]), + Err(ComposeError::InvalidLocations) + ); + composer.push(5, [0x22; 16], &[location(2, 5)]).unwrap(); + let object = composer.finish().unwrap(); + assert_eq!(object.length, 5); + assert_eq!(object.locations[0].logical_offset, 0); + assert!(object.etag.ends_with("-1")); +} + +#[test] +fn length_and_offset_overflow_are_rejected() { + let mut composer = MultipartComposer::new(u64::MAX); + let first = Location { + chunk_id: Some(ChunkId { high: 1, low: 1 }), + offset: 100, + length: 1, + logical_offset: 0, + logical_length: u64::MAX - 1, + }; + composer.push(u64::MAX - 1, [0; 16], &[first]).unwrap(); + assert_eq!( + composer.push(2, [0; 16], &[location(2, 2)]), + Err(ComposeError::OffsetOverflow) + ); + assert_eq!(composer.finish().unwrap().length, u64::MAX - 1); +} + +#[test] +fn empty_selection_is_rejected() { + assert!(matches!( + MultipartComposer::new(1).finish(), + Err(ComposeError::Empty) + )); +} From 367f875cba90c50edb997c66fe3112e06ddb794c Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 01:28:12 +0800 Subject: [PATCH 02/74] Share multipart selection and part accounting rules --- .../src/file/multipart_repository/parts.rs | 23 +++-- .../src/file/multipart_selection.rs | 20 +---- lib/crowdb-access-multipart/src/lib.rs | 4 + lib/crowdb-access-multipart/src/state.rs | 86 +++++++++++++++++++ .../tests/state_test.rs | 59 +++++++++++++ 5 files changed, 166 insertions(+), 26 deletions(-) create mode 100644 lib/crowdb-access-multipart/src/state.rs create mode 100644 lib/crowdb-access-multipart/tests/state_test.rs diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs index 178c87ed..21d4bde1 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs @@ -4,6 +4,7 @@ use crate::file::{MultipartPart, MultipartPartMutation, MultipartPhase, Multipar use crate::key::{CatalogScope, IcebergKey}; use crate::operation::mutation_identity; use crate::record::StorageRecord; +use crowdb_access_multipart::{reserve_part_accounting, PartAccounting}; use super::{check_live, increment, MultipartRepository}; @@ -180,15 +181,19 @@ impl MultipartRepository { before.validate_for(session)?; } let mut next = increment(session)?; - next.part_count = session - .part_count - .checked_add(u16::from(before.is_none())) - .ok_or(ValidationError::Record)?; - next.staged_bytes = session - .staged_bytes - .checked_sub(before.as_ref().map_or(0, MultipartPart::length)) - .and_then(|bytes| bytes.checked_add(part.length())) - .ok_or(ValidationError::Record)?; + let accounting = reserve_part_accounting( + PartAccounting { + count: session.part_count, + staged_bytes: session.staged_bytes, + }, + before.as_ref().map(MultipartPart::length), + part.length(), + session.limits.max_parts, + session.limits.max_staged_bytes, + ) + .map_err(|_| ValidationError::Record)?; + next.part_count = accounting.count; + next.staged_bytes = accounting.staged_bytes; next.pending = Some(MultipartPartMutation { before, after }); Ok(self.exchange(session, &next).await?.then_some(next)) } diff --git a/lib/crowdb-access-iceberg/src/file/multipart_selection.rs b/lib/crowdb-access-iceberg/src/file/multipart_selection.rs index 8ae47ec8..c519035d 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_selection.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_selection.rs @@ -1,6 +1,8 @@ use crate::error::ValidationError; use crate::operation::MAX_PAYLOAD_BYTES; use bincode::Options; +use crowdb_access_multipart::validate_selected_parts; +pub use crowdb_access_multipart::SelectedPart; use serde::{Deserialize, Serialize}; use sha2::{Digest, Sha256}; @@ -10,13 +12,6 @@ const MAGIC_V1: &[u8; 5] = b"ICMS\x01"; const MAGIC_V2: &[u8; 5] = b"ICMS\x02"; const ENTRY_BYTES: usize = 42; -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub struct SelectedPart { - pub number: u16, - pub revision: u64, - pub digest: [u8; 32], -} - #[derive(Clone, Debug, Eq, PartialEq)] pub struct MultipartSelection { parts: Vec, @@ -35,16 +30,7 @@ impl MultipartSelection { /// # Errors /// Rejects empty, oversized, unordered or duplicate part selections. pub fn new(parts: Vec) -> Result { - if parts.is_empty() || parts.len() > 10_000 { - return Err(ValidationError::Record); - } - let mut previous = 0; - for part in &parts { - if part.number <= previous || part.number > 10_000 || part.revision == 0 { - return Err(ValidationError::Record); - } - previous = part.number; - } + validate_selected_parts(&parts, 10_000).map_err(|_| ValidationError::Record)?; let count = u16::try_from(parts.len()).map_err(|_| ValidationError::Record)?; Ok(Self { parts, diff --git a/lib/crowdb-access-multipart/src/lib.rs b/lib/crowdb-access-multipart/src/lib.rs index eb7b0749..aafad0e8 100644 --- a/lib/crowdb-access-multipart/src/lib.rs +++ b/lib/crowdb-access-multipart/src/lib.rs @@ -12,6 +12,10 @@ use std::fmt::Write as _; use crowdb_protocol::chunkdb::rpc::Location; use md5::{Digest, Md5}; +mod state; + +pub use state::{reserve_part_accounting, validate_selected_parts, PartAccounting, SelectedPart, StateError}; + #[derive(Debug, thiserror::Error, PartialEq, Eq)] pub enum ComposeError { #[error("multipart part count exceeds 10000")] diff --git a/lib/crowdb-access-multipart/src/state.rs b/lib/crowdb-access-multipart/src/state.rs new file mode 100644 index 00000000..0a84aa40 --- /dev/null +++ b/lib/crowdb-access-multipart/src/state.rs @@ -0,0 +1,86 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Protocol-neutral part selection and reservation invariants. + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct SelectedPart { + pub number: u16, + pub revision: u64, + pub digest: [u8; 32], +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct PartAccounting { + pub count: u16, + pub staged_bytes: u64, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] +pub enum StateError { + #[error("multipart selection has no parts")] + EmptySelection, + #[error("multipart part number is invalid or out of order")] + InvalidPartNumber, + #[error("multipart selected revision is zero")] + InvalidRevision, + #[error("multipart part count exceeds its limit")] + PartLimit, + #[error("multipart staged byte total exceeds its limit")] + StagedLimit, + #[error("multipart part counters are inconsistent")] + InvalidAccounting, +} + +/// Validates one ordered completion selection independent of wire format. +/// +/// # Errors +/// Rejects empty, duplicate, descending, oversized or zero-revision entries. +pub fn validate_selected_parts(parts: &[SelectedPart], max_parts: u16) -> Result<(), StateError> { + if parts.is_empty() { + return Err(StateError::EmptySelection); + } + if parts.len() > usize::from(max_parts.min(10_000)) { + return Err(StateError::PartLimit); + } + let mut previous = 0; + for part in parts { + if part.number <= previous || part.number > max_parts.min(10_000) { + return Err(StateError::InvalidPartNumber); + } + if part.revision == 0 { + return Err(StateError::InvalidRevision); + } + previous = part.number; + } + Ok(()) +} + +/// Computes the counters to fence one new or replacement part publication. +/// +/// # Errors +/// Rejects count, byte and arithmetic limit violations before any metadata +/// mutation. A replacement keeps the part count and subtracts its old bytes. +pub fn reserve_part_accounting( + current: PartAccounting, + old_length: Option, + new_length: u64, + max_parts: u16, + max_staged_bytes: u64, +) -> Result { + let count = current + .count + .checked_add(u16::from(old_length.is_none())) + .filter(|count| *count <= max_parts.min(10_000)) + .ok_or(StateError::PartLimit)?; + let staged_bytes = current + .staged_bytes + .checked_sub(old_length.unwrap_or(0)) + .ok_or(StateError::InvalidAccounting)? + .checked_add(new_length) + .ok_or(StateError::StagedLimit)?; + if staged_bytes > max_staged_bytes { + return Err(StateError::StagedLimit); + } + Ok(PartAccounting { count, staged_bytes }) +} diff --git a/lib/crowdb-access-multipart/tests/state_test.rs b/lib/crowdb-access-multipart/tests/state_test.rs new file mode 100644 index 00000000..84b084ce --- /dev/null +++ b/lib/crowdb-access-multipart/tests/state_test.rs @@ -0,0 +1,59 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_multipart::{ + reserve_part_accounting, validate_selected_parts, PartAccounting, SelectedPart, StateError, +}; + +fn selected(number: u16, revision: u64) -> SelectedPart { + SelectedPart { + number, + revision, + digest: [1; 32], + } +} + +#[test] +fn selected_parts_require_strict_order_and_stable_revisions() { + assert_eq!(validate_selected_parts(&[], 10), Err(StateError::EmptySelection)); + assert_eq!( + validate_selected_parts(&[selected(2, 1), selected(2, 2)], 10), + Err(StateError::InvalidPartNumber) + ); + assert_eq!( + validate_selected_parts(&[selected(2, 1), selected(1, 2)], 10), + Err(StateError::InvalidPartNumber) + ); + assert_eq!( + validate_selected_parts(&[selected(1, 0)], 10), + Err(StateError::InvalidRevision) + ); + assert_eq!( + validate_selected_parts(&[selected(2, 1), selected(9, 4)], 10), + Ok(()) + ); +} + +#[test] +fn replacing_a_part_preserves_count_and_reclaims_its_old_credit() { + let current = PartAccounting { + count: 2, + staged_bytes: 15, + }; + let next = reserve_part_accounting(current, Some(10), 8, 3, 20).unwrap(); + assert_eq!( + next, + PartAccounting { + count: 2, + staged_bytes: 13 + } + ); + assert_eq!( + reserve_part_accounting(current, None, 6, 3, 20), + Err(StateError::StagedLimit) + ); + assert_eq!( + reserve_part_accounting(current, Some(16), 1, 3, 20), + Err(StateError::InvalidAccounting) + ); +} From d89cf3e88bdf80b9ea061929767c7c7691203161 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 01:33:55 +0800 Subject: [PATCH 03/74] Add durable S3 multipart record types --- Cargo.lock | 3 + doc/working/plan-s3-multipart.md | 2 + lib/crowdb-access-multipart/Cargo.toml | 1 + lib/crowdb-access-multipart/src/state.rs | 2 +- lib/crowdb-access-s3/Cargo.toml | 2 + lib/crowdb-access-s3/src/metadata.rs | 2 + lib/crowdb-access-s3/src/metadata/key.rs | 77 ++++++- .../src/metadata/multipart.rs | 218 ++++++++++++++++++ .../tests/metadata_key_test.rs | 20 ++ .../tests/metadata_multipart_test.rs | 113 +++++++++ 10 files changed, 438 insertions(+), 2 deletions(-) create mode 100644 lib/crowdb-access-s3/src/metadata/multipart.rs create mode 100644 lib/crowdb-access-s3/tests/metadata_multipart_test.rs diff --git a/Cargo.lock b/Cargo.lock index fe482562..ae46c778 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -657,6 +657,7 @@ version = "0.1.0-dev" dependencies = [ "crowdb-protocol", "md-5", + "serde", "thiserror 2.0.18", ] @@ -671,6 +672,7 @@ dependencies = [ "base64", "bincode", "chrono", + "crowdb-access-multipart", "crowdb-chunk-client", "crowdb-chunk-kv-client", "crowdb-protocol", @@ -682,6 +684,7 @@ dependencies = [ "md5", "percent-encoding", "rand 0.8.6", + "serde", "sha2", "subtle", "thiserror 2.0.18", diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 3d22d4b7..3f923c93 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -30,6 +30,8 @@ Iceberg multipart path. - [ ] **Durable S3 records**: add upload and part keys/records with raw 16-byte MD5, selected revision and cleanup state. Use bucket identity and object key as namespace scope; preserve immutable part data after replacement. + Versioned session/part records and ordered, binary-safe keys are in place; + persistence operations, replacement generations and cleanup state remain. - [ ] **S3 routes and wire**: classify create/upload/list/complete/abort/list uploads, parse bounded completion XML, emit compatible responses and errors. Preserve SigV4 authentication and existing basic routes. diff --git a/lib/crowdb-access-multipart/Cargo.toml b/lib/crowdb-access-multipart/Cargo.toml index 5a33c8e9..9b2177de 100644 --- a/lib/crowdb-access-multipart/Cargo.toml +++ b/lib/crowdb-access-multipart/Cargo.toml @@ -13,4 +13,5 @@ workspace = true [dependencies] crowdb-protocol = { path = "../crowdb-protocol" } md-5 = "0.10" +serde = { version = "1", features = ["derive"] } thiserror = { workspace = true } diff --git a/lib/crowdb-access-multipart/src/state.rs b/lib/crowdb-access-multipart/src/state.rs index 0a84aa40..5d48960a 100644 --- a/lib/crowdb-access-multipart/src/state.rs +++ b/lib/crowdb-access-multipart/src/state.rs @@ -3,7 +3,7 @@ //! Protocol-neutral part selection and reservation invariants. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, serde::Serialize, serde::Deserialize)] pub struct SelectedPart { pub number: u16, pub revision: u64, diff --git a/lib/crowdb-access-s3/Cargo.toml b/lib/crowdb-access-s3/Cargo.toml index 8c76e1f9..32aa7e7e 100644 --- a/lib/crowdb-access-s3/Cargo.toml +++ b/lib/crowdb-access-s3/Cargo.toml @@ -19,6 +19,7 @@ base64 = "0.22" bincode = "1" chrono = { version = "0.4", default-features = false, features = ["std"] } crowdb-chunk-client = { path = "../crowdb-chunk-client" } +crowdb-access-multipart = { path = "../crowdb-access-multipart" } crowdb-chunk-kv-client = { path = "../crowdb-chunk-kv-client" } crowdb-protocol = { path = "../crowdb-protocol" } flatbuffers = { workspace = true } @@ -29,6 +30,7 @@ md5 = "0.7" percent-encoding = "2" libc = "0.2" rand = "0.8" +serde = { version = "1", features = ["derive"] } sha2 = "0.10" subtle = "2" thiserror = { workspace = true } diff --git a/lib/crowdb-access-s3/src/metadata.rs b/lib/crowdb-access-s3/src/metadata.rs index decd1ea7..5c30e402 100644 --- a/lib/crowdb-access-s3/src/metadata.rs +++ b/lib/crowdb-access-s3/src/metadata.rs @@ -4,6 +4,7 @@ //! Durable S3 namespace metadata. mod key; +mod multipart; mod namespace; mod record; mod store; @@ -21,6 +22,7 @@ mod generated { } pub use key::{BucketId, MetadataKey, MetadataKeyError, TenantId}; +pub use multipart::{MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord}; pub use namespace::{BucketDeleteOutcome, BucketNamespace, BucketNamespaceError}; pub use record::{BucketNameRecord, MetadataRecordError, ObjectRecord}; pub use store::{ChunkKvMetadataStore, MetadataStoreError, PutIfAbsentOutcome}; diff --git a/lib/crowdb-access-s3/src/metadata/key.rs b/lib/crowdb-access-s3/src/metadata/key.rs index cfdc7492..3c4657e8 100644 --- a/lib/crowdb-access-s3/src/metadata/key.rs +++ b/lib/crowdb-access-s3/src/metadata/key.rs @@ -5,6 +5,8 @@ use std::fmt; const BUCKET_NAME_KIND: u8 = 1; const OBJECT_KIND: u8 = 2; +const MULTIPART_SESSION_KIND: u8 = 3; +const MULTIPART_PART_KIND: u8 = 4; const MAX_KEY_BYTES: usize = 1024; /// A tenant namespace identity. @@ -30,7 +32,7 @@ impl TenantId { } /// An immutable bucket identity. -#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd)] +#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd, serde::Serialize, serde::Deserialize)] pub struct BucketId([u8; 16]); impl BucketId { @@ -150,12 +152,84 @@ impl MetadataKey { end.push(OBJECT_KIND + 1); end } + + /// Starts the ordered multipart upload interval for one bucket. + #[must_use] + pub fn multipart_session_prefix(tenant: &TenantId, bucket: BucketId) -> Vec { + let mut key = namespace_prefix(tenant); + key.extend_from_slice(bucket.as_bytes()); + key.push(MULTIPART_SESSION_KIND); + key + } + + /// Ends the multipart upload interval for one bucket. + #[must_use] + pub fn multipart_session_end(tenant: &TenantId, bucket: BucketId) -> Vec { + let mut key = namespace_prefix(tenant); + key.extend_from_slice(bucket.as_bytes()); + key.push(MULTIPART_SESSION_KIND + 1); + key + } + + /// Identifies one upload by object key and stable upload ID. + /// + /// # Errors + /// Rejects an empty or oversized object key. + pub fn multipart_session( + tenant: &TenantId, + bucket: BucketId, + object: &[u8], + upload_id: &[u8; 16], + ) -> Result, MetadataKeyError> { + validate_component("object key", object)?; + let mut key = Self::multipart_session_prefix(tenant, bucket); + append_ordered_bytes(&mut key, object); + key.extend_from_slice(upload_id); + Ok(key) + } + + /// Starts the part interval of one upload. + #[must_use] + pub fn multipart_part_prefix(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { + let mut key = namespace_prefix(tenant); + key.extend_from_slice(bucket.as_bytes()); + key.push(MULTIPART_PART_KIND); + key.extend_from_slice(upload_id); + key + } + + /// Ends the part interval of one upload. + #[must_use] + pub fn multipart_part_end(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { + let mut key = Self::multipart_part_prefix(tenant, bucket, upload_id); + increment_lexicographic(&mut key); + key + } + + /// Identifies one part number independently of its replacement revision. + /// + /// # Errors + /// Rejects part numbers outside the S3 multipart range. + pub fn multipart_part( + tenant: &TenantId, + bucket: BucketId, + upload_id: &[u8; 16], + number: u16, + ) -> Result, MetadataKeyError> { + if number == 0 || number > 10_000 { + return Err(MetadataKeyError::InvalidPartNumber); + } + let mut key = Self::multipart_part_prefix(tenant, bucket, upload_id); + key.extend_from_slice(&number.to_be_bytes()); + Ok(key) + } } #[derive(Clone, Debug, Eq, PartialEq)] pub enum MetadataKeyError { Empty(&'static str), TooLong(&'static str), + InvalidPartNumber, } impl fmt::Display for MetadataKeyError { @@ -163,6 +237,7 @@ impl fmt::Display for MetadataKeyError { match self { Self::Empty(component) => write!(formatter, "{component} cannot be empty"), Self::TooLong(component) => write!(formatter, "{component} exceeds {MAX_KEY_BYTES} bytes"), + Self::InvalidPartNumber => write!(formatter, "multipart part number must be in 1..=10000"), } } } diff --git a/lib/crowdb-access-s3/src/metadata/multipart.rs b/lib/crowdb-access-s3/src/metadata/multipart.rs new file mode 100644 index 00000000..0aca3a1f --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart.rs @@ -0,0 +1,218 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Versioned durable S3 multipart session and part values. + +use bincode::Options as _; +use crowdb_access_multipart::{validate_selected_parts, MultipartComposer, SelectedPart}; +use crowdb_protocol::chunkdb::rpc::Location; +use serde::{Deserialize, Serialize}; + +use super::BucketId; + +const SESSION_MAGIC: [u8; 5] = *b"S3MS\x01"; +const PART_MAGIC: [u8; 5] = *b"S3MP\x01"; +const MAX_RECORD_BYTES: u64 = 1024 * 1024; +const MAX_OBJECT_KEY_BYTES: usize = 1024; +const MAX_CONTENT_TYPE_BYTES: usize = 1024; + +#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize)] +pub enum MultipartPhase { + Open, + Completing, + Publishing, + Published, + Aborted, +} + +#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] +pub struct MultipartSessionRecord { + pub bucket_id: BucketId, + pub object_key: Vec, + pub upload_id: [u8; 16], + pub revision: u64, + pub phase: MultipartPhase, + pub created_ms: u64, + pub expires_ms: u64, + pub content_type: String, + pub max_parts: u16, + pub max_part_bytes: u64, + pub max_object_bytes: u64, + pub max_staged_bytes: u64, + pub part_count: u16, + pub staged_bytes: u64, + pub selection: Option>, + pub etag: Option, +} + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct MultipartPartRecord { + pub bucket_id: BucketId, + pub upload_id: [u8; 16], + pub number: u16, + pub revision: u64, + pub modified_ms: u64, + pub length: u64, + pub raw_md5: [u8; 16], + pub locations: Vec, +} + +#[derive(Debug, Eq, PartialEq, thiserror::Error)] +pub enum MultipartRecordError { + #[error("invalid multipart record")] + Invalid, + #[error("multipart record exceeds its byte limit")] + TooLarge, + #[error("multipart record belongs to another key")] + Identity, +} + +impl MultipartSessionRecord { + /// Encodes one validated session with a distinct schema version. + /// + /// # Errors + /// Rejects incoherent phases, bounds and oversized records. + pub fn encode(&self) -> Result, MultipartRecordError> { + self.validate()?; + encode(SESSION_MAGIC, self) + } + + /// Decodes and binds one session to the requested bucket, key and upload. + /// + /// # Errors + /// Rejects corrupt, oversized, foreign or incoherent values. + pub fn decode( + bytes: &[u8], + bucket: BucketId, + object: &[u8], + upload_id: &[u8; 16], + ) -> Result { + let record: Self = decode(SESSION_MAGIC, bytes)?; + record.validate()?; + if record.bucket_id != bucket || record.object_key != object || &record.upload_id != upload_id { + return Err(MultipartRecordError::Identity); + } + Ok(record) + } + + fn validate(&self) -> Result<(), MultipartRecordError> { + if self.object_key.is_empty() + || self.object_key.len() > MAX_OBJECT_KEY_BYTES + || self.upload_id == [0; 16] + || self.revision == 0 + || self.created_ms >= self.expires_ms + || self.content_type.len() > MAX_CONTENT_TYPE_BYTES + || self.max_parts == 0 + || self.max_parts > 10_000 + || self.max_part_bytes == 0 + || self.max_part_bytes > self.max_object_bytes + || self.max_object_bytes > self.max_staged_bytes + || self.part_count > self.max_parts + || self.staged_bytes > self.max_staged_bytes + { + return Err(MultipartRecordError::Invalid); + } + if let Some(selection) = &self.selection { + validate_selected_parts(selection, self.max_parts).map_err(|_| MultipartRecordError::Invalid)?; + if selection.len() > usize::from(self.part_count) { + return Err(MultipartRecordError::Invalid); + } + } + let selected_count = self.selection.as_ref().map_or(0, Vec::len); + match self.phase { + MultipartPhase::Open if self.selection.is_none() && self.etag.is_none() => Ok(()), + MultipartPhase::Completing if self.selection.is_some() && self.etag.is_none() => Ok(()), + MultipartPhase::Publishing | MultipartPhase::Published + if self.selection.is_some() + && self + .etag + .as_ref() + .is_some_and(|etag| valid_etag(etag, selected_count)) => + { + Ok(()) + } + MultipartPhase::Aborted if self.etag.is_none() => Ok(()), + _ => Err(MultipartRecordError::Invalid), + } + } +} + +impl MultipartPartRecord { + /// Encodes one immutable selected part generation. + /// + /// # Errors + /// Rejects invalid identity, locations or oversized records. + pub fn encode(&self) -> Result, MultipartRecordError> { + self.validate()?; + encode(PART_MAGIC, self) + } + + /// Decodes and binds a part to the requested bucket, upload and number. + /// + /// # Errors + /// Rejects corrupt, oversized, foreign or incoherent values. + pub fn decode( + bytes: &[u8], + bucket: BucketId, + upload_id: &[u8; 16], + number: u16, + ) -> Result { + let record: Self = decode(PART_MAGIC, bytes)?; + record.validate()?; + if record.bucket_id != bucket || &record.upload_id != upload_id || record.number != number { + return Err(MultipartRecordError::Identity); + } + Ok(record) + } + + fn validate(&self) -> Result<(), MultipartRecordError> { + if self.upload_id == [0; 16] || self.number == 0 || self.number > 10_000 || self.revision == 0 { + return Err(MultipartRecordError::Invalid); + } + let mut composer = MultipartComposer::new(self.length); + composer + .push(self.length, self.raw_md5, &self.locations) + .map_err(|_| MultipartRecordError::Invalid)?; + composer.finish().map_err(|_| MultipartRecordError::Invalid)?; + Ok(()) + } +} + +fn valid_etag(etag: &str, selected_count: usize) -> bool { + let Some((digest, count)) = etag.split_once('-') else { + return false; + }; + digest.len() == 32 + && digest + .bytes() + .all(|byte| byte.is_ascii_hexdigit() && !byte.is_ascii_uppercase()) + && count + .parse::() + .is_ok_and(|count| count > 0 && usize::from(count) == selected_count) +} + +fn encode(magic: [u8; 5], value: &T) -> Result, MultipartRecordError> { + let mut bytes = magic.to_vec(); + let encoded = bincode::DefaultOptions::new() + .with_fixint_encoding() + .with_limit(MAX_RECORD_BYTES - 5) + .serialize(value) + .map_err(|_| MultipartRecordError::TooLarge)?; + bytes.extend_from_slice(&encoded); + Ok(bytes) +} + +fn decode Deserialize<'de>>(magic: [u8; 5], bytes: &[u8]) -> Result { + if bytes.len() as u64 > MAX_RECORD_BYTES { + return Err(MultipartRecordError::TooLarge); + } + if bytes.get(..5) != Some(&magic[..]) { + return Err(MultipartRecordError::Invalid); + } + bincode::DefaultOptions::new() + .with_fixint_encoding() + .with_limit(MAX_RECORD_BYTES - 5) + .reject_trailing_bytes() + .deserialize(&bytes[5..]) + .map_err(|_| MultipartRecordError::Invalid) +} diff --git a/lib/crowdb-access-s3/tests/metadata_key_test.rs b/lib/crowdb-access-s3/tests/metadata_key_test.rs index c774e385..4fc5abd7 100644 --- a/lib/crowdb-access-s3/tests/metadata_key_test.rs +++ b/lib/crowdb-access-s3/tests/metadata_key_test.rs @@ -61,3 +61,23 @@ fn maximum_binary_object_key_stays_within_its_bucket_interval() { assert!(MetadataKey::object_prefix(&tenant, bucket) < encoded); assert!(encoded < MetadataKey::object_end(&tenant, bucket)); } + +#[test] +fn multipart_keys_keep_uploads_and_parts_in_separate_bounded_intervals() { + let tenant = TenantId::new(b"tenant".to_vec()).unwrap(); + let bucket = BucketId::new([5; 16]); + let upload = [9; 16]; + let session = MetadataKey::multipart_session(&tenant, bucket, b"a\0", &upload).unwrap(); + let start = MetadataKey::multipart_session_prefix(&tenant, bucket); + let end = MetadataKey::multipart_session_end(&tenant, bucket); + assert!(start < session && session < end); + let part_start = MetadataKey::multipart_part_prefix(&tenant, bucket, &upload); + let part_end = MetadataKey::multipart_part_end(&tenant, bucket, &upload); + let first = MetadataKey::multipart_part(&tenant, bucket, &upload, 1).unwrap(); + let last = MetadataKey::multipart_part(&tenant, bucket, &upload, 10_000).unwrap(); + assert!(part_start < first && first < last && last < part_end); + assert!(MetadataKey::multipart_part(&tenant, bucket, &upload, 0).is_err()); + assert!(MetadataKey::multipart_part(&tenant, bucket, &upload, 10_001).is_err()); + assert!(MetadataKey::object_end(&tenant, bucket) <= start); + assert!(end <= part_start); +} diff --git a/lib/crowdb-access-s3/tests/metadata_multipart_test.rs b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs new file mode 100644 index 00000000..810575b7 --- /dev/null +++ b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs @@ -0,0 +1,113 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_multipart::SelectedPart; +use crowdb_access_s3::metadata::{ + BucketId, MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord, +}; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; + +fn session() -> MultipartSessionRecord { + MultipartSessionRecord { + bucket_id: BucketId::new([3; 16]), + object_key: b"key\0part".to_vec(), + upload_id: [7; 16], + revision: 1, + phase: MultipartPhase::Open, + created_ms: 100, + expires_ms: 200, + content_type: "application/octet-stream".into(), + max_parts: 10, + max_part_bytes: 100, + max_object_bytes: 500, + max_staged_bytes: 1_000, + part_count: 0, + staged_bytes: 0, + selection: None, + etag: None, + } +} + +fn part() -> MultipartPartRecord { + MultipartPartRecord { + bucket_id: BucketId::new([3; 16]), + upload_id: [7; 16], + number: 1, + revision: 2, + modified_ms: 150, + length: 5, + raw_md5: [9; 16], + locations: vec![Location { + chunk_id: Some(ChunkId { high: 1, low: 2 }), + offset: 10, + length: 39, + logical_offset: 0, + logical_length: 5, + }], + } +} + +#[test] +fn session_and_part_records_round_trip_only_under_their_own_keys() { + let session = session(); + let bytes = session.encode().unwrap(); + assert_eq!( + MultipartSessionRecord::decode(&bytes, session.bucket_id, &session.object_key, &session.upload_id) + .unwrap(), + session + ); + assert_eq!( + MultipartSessionRecord::decode(&bytes, session.bucket_id, b"other", &session.upload_id), + Err(MultipartRecordError::Identity) + ); + + let part = part(); + let bytes = part.encode().unwrap(); + assert_eq!( + MultipartPartRecord::decode(&bytes, part.bucket_id, &part.upload_id, part.number).unwrap(), + part + ); + assert_eq!( + MultipartPartRecord::decode(&bytes, part.bucket_id, &part.upload_id, 2), + Err(MultipartRecordError::Identity) + ); +} + +#[test] +fn completed_session_requires_a_matching_selected_count_and_etag() { + let mut session = session(); + session.phase = MultipartPhase::Publishing; + session.part_count = 1; + session.staged_bytes = 5; + session.selection = Some(vec![SelectedPart { + number: 1, + revision: 2, + digest: [4; 32], + }]); + session.etag = Some("11111111111111111111111111111111-2".into()); + assert_eq!(session.encode(), Err(MultipartRecordError::Invalid)); + session.etag = Some("11111111111111111111111111111111-1".into()); + assert!(session.encode().is_ok()); + session.phase = MultipartPhase::Open; + assert_eq!(session.encode(), Err(MultipartRecordError::Invalid)); +} + +#[test] +fn malformed_or_unbounded_records_fail_before_exposure() { + let mut invalid_part = part(); + invalid_part.locations[0].logical_offset = 1; + assert_eq!(invalid_part.encode(), Err(MultipartRecordError::Invalid)); + let part = part(); + let mut bytes = part.encode().unwrap(); + bytes.push(0); + assert_eq!( + MultipartPartRecord::decode(&bytes, part.bucket_id, &part.upload_id, part.number), + Err(MultipartRecordError::Invalid) + ); + let oversized = vec![0; 1024 * 1024 + 1]; + assert_eq!( + MultipartSessionRecord::decode(&oversized, session().bucket_id, b"key", &[7; 16]), + Err(MultipartRecordError::TooLarge) + ); +} From a422511d26a09aa15aa173feead6a3354eb6629a Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 01:39:29 +0800 Subject: [PATCH 04/74] Add CAS-backed S3 multipart session repository --- doc/working/plan-s3-multipart.md | 4 +- lib/crowdb-access-s3/Cargo.toml | 2 +- lib/crowdb-access-s3/src/metadata.rs | 2 + .../src/metadata/multipart_repository.rs | 222 +++++++++++++++ .../tests/multipart_repository_test.rs | 268 ++++++++++++++++++ 5 files changed, 496 insertions(+), 2 deletions(-) create mode 100644 lib/crowdb-access-s3/src/metadata/multipart_repository.rs create mode 100644 lib/crowdb-access-s3/tests/multipart_repository_test.rs diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 3f923c93..8b2fe580 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -31,7 +31,9 @@ Iceberg multipart path. MD5, selected revision and cleanup state. Use bucket identity and object key as namespace scope; preserve immutable part data after replacement. Versioned session/part records and ordered, binary-safe keys are in place; - persistence operations, replacement generations and cleanup state remain. + CAS-backed begin, phase transition and part replacement now use exact-value + confirmation after lost replies. Completion snapshots, HTTP wiring and + cleanup state remain. - [ ] **S3 routes and wire**: classify create/upload/list/complete/abort/list uploads, parse bounded completion XML, emit compatible responses and errors. Preserve SigV4 authentication and existing basic routes. diff --git a/lib/crowdb-access-s3/Cargo.toml b/lib/crowdb-access-s3/Cargo.toml index 32aa7e7e..80ded829 100644 --- a/lib/crowdb-access-s3/Cargo.toml +++ b/lib/crowdb-access-s3/Cargo.toml @@ -41,4 +41,4 @@ zeroize = { version = "1", features = ["derive"] } [build-dependencies] [dev-dependencies] -tokio = { workspace = true, features = ["macros", "rt-multi-thread"] } +tokio = { workspace = true, features = ["macros", "rt-multi-thread", "sync"] } diff --git a/lib/crowdb-access-s3/src/metadata.rs b/lib/crowdb-access-s3/src/metadata.rs index 5c30e402..d78285d7 100644 --- a/lib/crowdb-access-s3/src/metadata.rs +++ b/lib/crowdb-access-s3/src/metadata.rs @@ -5,6 +5,7 @@ mod key; mod multipart; +mod multipart_repository; mod namespace; mod record; mod store; @@ -23,6 +24,7 @@ mod generated { pub use key::{BucketId, MetadataKey, MetadataKeyError, TenantId}; pub use multipart::{MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord}; +pub use multipart_repository::{MultipartRepository, MultipartRepositoryError}; pub use namespace::{BucketDeleteOutcome, BucketNamespace, BucketNamespaceError}; pub use record::{BucketNameRecord, MetadataRecordError, ObjectRecord}; pub use store::{ChunkKvMetadataStore, MetadataStoreError, PutIfAbsentOutcome}; diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs new file mode 100644 index 00000000..99958437 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -0,0 +1,222 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! CAS-backed multipart authority in the S3 Chunk-KV namespace. + +use std::sync::Arc; + +use super::{ + ChunkKvMetadataStore, MetadataKey, MetadataKeyError, MetadataStoreError, MultipartPartRecord, + MultipartPhase, MultipartRecordError, MultipartSessionRecord, PutIfAbsentOutcome, TenantId, +}; + +#[derive(Debug, thiserror::Error)] +pub enum MultipartRepositoryError { + #[error(transparent)] + Key(#[from] MetadataKeyError), + #[error(transparent)] + Record(#[from] MultipartRecordError), + #[error(transparent)] + Store(#[from] MetadataStoreError), + #[error("multipart operation conflicts with durable state")] + Conflict, +} + +pub struct MultipartRepository { + store: Arc, + tenant: TenantId, +} + +impl MultipartRepository { + #[must_use] + pub fn new(store: Arc, tenant: TenantId) -> Self { + Self { store, tenant } + } + + /// Creates one upload or confirms the exact record after a lost reply. + /// + /// # Errors + /// Rejects a different record at the same upload key or unavailable store. + pub async fn begin( + &self, + session: &MultipartSessionRecord, + ) -> Result { + if session.phase != MultipartPhase::Open || session.revision != 1 { + return Err(MultipartRepositoryError::Conflict); + } + let key = self.session_key(session)?; + let value = session.encode()?; + let result = self.store.put_if_absent(key, value.clone()).await; + match result { + Ok(PutIfAbsentOutcome::Inserted { .. }) => Ok(session.clone()), + Ok(PutIfAbsentOutcome::Existing(existing)) if existing.value == value => Ok(session.clone()), + Ok(PutIfAbsentOutcome::Existing(_)) => Err(MultipartRepositoryError::Conflict), + Err(error) => { + if self.load(session).await?.as_ref() == Some(session) { + Ok(session.clone()) + } else { + Err(error.into()) + } + } + } + } + + /// Reads one upload through its exact bucket/object/upload identity. + /// + /// # Errors + /// Rejects corrupt or foreign values and storage failures. + pub async fn load( + &self, + identity: &MultipartSessionRecord, + ) -> Result, MultipartRepositoryError> { + let key = self.session_key(identity)?; + self.store + .get(key) + .await? + .map(|value| { + MultipartSessionRecord::decode( + &value.value, + identity.bucket_id, + &identity.object_key, + &identity.upload_id, + ) + .map_err(Into::into) + }) + .transpose() + } + + /// Moves one session phase under an exact-value CAS fence. + /// + /// # Errors + /// Rejects invalid revisions, changed identity and uncertain storage + /// writes that cannot be confirmed by a matching read. + pub async fn exchange( + &self, + previous: &MultipartSessionRecord, + next: &MultipartSessionRecord, + ) -> Result { + if previous.bucket_id != next.bucket_id + || previous.object_key != next.object_key + || previous.upload_id != next.upload_id + || previous.revision.checked_add(1) != Some(next.revision) + { + return Err(MultipartRepositoryError::Conflict); + } + let key = self.session_key(previous)?; + let expected = previous.encode()?; + let value = next.encode()?; + match self.store.compare_exchange(key, expected, value).await { + Ok(true) => Ok(true), + Ok(false) => Ok(self.load(next).await?.as_ref() == Some(next)), + Err(error) => { + if self.load(next).await?.as_ref() == Some(next) { + Ok(true) + } else { + Err(error.into()) + } + } + } + } + + /// Reads one current part generation. + /// + /// # Errors + /// Rejects corrupt or foreign records and storage failures. + pub async fn part( + &self, + session: &MultipartSessionRecord, + number: u16, + ) -> Result, MultipartRepositoryError> { + let key = MetadataKey::multipart_part(&self.tenant, session.bucket_id, &session.upload_id, number)?; + self.store + .get(key) + .await? + .map(|value| { + MultipartPartRecord::decode(&value.value, session.bucket_id, &session.upload_id, number) + .map_err(Into::into) + }) + .transpose() + } + + /// Conditionally publishes an independently streamed part generation. + /// + /// Distinct part numbers write independent keys. The conservative + /// `max_parts * max_part_bytes` bound prevents aggregate staged bytes from + /// exceeding the session budget even when all parts arrive concurrently. + /// + /// # Errors + /// Rejects closed, expired or foreign sessions, invalid parts and + /// unconfirmed storage errors. A competing writer returns `None`. + pub async fn put_stream_part( + &self, + session: &MultipartSessionRecord, + part: &MultipartPartRecord, + now_ms: u64, + ) -> Result, MultipartRepositoryError> { + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.phase != MultipartPhase::Open + || now_ms < current.created_ms + || now_ms >= current.expires_ms + || u64::from(current.max_parts) + .checked_mul(current.max_part_bytes) + .map_or(true, |bytes| bytes > current.max_staged_bytes) + || part.bucket_id != current.bucket_id + || part.upload_id != current.upload_id + || part.length > current.max_part_bytes + || part.number > current.max_parts + { + return Err(MultipartRepositoryError::Conflict); + } + let before = self.part(¤t, part.number).await?; + let mut after = part.clone(); + after.revision = before + .as_ref() + .map_or(Some(1), |before| before.revision.checked_add(1)) + .ok_or(MultipartRepositoryError::Conflict)?; + after.modified_ms = now_ms; + let value = after.encode()?; + let key = + MetadataKey::multipart_part(&self.tenant, current.bucket_id, ¤t.upload_id, part.number)?; + let outcome = if let Some(before) = &before { + self.store + .compare_exchange(key, before.encode()?, value.clone()) + .await + .map(|applied| applied.then_some(after.clone())) + } else { + self.store + .put_if_absent(key, value.clone()) + .await + .map(|outcome| match outcome { + PutIfAbsentOutcome::Inserted { .. } => Some(after.clone()), + PutIfAbsentOutcome::Existing(existing) if existing.value == value => Some(after.clone()), + PutIfAbsentOutcome::Existing(_) => None, + }) + }; + match outcome { + Ok(Some(result)) => Ok(Some(result)), + Ok(None) => Ok(self + .part(¤t, part.number) + .await? + .filter(|existing| existing == &after)), + Err(error) => { + if self.part(¤t, part.number).await?.as_ref() == Some(&after) { + Ok(Some(after)) + } else { + Err(error.into()) + } + } + } + } + + fn session_key(&self, session: &MultipartSessionRecord) -> Result, MetadataKeyError> { + MetadataKey::multipart_session( + &self.tenant, + session.bucket_id, + &session.object_key, + &session.upload_id, + ) + } +} diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs new file mode 100644 index 00000000..62902dd1 --- /dev/null +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -0,0 +1,268 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::HashMap; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::Arc; +use std::time::Duration; + +use async_trait::async_trait; +use crowdb_access_multipart::SelectedPart; +use crowdb_access_s3::metadata::{ + BucketId, ChunkKvMetadataStore, MultipartPartRecord, MultipartPhase, MultipartRepository, + MultipartRepositoryError, MultipartSessionRecord, TenantId, +}; +use crowdb_chunk_kv_client::{ + ChunkKvClient, ChunkKvRangeCatalogSource, ChunkKvTransport, ClientConfig, ClientError, Result, +}; +use crowdb_protocol::chunk_kv::{ + ChunkKvRangeCatalogEntry, ChunkKvRangeCatalogHead, ChunkKvRangeCatalogPage, ChunkKvRangeCatalogPageRef, + ChunkKvRangeCatalogPartitionState, ChunkKvResponse, Id128, KeyRange, OperationResult, OwnerDescriptor, + PartitionArtifact, PointOperation, PointRequest, RpcCompareCondition, RpcValue, +}; +use crowdb_protocol::chunk_stream::StreamName; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; +use tokio::sync::{mpsc, oneshot}; + +struct Catalog(ChunkKvRangeCatalogHead, Vec); + +#[async_trait] +impl ChunkKvRangeCatalogSource for Catalog { + async fn load(&self) -> Result<(ChunkKvRangeCatalogHead, Vec)> { + Ok((self.0.clone(), self.1.clone())) + } +} + +struct ActorTransport(mpsc::UnboundedSender<(PointOperation, oneshot::Sender>)>); + +#[async_trait] +impl ChunkKvTransport for ActorTransport { + async fn point(&self, _: &str, request: &PointRequest) -> Result { + let (response, receiver) = oneshot::channel(); + self.0 + .send((request.operation.clone(), response)) + .expect("actor is running"); + receiver.await.expect("actor replies") + } + + async fn seek(&self, _: &str, _: &crowdb_protocol::chunk_kv::SeekRequest) -> Result { + unreachable!() + } + + async fn scan(&self, _: &str, _: &crowdb_protocol::chunk_kv::ScanRequest) -> Result { + unreachable!() + } +} + +fn reply(operation: PointOperation, values: &mut HashMap, RpcValue>) -> ChunkKvResponse { + let result = match operation { + PointOperation::Get { key } => OperationResult::Value(values.get(&key).cloned()), + PointOperation::PutIfAbsent { key, value } => { + let observed = values.get(&key).cloned(); + let applied = observed.is_none(); + if applied { + values.insert( + key.clone(), + RpcValue { + key, + value, + revision: 1, + }, + ); + } + OperationResult::Mutation { + applied, + revision: applied.then_some(1), + observed, + } + } + PointOperation::CompareExchange { + key, + condition: RpcCompareCondition::Value(expected), + value, + } => { + let observed = values.get(&key).cloned(); + let applied = observed + .as_ref() + .is_some_and(|observed| observed.value == expected); + let revision = observed.as_ref().map_or(1, |observed| observed.revision + 1); + if applied { + values.insert(key.clone(), RpcValue { key, value, revision }); + } + OperationResult::Mutation { + applied, + revision: applied.then_some(revision), + observed, + } + } + other => panic!("unexpected operation: {other:?}"), + }; + ChunkKvResponse { + map_revision: 1, + journal_position: None, + result: Ok(result), + } +} + +async fn repository() -> (MultipartRepository, Arc) { + let (sender, mut receiver) = + mpsc::unbounded_channel::<(PointOperation, oneshot::Sender>)>(); + let lose_reply = Arc::new(AtomicBool::new(false)); + let actor_lose_reply = Arc::clone(&lose_reply); + tokio::spawn(async move { + let mut values = HashMap::new(); + while let Some((operation, response)) = receiver.recv().await { + let is_mutation = !matches!(operation, PointOperation::Get { .. }); + let result = reply(operation, &mut values); + if is_mutation && actor_lose_reply.swap(false, Ordering::SeqCst) { + let _ = response.send(Err(ClientError::Transport("committed reply lost".into()))); + } else { + let _ = response.send(Ok(result)); + } + } + }); + let mut page = ChunkKvRangeCatalogPage { + generation: 1, + page_index: 0, + entries: vec![ChunkKvRangeCatalogEntry { + partition_id: Id128 { high: 1, low: 1 }, + range: KeyRange { + start: Vec::new(), + end: None, + }, + owner: OwnerDescriptor { + instance_id: 1, + rpc_endpoint: "owner".into(), + }, + owner_epoch: 1, + state: ChunkKvRangeCatalogPartitionState::Serving, + artifact: PartitionArtifact { + tree_id: 1, + stream_name: StreamName { high: 1, low: 1 }, + tail_overlay: None, + }, + transition_id: None, + }], + checksum: [0; 32], + }; + page.seal().unwrap(); + let mut head = ChunkKvRangeCatalogHead { + generation: 1, + previous_generation: None, + pages: vec![ChunkKvRangeCatalogPageRef { + page_generation: 1, + page_index: 0, + first_key: Vec::new(), + page_checksum: page.checksum, + }], + checksum: [0; 32], + }; + head.seal().unwrap(); + let client = Arc::new( + ChunkKvClient::new( + ClientConfig { + operation_timeout: Duration::from_secs(1), + retry_backoff: Duration::from_millis(1), + ..ClientConfig::default() + }, + Arc::new(Catalog(head, vec![page])), + Arc::new(ActorTransport(sender)), + ) + .unwrap(), + ); + client.refresh_catalog().await.unwrap(); + ( + MultipartRepository::new( + Arc::new(ChunkKvMetadataStore::new(client)), + TenantId::new(b"tenant".to_vec()).unwrap(), + ), + lose_reply, + ) +} + +fn session() -> MultipartSessionRecord { + MultipartSessionRecord { + bucket_id: BucketId::new([3; 16]), + object_key: b"object".to_vec(), + upload_id: [7; 16], + revision: 1, + phase: MultipartPhase::Open, + created_ms: 100, + expires_ms: 200, + content_type: "application/octet-stream".into(), + max_parts: 10, + max_part_bytes: 100, + max_object_bytes: 500, + max_staged_bytes: 1_000, + part_count: 0, + staged_bytes: 0, + selection: None, + etag: None, + } +} + +fn part() -> MultipartPartRecord { + MultipartPartRecord { + bucket_id: BucketId::new([3; 16]), + upload_id: [7; 16], + number: 1, + revision: 1, + modified_ms: 0, + length: 5, + raw_md5: [9; 16], + locations: vec![Location { + chunk_id: Some(ChunkId { high: 1, low: 2 }), + offset: 10, + length: 39, + logical_offset: 0, + logical_length: 5, + }], + } +} + +#[tokio::test] +async fn session_cas_and_independent_part_replacement_obey_the_freeze() { + let (repository, lose_reply) = repository().await; + let session = session(); + assert_eq!(repository.begin(&session).await.unwrap(), session); + assert_eq!(repository.begin(&session).await.unwrap(), session); + let mut changed = session.clone(); + changed.content_type = "text/plain".into(); + assert!(matches!( + repository.begin(&changed).await, + Err(MultipartRepositoryError::Conflict) + )); + + let first = repository + .put_stream_part(&session, &part(), 110) + .await + .unwrap() + .unwrap(); + assert_eq!(first.revision, 1); + let second = repository + .put_stream_part(&session, &part(), 111) + .await + .unwrap() + .unwrap(); + assert_eq!(second.revision, 2); + assert_eq!(repository.part(&session, 1).await.unwrap(), Some(second)); + + let mut frozen = session.clone(); + frozen.revision = 2; + frozen.phase = MultipartPhase::Completing; + frozen.part_count = 1; + frozen.staged_bytes = 5; + frozen.selection = Some(vec![SelectedPart { + number: 1, + revision: 2, + digest: [4; 32], + }]); + lose_reply.store(true, Ordering::SeqCst); + assert!(repository.exchange(&session, &frozen).await.unwrap()); + assert!(repository.exchange(&session, &frozen).await.unwrap()); + assert!(matches!( + repository.put_stream_part(&session, &part(), 112).await, + Err(MultipartRepositoryError::Conflict) + )); +} From 9a3a91b0b4636e38dd7973b01845182ac05e31f6 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 01:51:59 +0800 Subject: [PATCH 05/74] Fence and publish S3 multipart generations --- doc/working/plan-s3-multipart.md | 11 +- lib/crowdb-access-s3/src/integrity.rs | 29 +++ lib/crowdb-access-s3/src/metadata.rs | 2 +- lib/crowdb-access-s3/src/metadata/key.rs | 24 +++ .../src/metadata/multipart.rs | 37 +++- .../src/metadata/multipart_repository.rs | 63 ++++++ .../multipart_repository/completion.rs | 143 ++++++++++++ .../multipart_repository/publication.rs | 132 ++++++++++++ lib/crowdb-access-s3/src/metadata/record.rs | 4 + lib/crowdb-access-s3/src/retrieval.rs | 4 +- lib/crowdb-access-s3/tests/integrity_test.rs | 16 ++ .../tests/metadata_key_test.rs | 2 + .../tests/metadata_multipart_test.rs | 6 + .../tests/multipart_repository_test.rs | 204 +++++++++++++++++- 14 files changed, 662 insertions(+), 15 deletions(-) create mode 100644 lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs create mode 100644 lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 8b2fe580..c9f300d5 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -32,8 +32,9 @@ Iceberg multipart path. as namespace scope; preserve immutable part data after replacement. Versioned session/part records and ordered, binary-safe keys are in place; CAS-backed begin, phase transition and part replacement now use exact-value - confirmation after lost replies. Completion snapshots, HTTP wiring and - cleanup state remain. + confirmation after lost replies. Completion snapshots and a predecessor-fenced + metadata-only object publication path are in place. HTTP wiring and cleanup + state remain. - [ ] **S3 routes and wire**: classify create/upload/list/complete/abort/list uploads, parse bounded completion XML, emit compatible responses and errors. Preserve SigV4 authentication and existing basic routes. @@ -41,7 +42,11 @@ Iceberg multipart path. budget, persist part location/integrity before success, reconcile lost replies. - [ ] **Atomic completion**: fence selected part generations, validate order, count, size and checksum, compose locations through the shared core, and - publish one immutable object generation without reading part bytes. + publish one immutable object generation without reading part bytes. The S3 + adapter now freezes selection under session CAS, validates it again before + object-key CAS, and confirms exact publication after response loss. An + immutable generation records preserve selected bytes across a concurrent + part-number replacement. The HTTP path and end-to-end publication test remain. - [ ] **Abort and expiry**: make terminal states idempotent, queue unreachable private part data for bounded cleanup, and protect active/read-pinned data. diff --git a/lib/crowdb-access-s3/src/integrity.rs b/lib/crowdb-access-s3/src/integrity.rs index 20839097..550369f1 100644 --- a/lib/crowdb-access-s3/src/integrity.rs +++ b/lib/crowdb-access-s3/src/integrity.rs @@ -112,3 +112,32 @@ impl SinglePartIntegrity { Ok(result) } } + +/// Encodes a composite multipart `ETag` as a distinct 18-byte metadata +/// checksum marker: 16 digest bytes followed by the part count. +#[must_use] +pub fn multipart_checksum_marker(etag: &str) -> Option<[u8; 18]> { + let (digest, count) = etag.split_once('-')?; + if digest.len() != 32 || count.starts_with('0') { + return None; + } + let count: u16 = count.parse().ok()?; + if count == 0 || count > 10_000 { + return None; + } + let mut marker = [0_u8; 18]; + for (byte, pair) in marker[..16].iter_mut().zip(digest.as_bytes().chunks_exact(2)) { + let pair = std::str::from_utf8(pair).ok()?; + if pair.bytes().any(|byte| byte.is_ascii_uppercase()) { + return None; + } + *byte = u8::from_str_radix(pair, 16).ok()?; + } + marker[16..].copy_from_slice(&count.to_be_bytes()); + Some(marker) +} + +#[must_use] +pub fn is_multipart_checksum(checksum: &[u8], etag: &str) -> bool { + checksum.len() == 18 && multipart_checksum_marker(etag).is_some_and(|marker| checksum == marker) +} diff --git a/lib/crowdb-access-s3/src/metadata.rs b/lib/crowdb-access-s3/src/metadata.rs index d78285d7..edee280f 100644 --- a/lib/crowdb-access-s3/src/metadata.rs +++ b/lib/crowdb-access-s3/src/metadata.rs @@ -24,7 +24,7 @@ mod generated { pub use key::{BucketId, MetadataKey, MetadataKeyError, TenantId}; pub use multipart::{MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord}; -pub use multipart_repository::{MultipartRepository, MultipartRepositoryError}; +pub use multipart_repository::{CompletionPart, MultipartRepository, MultipartRepositoryError}; pub use namespace::{BucketDeleteOutcome, BucketNamespace, BucketNamespaceError}; pub use record::{BucketNameRecord, MetadataRecordError, ObjectRecord}; pub use store::{ChunkKvMetadataStore, MetadataStoreError, PutIfAbsentOutcome}; diff --git a/lib/crowdb-access-s3/src/metadata/key.rs b/lib/crowdb-access-s3/src/metadata/key.rs index 3c4657e8..8942ed44 100644 --- a/lib/crowdb-access-s3/src/metadata/key.rs +++ b/lib/crowdb-access-s3/src/metadata/key.rs @@ -7,6 +7,7 @@ const BUCKET_NAME_KIND: u8 = 1; const OBJECT_KIND: u8 = 2; const MULTIPART_SESSION_KIND: u8 = 3; const MULTIPART_PART_KIND: u8 = 4; +const MULTIPART_PART_GENERATION_KIND: u8 = 5; const MAX_KEY_BYTES: usize = 1024; /// A tenant namespace identity. @@ -223,6 +224,29 @@ impl MetadataKey { key.extend_from_slice(&number.to_be_bytes()); Ok(key) } + + /// Identifies an immutable generation retained after part-number replacement. + /// + /// # Errors + /// Rejects invalid part numbers or a zero revision. + pub fn multipart_part_generation( + tenant: &TenantId, + bucket: BucketId, + upload_id: &[u8; 16], + number: u16, + revision: u64, + ) -> Result, MetadataKeyError> { + if number == 0 || number > 10_000 || revision == 0 { + return Err(MetadataKeyError::InvalidPartNumber); + } + let mut key = namespace_prefix(tenant); + key.extend_from_slice(bucket.as_bytes()); + key.push(MULTIPART_PART_GENERATION_KIND); + key.extend_from_slice(upload_id); + key.extend_from_slice(&number.to_be_bytes()); + key.extend_from_slice(&revision.to_be_bytes()); + Ok(key) + } } #[derive(Clone, Debug, Eq, PartialEq)] diff --git a/lib/crowdb-access-s3/src/metadata/multipart.rs b/lib/crowdb-access-s3/src/metadata/multipart.rs index 0aca3a1f..c9b3fd23 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart.rs @@ -7,6 +7,7 @@ use bincode::Options as _; use crowdb_access_multipart::{validate_selected_parts, MultipartComposer, SelectedPart}; use crowdb_protocol::chunkdb::rpc::Location; use serde::{Deserialize, Serialize}; +use sha2::{Digest as _, Sha256}; use super::BucketId; @@ -42,6 +43,9 @@ pub struct MultipartSessionRecord { pub part_count: u16, pub staged_bytes: u64, pub selection: Option>, + pub completion_request_digest: Option<[u8; 32]>, + pub publication_ms: Option, + pub object_predecessor: Option>, pub etag: Option, } @@ -120,10 +124,31 @@ impl MultipartSessionRecord { } let selected_count = self.selection.as_ref().map_or(0, Vec::len); match self.phase { - MultipartPhase::Open if self.selection.is_none() && self.etag.is_none() => Ok(()), - MultipartPhase::Completing if self.selection.is_some() && self.etag.is_none() => Ok(()), + MultipartPhase::Open + if self.selection.is_none() + && self.completion_request_digest.is_none() + && self.publication_ms.is_none() + && self.object_predecessor.is_none() + && self.etag.is_none() => + { + Ok(()) + } + MultipartPhase::Completing + if self.selection.is_some() + && self.completion_request_digest.is_some() + && self.publication_ms.is_none() + && self.object_predecessor.is_none() + && self.etag.is_none() => + { + Ok(()) + } MultipartPhase::Publishing | MultipartPhase::Published if self.selection.is_some() + && self.completion_request_digest.is_some() + && self + .publication_ms + .is_some_and(|time| time >= self.created_ms && time < self.expires_ms) + && self.object_predecessor.is_some() && self .etag .as_ref() @@ -138,6 +163,14 @@ impl MultipartSessionRecord { } impl MultipartPartRecord { + /// Binds a completion selection to this exact persisted part generation. + /// + /// # Errors + /// Rejects an invalid part record before calculating its identity. + pub fn selection_digest(&self) -> Result<[u8; 32], MultipartRecordError> { + Ok(Sha256::digest(self.encode()?).into()) + } + /// Encodes one immutable selected part generation. /// /// # Errors diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index 99958437..5bdcfdf8 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -10,6 +10,11 @@ use super::{ MultipartPhase, MultipartRecordError, MultipartSessionRecord, PutIfAbsentOutcome, TenantId, }; +mod completion; +mod publication; + +pub use completion::CompletionPart; + #[derive(Debug, thiserror::Error)] pub enum MultipartRepositoryError { #[error(transparent)] @@ -20,6 +25,10 @@ pub enum MultipartRepositoryError { Store(#[from] MetadataStoreError), #[error("multipart operation conflicts with durable state")] Conflict, + #[error("multipart completion references a missing or changed part")] + InvalidPart, + #[error("a nonfinal multipart part is smaller than 5 MiB")] + EntityTooSmall, } pub struct MultipartRepository { @@ -138,6 +147,37 @@ impl MultipartRepository { .transpose() } + /// Reads one immutable part generation, including replaced generations. + /// + /// # Errors + /// Rejects corrupt or foreign generation bytes and unavailable storage. + pub async fn part_generation( + &self, + session: &MultipartSessionRecord, + number: u16, + revision: u64, + ) -> Result, MultipartRepositoryError> { + let key = MetadataKey::multipart_part_generation( + &self.tenant, + session.bucket_id, + &session.upload_id, + number, + revision, + )?; + self.store + .get(key) + .await? + .map(|value| { + let part = + MultipartPartRecord::decode(&value.value, session.bucket_id, &session.upload_id, number)?; + if part.revision != revision { + return Err(MultipartRepositoryError::InvalidPart); + } + Ok(part) + }) + .transpose() + } + /// Conditionally publishes an independently streamed part generation. /// /// Distinct part numbers write independent keys. The conservative @@ -178,6 +218,29 @@ impl MultipartRepository { .ok_or(MultipartRepositoryError::Conflict)?; after.modified_ms = now_ms; let value = after.encode()?; + let generation_key = MetadataKey::multipart_part_generation( + &self.tenant, + current.bucket_id, + ¤t.upload_id, + part.number, + after.revision, + )?; + let generation_write = self.store.put_if_absent(generation_key, value.clone()).await; + match generation_write { + Ok(PutIfAbsentOutcome::Inserted { .. }) => {} + Ok(PutIfAbsentOutcome::Existing(existing)) if existing.value == value => {} + Ok(PutIfAbsentOutcome::Existing(_)) => return Ok(None), + Err(error) => { + if self + .part_generation(¤t, part.number, after.revision) + .await? + .as_ref() + != Some(&after) + { + return Err(error.into()); + } + } + } let key = MetadataKey::multipart_part(&self.tenant, current.bucket_id, ¤t.upload_id, part.number)?; let outcome = if let Some(before) = &before { diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs new file mode 100644 index 00000000..293f1d76 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs @@ -0,0 +1,143 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Freeze the exact selected part generations before object publication. + +use crowdb_access_multipart::{validate_selected_parts, MultipartComposer, SelectedPart}; +use sha2::{Digest as _, Sha256}; + +use super::{MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord}; + +const MIN_NONFINAL_PART_BYTES: u64 = 5 * 1024 * 1024; + +/// One S3 `CompleteMultipartUpload` part reference in request order. +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct CompletionPart { + pub number: u16, + pub etag: String, +} + +impl MultipartRepository { + /// Validates and freezes an ordered part selection under the session CAS. + /// + /// The part records can still race with this phase write. Publication + /// rechecks every recorded digest and refuses changed generations. + /// + /// # Errors + /// Rejects missing, duplicate, undersized or mismatched parts and + /// conflicting/expired uploads. No object is published by this step. + pub async fn freeze_completion( + &self, + session: &MultipartSessionRecord, + requested: &[CompletionPart], + now_ms: u64, + ) -> Result, MultipartRepositoryError> { + let request_digest = request_digest(requested)?; + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if matches!( + current.phase, + MultipartPhase::Publishing | MultipartPhase::Published + ) && current.completion_request_digest == Some(request_digest) + { + return Ok(Some(current)); + } + if current != *session + || current.phase != MultipartPhase::Open + || now_ms < current.created_ms + || now_ms >= current.expires_ms + || requested.is_empty() + || requested.len() > usize::from(current.max_parts) + { + return Err(MultipartRepositoryError::Conflict); + } + let mut composer = MultipartComposer::new(current.max_object_bytes); + let mut selection = Vec::with_capacity(requested.len()); + let mut previous = 0; + for (index, request) in requested.iter().enumerate() { + if request.number <= previous || request.number > current.max_parts { + return Err(MultipartRepositoryError::InvalidPart); + } + previous = request.number; + let part = self + .part(¤t, request.number) + .await? + .ok_or(MultipartRepositoryError::InvalidPart)?; + if index + 1 < requested.len() && part.length < MIN_NONFINAL_PART_BYTES { + return Err(MultipartRepositoryError::EntityTooSmall); + } + if !etag_matches(&request.etag, &part.raw_md5) { + return Err(MultipartRepositoryError::InvalidPart); + } + composer + .push(part.length, part.raw_md5, &part.locations) + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + selection.push(SelectedPart { + number: part.number, + revision: part.revision, + digest: part.selection_digest()?, + }); + } + validate_selected_parts(&selection, current.max_parts) + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + let assembled = composer + .finish() + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + let mut next = current.clone(); + next.revision = current + .revision + .checked_add(1) + .ok_or(MultipartRepositoryError::Conflict)?; + next.phase = MultipartPhase::Publishing; + next.part_count = + u16::try_from(selection.len()).map_err(|_| MultipartRepositoryError::InvalidPart)?; + next.staged_bytes = assembled.length; + next.selection = Some(selection); + next.completion_request_digest = Some(request_digest); + next.publication_ms = Some(now_ms); + let object_key = super::MetadataKey::object(&self.tenant, current.bucket_id, ¤t.object_key)?; + next.object_predecessor = Some( + self.store + .get(object_key) + .await? + .map(|value| Sha256::digest(value.value).into()), + ); + next.etag = Some(assembled.etag); + Ok(self.exchange(¤t, &next).await?.then_some(next)) + } +} + +fn etag_matches(value: &str, raw_md5: &[u8; 16]) -> bool { + parse_raw_md5(value).is_some_and(|actual| &actual == raw_md5) +} + +fn request_digest(requested: &[CompletionPart]) -> Result<[u8; 32], MultipartRepositoryError> { + let mut sha = Sha256::new(); + sha.update( + u16::try_from(requested.len()) + .map_err(|_| MultipartRepositoryError::InvalidPart)? + .to_be_bytes(), + ); + for part in requested { + sha.update(part.number.to_be_bytes()); + sha.update(parse_raw_md5(&part.etag).ok_or(MultipartRepositoryError::InvalidPart)?); + } + Ok(sha.finalize().into()) +} + +fn parse_raw_md5(value: &str) -> Option<[u8; 16]> { + let unquoted = value + .strip_prefix('"') + .and_then(|value| value.strip_suffix('"')) + .unwrap_or(value); + if unquoted.len() != 32 { + return None; + } + let mut raw = [0_u8; 16]; + for (output, pair) in raw.iter_mut().zip(unquoted.as_bytes().chunks_exact(2)) { + *output = u8::from_str_radix(std::str::from_utf8(pair).ok()?, 16).ok()?; + } + Some(raw) +} diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs new file mode 100644 index 00000000..0f9cffe9 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs @@ -0,0 +1,132 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Publish the frozen multipart object through one object-key mutation. + +use crowdb_access_multipart::MultipartComposer; +use sha2::{Digest as _, Sha256}; + +use super::{ + MetadataKey, MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, + PutIfAbsentOutcome, +}; +use crate::integrity::multipart_checksum_marker; +use crate::metadata::ObjectRecord; + +impl MultipartRepository { + /// Publishes selected durable part locations without rereading part bytes. + /// + /// # Errors + /// Rejects changed part generations or object predecessors and unconfirmed + /// storage failures. A retry confirms only an exact published generation. + pub async fn publish_completion( + &self, + session: &MultipartSessionRecord, + ) -> Result { + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.phase == MultipartPhase::Published { + return Ok(current); + } + if current.phase != MultipartPhase::Publishing { + return Err(MultipartRepositoryError::Conflict); + } + let selection = current + .selection + .as_ref() + .ok_or(MultipartRepositoryError::Conflict)?; + let mut composer = MultipartComposer::new(current.max_object_bytes); + for selected in selection { + let part = self + .part_generation(¤t, selected.number, selected.revision) + .await? + .ok_or(MultipartRepositoryError::InvalidPart)?; + if part.revision != selected.revision || part.selection_digest()? != selected.digest { + return Err(MultipartRepositoryError::InvalidPart); + } + composer + .push(part.length, part.raw_md5, &part.locations) + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + } + let assembled = composer + .finish() + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + if assembled.length != current.staged_bytes || Some(&assembled.etag) != current.etag.as_ref() { + return Err(MultipartRepositoryError::InvalidPart); + } + let published_at = current.publication_ms.ok_or(MultipartRepositoryError::Conflict)?; + let marker = multipart_checksum_marker(&assembled.etag).ok_or(MultipartRepositoryError::Conflict)?; + let object = ObjectRecord { + bucket_id: current.bucket_id, + key: current.object_key.clone(), + logical_length: assembled.length, + checksum: marker.to_vec(), + etag: assembled.etag, + created_at_ms: published_at, + modified_at_ms: published_at, + content_type: current.content_type.clone(), + attributes: Vec::new(), + data_reference: bincode::serialize(&assembled.locations) + .map_err(|_| MultipartRepositoryError::Conflict)?, + data_length: assembled.length, + }; + let value = object.encode().map_err(|_| MultipartRepositoryError::Conflict)?; + let key = MetadataKey::object(&self.tenant, current.bucket_id, ¤t.object_key)?; + let observed = self.store.get(key.clone()).await?; + if observed.as_ref().is_some_and(|entry| entry.value == value) { + return self.finish_publication(¤t).await; + } + let expected_digest = current + .object_predecessor + .ok_or(MultipartRepositoryError::Conflict)?; + if observed.as_ref().map(|entry| Sha256::digest(&entry.value).into()) != expected_digest { + return Err(MultipartRepositoryError::Conflict); + } + let mutation = if let Some(before) = observed { + self.store + .compare_exchange(key.clone(), before.value, value.clone()) + .await + } else { + self.store + .put_if_absent(key.clone(), value.clone()) + .await + .map(|outcome| matches!(outcome, PutIfAbsentOutcome::Inserted { .. })) + }; + match mutation { + Ok(true) => {} + Ok(false) => { + if self.store.get(key).await?.as_ref().map(|entry| &entry.value) != Some(&value) { + return Err(MultipartRepositoryError::Conflict); + } + } + Err(error) => { + if self.store.get(key).await?.as_ref().map(|entry| &entry.value) != Some(&value) { + return Err(error.into()); + } + } + } + self.finish_publication(¤t).await + } + + async fn finish_publication( + &self, + current: &MultipartSessionRecord, + ) -> Result { + let mut next = current.clone(); + next.revision = current + .revision + .checked_add(1) + .ok_or(MultipartRepositoryError::Conflict)?; + next.phase = MultipartPhase::Published; + if self.exchange(current, &next).await? { + Ok(next) + } else { + self.load(current) + .await? + .filter(|record| record.phase == MultipartPhase::Published) + .ok_or(MultipartRepositoryError::Conflict) + } + } +} diff --git a/lib/crowdb-access-s3/src/metadata/record.rs b/lib/crowdb-access-s3/src/metadata/record.rs index 814cd6b5..4cac5e5d 100644 --- a/lib/crowdb-access-s3/src/metadata/record.rs +++ b/lib/crowdb-access-s3/src/metadata/record.rs @@ -172,6 +172,10 @@ fn validate_object(record: &ObjectRecord) -> Result<(), MetadataRecordError> { required("etag", record.etag.as_bytes())?; required("content_type", record.content_type.as_bytes())?; required("data_reference", &record.data_reference)?; + if record.checksum.len() == 18 && !crate::integrity::is_multipart_checksum(&record.checksum, &record.etag) + { + return Err(MetadataRecordError::Invalid); + } if record.logical_length != record.data_length { return Err(MetadataRecordError::DataLength); } diff --git a/lib/crowdb-access-s3/src/retrieval.rs b/lib/crowdb-access-s3/src/retrieval.rs index 732e2436..901d6a29 100644 --- a/lib/crowdb-access-s3/src/retrieval.rs +++ b/lib/crowdb-access-s3/src/retrieval.rs @@ -126,7 +126,9 @@ pub fn prepare_get( inner: client .read_range_stream(&locations, interval.start, interval.end) .map_err(|_| RetrievalError::ChunkRead)?, - integrity: requested.is_none().then(SinglePartIntegrity::default), + integrity: (requested.is_none() + && !crate::integrity::is_multipart_checksum(&record.checksum, &record.etag)) + .then(SinglePartIntegrity::default), expected_checksum: record.checksum.clone(), terminal: false, }) diff --git a/lib/crowdb-access-s3/tests/integrity_test.rs b/lib/crowdb-access-s3/tests/integrity_test.rs index 9588e569..c03c729e 100644 --- a/lib/crowdb-access-s3/tests/integrity_test.rs +++ b/lib/crowdb-access-s3/tests/integrity_test.rs @@ -4,6 +4,22 @@ use crowdb_access_s3::integrity::{IntegrityError, SinglePartIntegrity}; use hyper::body::Bytes; +#[test] +fn multipart_marker_binds_composite_digest_and_part_count() { + use crowdb_access_s3::integrity::{is_multipart_checksum, multipart_checksum_marker}; + + let etag = "b4ab393b73e0e71830bf2bf0e63c4d91-2"; + let marker = multipart_checksum_marker(etag).unwrap(); + assert_eq!(marker.len(), 18); + assert_eq!(&marker[16..], &[0, 2]); + assert!(is_multipart_checksum(&marker, etag)); + assert!(!is_multipart_checksum( + &marker, + "b4ab393b73e0e71830bf2bf0e63c4d91-3" + )); + assert!(multipart_checksum_marker("b4ab393b73e0e71830bf2bf0e63c4d91-0").is_none()); +} + #[test] fn single_part_etag_is_independent_of_body_frame_boundaries() { let mut fragmented = SinglePartIntegrity::default(); diff --git a/lib/crowdb-access-s3/tests/metadata_key_test.rs b/lib/crowdb-access-s3/tests/metadata_key_test.rs index 4fc5abd7..ff16a33c 100644 --- a/lib/crowdb-access-s3/tests/metadata_key_test.rs +++ b/lib/crowdb-access-s3/tests/metadata_key_test.rs @@ -80,4 +80,6 @@ fn multipart_keys_keep_uploads_and_parts_in_separate_bounded_intervals() { assert!(MetadataKey::multipart_part(&tenant, bucket, &upload, 10_001).is_err()); assert!(MetadataKey::object_end(&tenant, bucket) <= start); assert!(end <= part_start); + let generation = MetadataKey::multipart_part_generation(&tenant, bucket, &upload, 1, 2).unwrap(); + assert!(part_end < generation); } diff --git a/lib/crowdb-access-s3/tests/metadata_multipart_test.rs b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs index 810575b7..fe46fa54 100644 --- a/lib/crowdb-access-s3/tests/metadata_multipart_test.rs +++ b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs @@ -25,6 +25,9 @@ fn session() -> MultipartSessionRecord { part_count: 0, staged_bytes: 0, selection: None, + completion_request_digest: None, + publication_ms: None, + object_predecessor: None, etag: None, } } @@ -85,6 +88,9 @@ fn completed_session_requires_a_matching_selected_count_and_etag() { revision: 2, digest: [4; 32], }]); + session.completion_request_digest = Some([5; 32]); + session.publication_ms = Some(150); + session.object_predecessor = Some(None); session.etag = Some("11111111111111111111111111111111-2".into()); assert_eq!(session.encode(), Err(MultipartRecordError::Invalid)); session.etag = Some("11111111111111111111111111111111-1".into()); diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs index 62902dd1..7ebe56ca 100644 --- a/lib/crowdb-access-s3/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -9,8 +9,8 @@ use std::time::Duration; use async_trait::async_trait; use crowdb_access_multipart::SelectedPart; use crowdb_access_s3::metadata::{ - BucketId, ChunkKvMetadataStore, MultipartPartRecord, MultipartPhase, MultipartRepository, - MultipartRepositoryError, MultipartSessionRecord, TenantId, + BucketId, ChunkKvMetadataStore, CompletionPart, MetadataKey, MultipartPartRecord, MultipartPhase, + MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, ObjectRecord, TenantId, }; use crowdb_chunk_kv_client::{ ChunkKvClient, ChunkKvRangeCatalogSource, ChunkKvTransport, ClientConfig, ClientError, Result, @@ -105,7 +105,7 @@ fn reply(operation: PointOperation, values: &mut HashMap, RpcValue>) -> } } -async fn repository() -> (MultipartRepository, Arc) { +async fn repository() -> (MultipartRepository, Arc, Arc) { let (sender, mut receiver) = mpsc::unbounded_channel::<(PointOperation, oneshot::Sender>)>(); let lose_reply = Arc::new(AtomicBool::new(false)); @@ -172,12 +172,11 @@ async fn repository() -> (MultipartRepository, Arc) { .unwrap(), ); client.refresh_catalog().await.unwrap(); + let store = Arc::new(ChunkKvMetadataStore::new(client)); ( - MultipartRepository::new( - Arc::new(ChunkKvMetadataStore::new(client)), - TenantId::new(b"tenant".to_vec()).unwrap(), - ), + MultipartRepository::new(Arc::clone(&store), TenantId::new(b"tenant".to_vec()).unwrap()), lose_reply, + store, ) } @@ -198,6 +197,9 @@ fn session() -> MultipartSessionRecord { part_count: 0, staged_bytes: 0, selection: None, + completion_request_digest: None, + publication_ms: None, + object_predecessor: None, etag: None, } } @@ -223,7 +225,7 @@ fn part() -> MultipartPartRecord { #[tokio::test] async fn session_cas_and_independent_part_replacement_obey_the_freeze() { - let (repository, lose_reply) = repository().await; + let (repository, lose_reply, _) = repository().await; let session = session(); assert_eq!(repository.begin(&session).await.unwrap(), session); assert_eq!(repository.begin(&session).await.unwrap(), session); @@ -247,6 +249,10 @@ async fn session_cas_and_independent_part_replacement_obey_the_freeze() { .unwrap(); assert_eq!(second.revision, 2); assert_eq!(repository.part(&session, 1).await.unwrap(), Some(second)); + assert_eq!( + repository.part_generation(&session, 1, 1).await.unwrap(), + Some(first) + ); let mut frozen = session.clone(); frozen.revision = 2; @@ -258,6 +264,8 @@ async fn session_cas_and_independent_part_replacement_obey_the_freeze() { revision: 2, digest: [4; 32], }]); + frozen.completion_request_digest = Some([5; 32]); + frozen.publication_ms = None; lose_reply.store(true, Ordering::SeqCst); assert!(repository.exchange(&session, &frozen).await.unwrap()); assert!(repository.exchange(&session, &frozen).await.unwrap()); @@ -266,3 +274,183 @@ async fn session_cas_and_independent_part_replacement_obey_the_freeze() { Err(MultipartRepositoryError::Conflict) )); } + +#[tokio::test] +async fn completion_freezes_exact_part_revision_and_replays_the_same_request() { + let (repository, _, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + let first = repository + .put_stream_part(&session, &part(), 110) + .await + .unwrap() + .unwrap(); + let requested = [CompletionPart { + number: 1, + etag: format!("\"{}\"", "09".repeat(16)), + }]; + let frozen = repository + .freeze_completion(&session, &requested, 120) + .await + .unwrap() + .unwrap(); + assert_eq!(frozen.phase, MultipartPhase::Publishing); + assert_eq!(frozen.selection.as_ref().unwrap()[0].revision, first.revision); + assert!(frozen.etag.as_ref().unwrap().ends_with("-1")); + assert_eq!( + repository + .freeze_completion(&session, &requested, 121) + .await + .unwrap(), + Some(frozen.clone()) + ); + assert!(matches!( + repository.put_stream_part(&session, &part(), 122).await, + Err(MultipartRepositoryError::Conflict) + )); + let published = repository.publish_completion(&frozen).await.unwrap(); + assert_eq!(published.phase, MultipartPhase::Published); + assert_eq!(repository.publish_completion(&session).await.unwrap(), published); + let key = MetadataKey::object( + &TenantId::new(b"tenant".to_vec()).unwrap(), + session.bucket_id, + &session.object_key, + ) + .unwrap(); + let object = ObjectRecord::decode(&store.get(key).await.unwrap().unwrap().value).unwrap(); + assert_eq!(object.logical_length, 5); + assert_eq!(object.etag, frozen.etag.unwrap()); + assert_eq!(object.checksum.len(), 18); + let locations: Vec = bincode::deserialize(&object.data_reference).unwrap(); + assert_eq!(locations, part().locations); +} + +#[tokio::test] +async fn completion_rejects_undersized_nonfinal_and_wrong_etag() { + let (repository, _, _) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + let mut second = part(); + second.number = 2; + repository.put_stream_part(&session, &second, 111).await.unwrap(); + let etag = "09".repeat(16); + let request = [ + CompletionPart { + number: 1, + etag: etag.clone(), + }, + CompletionPart { number: 2, etag }, + ]; + assert!(matches!( + repository.freeze_completion(&session, &request, 120).await, + Err(MultipartRepositoryError::EntityTooSmall) + )); + let wrong = [CompletionPart { + number: 1, + etag: "00".repeat(16), + }]; + assert!(matches!( + repository.freeze_completion(&session, &wrong, 120).await, + Err(MultipartRepositoryError::InvalidPart) + )); + assert_eq!(repository.load(&session).await.unwrap(), Some(session)); +} + +#[tokio::test] +async fn publication_recovers_a_lost_reply_without_replacing_a_competing_object() { + let (repository, lose_reply, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + let request = [CompletionPart { + number: 1, + etag: "09".repeat(16), + }]; + let frozen = repository + .freeze_completion(&session, &request, 120) + .await + .unwrap() + .unwrap(); + lose_reply.store(true, Ordering::SeqCst); + let published = repository.publish_completion(&frozen).await.unwrap(); + assert_eq!(published.phase, MultipartPhase::Published); + assert_eq!(repository.publish_completion(&frozen).await.unwrap(), published); + + let mut second = session.clone(); + second.upload_id = [8; 16]; + repository.begin(&second).await.unwrap(); + repository + .put_stream_part(&second, &part_for(&second), 130) + .await + .unwrap(); + let frozen_second = repository + .freeze_completion(&second, &request, 140) + .await + .unwrap() + .unwrap(); + let key = MetadataKey::object( + &TenantId::new(b"tenant".to_vec()).unwrap(), + second.bucket_id, + &second.object_key, + ) + .unwrap(); + let previous = store.get(key.clone()).await.unwrap().unwrap(); + assert!(store + .compare_exchange(key.clone(), previous.value, b"competing generation".to_vec()) + .await + .unwrap()); + assert!(matches!( + repository.publish_completion(&frozen_second).await, + Err(MultipartRepositoryError::Conflict) + )); + assert_eq!( + store.get(key).await.unwrap().unwrap().value, + b"competing generation" + ); +} + +#[tokio::test] +async fn frozen_part_generation_survives_a_late_pointer_change() { + let (repository, _, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + let selected = repository + .put_stream_part(&session, &part(), 111) + .await + .unwrap() + .unwrap(); + let request = [CompletionPart { + number: 1, + etag: "09".repeat(16), + }]; + let frozen = repository + .freeze_completion(&session, &request, 120) + .await + .unwrap() + .unwrap(); + let pointer_key = MetadataKey::multipart_part( + &TenantId::new(b"tenant".to_vec()).unwrap(), + session.bucket_id, + &session.upload_id, + 1, + ) + .unwrap(); + let mut late = selected.clone(); + late.revision = 3; + late.locations[0].offset = 77; + let previous = store.get(pointer_key.clone()).await.unwrap().unwrap(); + assert!(store + .compare_exchange(pointer_key, previous.value, late.encode().unwrap()) + .await + .unwrap()); + let published = repository.publish_completion(&frozen).await.unwrap(); + assert_eq!(published.phase, MultipartPhase::Published); +} + +fn part_for(session: &MultipartSessionRecord) -> MultipartPartRecord { + let mut value = part(); + value.upload_id = session.upload_id; + value +} From 5a6ff1b682c6352f1edcf88d2b4491aae49a0de6 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 01:56:18 +0800 Subject: [PATCH 06/74] Share multipart phases and abort transition --- doc/working/plan-s3-multipart.md | 8 +++- .../src/file/multipart.rs | 10 +---- lib/crowdb-access-multipart/src/lib.rs | 5 ++- lib/crowdb-access-multipart/src/state.rs | 11 +++++ .../src/metadata/multipart.rs | 9 +--- .../src/metadata/multipart_repository.rs | 1 + .../metadata/multipart_repository/terminal.rs | 42 +++++++++++++++++++ .../tests/multipart_repository_test.rs | 25 +++++++++++ 8 files changed, 91 insertions(+), 20 deletions(-) create mode 100644 lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index c9f300d5..241a7688 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -22,8 +22,10 @@ Iceberg multipart path. - [~] **Durable transition core**: isolate session/part states, replacement generations, completion selection, abort and recovery transitions from Iceberg catalog-specific keys and records. Keep store CAS and namespace - adaptation in each protocol. Files: shared multipart crate, Iceberg file - repository, S3 metadata store. + adaptation in each protocol. Shared phase vocabulary, selection validation, + accounting and location composition are now used by both adapters; storage + CAS and durable record layouts remain protocol-specific. Files: shared + multipart crate, Iceberg file repository, S3 metadata store. ## S3 adapter and HTTP @@ -49,6 +51,8 @@ Iceberg multipart path. part-number replacement. The HTTP path and end-to-end publication test remain. - [ ] **Abort and expiry**: make terminal states idempotent, queue unreachable private part data for bounded cleanup, and protect active/read-pinned data. + The S3 adapter now has an idempotent, response-loss-safe logical abort; the + durable cleanup queue, expiry scan and read-pin protection remain. ## Acceptance and cleanup diff --git a/lib/crowdb-access-iceberg/src/file/multipart.rs b/lib/crowdb-access-iceberg/src/file/multipart.rs index 0c75546c..2c51e47a 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart.rs @@ -35,15 +35,7 @@ impl MultipartLimits { } } -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum MultipartPhase { - Open, - Completing, - Publishing, - Published, - Aborted, - Conflicted, -} +pub use crowdb_access_multipart::MultipartPhase; #[derive(Clone, Debug, Eq, PartialEq)] pub struct MultipartCompletion { diff --git a/lib/crowdb-access-multipart/src/lib.rs b/lib/crowdb-access-multipart/src/lib.rs index aafad0e8..a494db87 100644 --- a/lib/crowdb-access-multipart/src/lib.rs +++ b/lib/crowdb-access-multipart/src/lib.rs @@ -14,7 +14,10 @@ use md5::{Digest, Md5}; mod state; -pub use state::{reserve_part_accounting, validate_selected_parts, PartAccounting, SelectedPart, StateError}; +pub use state::{ + reserve_part_accounting, validate_selected_parts, MultipartPhase, PartAccounting, SelectedPart, + StateError, +}; #[derive(Debug, thiserror::Error, PartialEq, Eq)] pub enum ComposeError { diff --git a/lib/crowdb-access-multipart/src/state.rs b/lib/crowdb-access-multipart/src/state.rs index 5d48960a..10c91c46 100644 --- a/lib/crowdb-access-multipart/src/state.rs +++ b/lib/crowdb-access-multipart/src/state.rs @@ -3,6 +3,17 @@ //! Protocol-neutral part selection and reservation invariants. +/// Durable multipart transition phase shared by S3 and Iceberg. +#[derive(Clone, Copy, Debug, Eq, PartialEq, serde::Serialize, serde::Deserialize)] +pub enum MultipartPhase { + Open, + Completing, + Publishing, + Published, + Aborted, + Conflicted, +} + #[derive(Clone, Copy, Debug, Eq, PartialEq, serde::Serialize, serde::Deserialize)] pub struct SelectedPart { pub number: u16, diff --git a/lib/crowdb-access-s3/src/metadata/multipart.rs b/lib/crowdb-access-s3/src/metadata/multipart.rs index c9b3fd23..93cfbb3b 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart.rs @@ -17,14 +17,7 @@ const MAX_RECORD_BYTES: u64 = 1024 * 1024; const MAX_OBJECT_KEY_BYTES: usize = 1024; const MAX_CONTENT_TYPE_BYTES: usize = 1024; -#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize)] -pub enum MultipartPhase { - Open, - Completing, - Publishing, - Published, - Aborted, -} +pub use crowdb_access_multipart::MultipartPhase; #[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] pub struct MultipartSessionRecord { diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index 5bdcfdf8..e509c8ef 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -12,6 +12,7 @@ use super::{ mod completion; mod publication; +mod terminal; pub use completion::CompletionPart; diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs new file mode 100644 index 00000000..8dd33943 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs @@ -0,0 +1,42 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Terminal multipart session transitions. + +use super::{MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord}; + +impl MultipartRepository { + /// Logically aborts an open upload, retaining part evidence for cleanup. + /// + /// # Errors + /// Rejects an upload whose completion or publication already won. + pub async fn abort( + &self, + session: &MultipartSessionRecord, + ) -> Result { + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.phase == MultipartPhase::Aborted { + return Ok(current); + } + if current.phase != MultipartPhase::Open { + return Err(MultipartRepositoryError::Conflict); + } + let mut next = current.clone(); + next.revision = current + .revision + .checked_add(1) + .ok_or(MultipartRepositoryError::Conflict)?; + next.phase = MultipartPhase::Aborted; + if self.exchange(¤t, &next).await? { + Ok(next) + } else { + self.load(session) + .await? + .filter(|record| record.phase == MultipartPhase::Aborted) + .ok_or(MultipartRepositoryError::Conflict) + } + } +} diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs index 7ebe56ca..49777526 100644 --- a/lib/crowdb-access-s3/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -454,3 +454,28 @@ fn part_for(session: &MultipartSessionRecord) -> MultipartPartRecord { value.upload_id = session.upload_id; value } + +#[tokio::test] +async fn abort_confirms_lost_reply_and_rejects_part_publication() { + let (repository, lose_reply, _) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + lose_reply.store(true, Ordering::SeqCst); + let aborted = repository.abort(&session).await.unwrap(); + assert_eq!(aborted.phase, MultipartPhase::Aborted); + assert_eq!(repository.abort(&session).await.unwrap(), aborted); + assert!(matches!( + repository.put_stream_part(&session, &part(), 120).await, + Err(MultipartRepositoryError::Conflict) + )); + assert_eq!( + repository + .part_generation(&session, 1, 1) + .await + .unwrap() + .unwrap() + .length, + 5 + ); +} From 4825f4dcfca1d39256e1da45d18ba2172e57aea3 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:02:44 +0800 Subject: [PATCH 07/74] Share multipart admission bounds --- doc/working/plan-s3-multipart.md | 7 +++-- .../src/file/multipart.rs | 13 +++++---- lib/crowdb-access-multipart/src/lib.rs | 4 +-- lib/crowdb-access-multipart/src/state.rs | 20 +++++++++++++ .../tests/state_test.rs | 29 ++++++++++++++++++- .../src/metadata/multipart.rs | 14 +++++---- 6 files changed, 70 insertions(+), 17 deletions(-) diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 241a7688..a543687a 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -22,9 +22,10 @@ Iceberg multipart path. - [~] **Durable transition core**: isolate session/part states, replacement generations, completion selection, abort and recovery transitions from Iceberg catalog-specific keys and records. Keep store CAS and namespace - adaptation in each protocol. Shared phase vocabulary, selection validation, - accounting and location composition are now used by both adapters; storage - CAS and durable record layouts remain protocol-specific. Files: shared + adaptation in each protocol. Shared phase vocabulary, admission bounds, + selection validation, accounting and location composition are now used by + both adapters; storage CAS and durable record layouts remain protocol-specific. + Files: shared multipart crate, Iceberg file repository, S3 metadata store. ## S3 adapter and HTTP diff --git a/lib/crowdb-access-iceberg/src/file/multipart.rs b/lib/crowdb-access-iceberg/src/file/multipart.rs index 2c51e47a..505c17d1 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart.rs @@ -2,6 +2,7 @@ use crate::catalog::CatalogContext; use crate::error::ValidationError; use crate::key::{CatalogScope, FileId, IcebergKey, OperationId}; use crate::operation::PayloadReference; +use crowdb_access_multipart::MultipartBounds; use sha2::{Digest, Sha256}; use std::fmt::Write; @@ -20,11 +21,13 @@ impl MultipartLimits { /// # Errors /// Rejects missing or incoherent independent multipart limits. pub fn validate(self) -> Result<(), ValidationError> { - if self.max_parts == 0 - || self.max_parts > 10_000 - || self.max_part_bytes == 0 - || self.max_part_bytes > self.max_file_bytes - || self.max_file_bytes > self.max_staged_bytes + if !(MultipartBounds { + max_parts: self.max_parts, + max_part_bytes: self.max_part_bytes, + max_object_bytes: self.max_file_bytes, + max_staged_bytes: self.max_staged_bytes, + }) + .valid() || self.max_staged_bytes > u64::MAX / 8 || self.ttl_ms == 0 || self.ttl_ms > 7 * 24 * 60 * 60 * 1000 diff --git a/lib/crowdb-access-multipart/src/lib.rs b/lib/crowdb-access-multipart/src/lib.rs index a494db87..87ec7f53 100644 --- a/lib/crowdb-access-multipart/src/lib.rs +++ b/lib/crowdb-access-multipart/src/lib.rs @@ -15,8 +15,8 @@ use md5::{Digest, Md5}; mod state; pub use state::{ - reserve_part_accounting, validate_selected_parts, MultipartPhase, PartAccounting, SelectedPart, - StateError, + reserve_part_accounting, validate_selected_parts, MultipartBounds, MultipartPhase, PartAccounting, + SelectedPart, StateError, }; #[derive(Debug, thiserror::Error, PartialEq, Eq)] diff --git a/lib/crowdb-access-multipart/src/state.rs b/lib/crowdb-access-multipart/src/state.rs index 10c91c46..5352bc50 100644 --- a/lib/crowdb-access-multipart/src/state.rs +++ b/lib/crowdb-access-multipart/src/state.rs @@ -14,6 +14,26 @@ pub enum MultipartPhase { Conflicted, } +/// Protocol-neutral admission bounds for a durable multipart upload. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct MultipartBounds { + pub max_parts: u16, + pub max_part_bytes: u64, + pub max_object_bytes: u64, + pub max_staged_bytes: u64, +} + +impl MultipartBounds { + #[must_use] + pub const fn valid(self) -> bool { + self.max_parts > 0 + && self.max_parts <= 10_000 + && self.max_part_bytes > 0 + && self.max_part_bytes <= self.max_object_bytes + && self.max_object_bytes <= self.max_staged_bytes + } +} + #[derive(Clone, Copy, Debug, Eq, PartialEq, serde::Serialize, serde::Deserialize)] pub struct SelectedPart { pub number: u16, diff --git a/lib/crowdb-access-multipart/tests/state_test.rs b/lib/crowdb-access-multipart/tests/state_test.rs index 84b084ce..66eeeb21 100644 --- a/lib/crowdb-access-multipart/tests/state_test.rs +++ b/lib/crowdb-access-multipart/tests/state_test.rs @@ -2,7 +2,8 @@ // Licensed under the Apache License, Version 2.0. use crowdb_access_multipart::{ - reserve_part_accounting, validate_selected_parts, PartAccounting, SelectedPart, StateError, + reserve_part_accounting, validate_selected_parts, MultipartBounds, PartAccounting, SelectedPart, + StateError, }; fn selected(number: u16, revision: u64) -> SelectedPart { @@ -13,6 +14,32 @@ fn selected(number: u16, revision: u64) -> SelectedPart { } } +#[test] +fn both_protocols_share_the_same_multipart_admission_bounds() { + let valid = MultipartBounds { + max_parts: 10_000, + max_part_bytes: 10, + max_object_bytes: 100, + max_staged_bytes: 200, + }; + assert!(valid.valid()); + assert!(!MultipartBounds { + max_parts: 10_001, + ..valid + } + .valid()); + assert!(!MultipartBounds { + max_part_bytes: 101, + ..valid + } + .valid()); + assert!(!MultipartBounds { + max_staged_bytes: 99, + ..valid + } + .valid()); +} + #[test] fn selected_parts_require_strict_order_and_stable_revisions() { assert_eq!(validate_selected_parts(&[], 10), Err(StateError::EmptySelection)); diff --git a/lib/crowdb-access-s3/src/metadata/multipart.rs b/lib/crowdb-access-s3/src/metadata/multipart.rs index 93cfbb3b..559e5753 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart.rs @@ -4,7 +4,7 @@ //! Versioned durable S3 multipart session and part values. use bincode::Options as _; -use crowdb_access_multipart::{validate_selected_parts, MultipartComposer, SelectedPart}; +use crowdb_access_multipart::{validate_selected_parts, MultipartBounds, MultipartComposer, SelectedPart}; use crowdb_protocol::chunkdb::rpc::Location; use serde::{Deserialize, Serialize}; use sha2::{Digest as _, Sha256}; @@ -99,11 +99,13 @@ impl MultipartSessionRecord { || self.revision == 0 || self.created_ms >= self.expires_ms || self.content_type.len() > MAX_CONTENT_TYPE_BYTES - || self.max_parts == 0 - || self.max_parts > 10_000 - || self.max_part_bytes == 0 - || self.max_part_bytes > self.max_object_bytes - || self.max_object_bytes > self.max_staged_bytes + || !(MultipartBounds { + max_parts: self.max_parts, + max_part_bytes: self.max_part_bytes, + max_object_bytes: self.max_object_bytes, + max_staged_bytes: self.max_staged_bytes, + }) + .valid() || self.part_count > self.max_parts || self.staged_bytes > self.max_staged_bytes { From 8fd048d5c35c195f79ad9cdffacdff0f787c46ef Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:03:00 +0800 Subject: [PATCH 08/74] Document container crash collector boundary --- container/single-node-container/README.md | 26 +++++++++++++++++++++++ doc/working/plan-console-authority.md | 11 ++++++---- 2 files changed, 33 insertions(+), 4 deletions(-) diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index e8b5b3db..3b5b2029 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -29,3 +29,29 @@ building a Docker image. The release workflow archives the verified directory and packages those same files in its publish job, without recompiling them. Docker Hub publication is manual; actual publication verification is deferred until administrator preparation is complete. + +## Crash collection boundary + +The image does not configure the host's Linux core collector. Inspect +`/proc/sys/kernel/core_pattern` on the Docker host before expecting a dump in +the mounted data volume. A leading `|` sends a crash to a host-side collector; +relative file patterns write in the crashing process's working directory. +The container does not currently set a private core working directory or a +core size limit, and it does not provide dump retention or exact-build debug +symbols. Do not assume `/opt/crowdb/data` contains a core after a crash. + +- On a systemd-coredump host, use `coredumpctl list` and `coredumpctl dump` + on the host to locate and export a captured dump. +- On an Ubuntu Apport host, use the host's Apport report and core extraction + workflow. A pipe pattern does not create a volume file. +- On Docker Desktop, inspect the Linux VM's collector. The desktop host's + native crash directory is not the container's core directory. + +Core dumps can contain credentials and user data. Store exports privately, +apply host retention policy, and match the exact image revision and binary +build when symbolizing. The required bounded volume collection and symbol +distribution remain tracked by R188. + +Collector behavior follows the [Linux core pattern documentation](https://docs.kernel.org/admin-guide/sysctl/kernel.html), +[systemd-coredump manual](https://www.freedesktop.org/software/systemd/man/250/systemd-coredump.socket.html), +and [Ubuntu Apport documentation](https://ubuntu.com/project/docs/contributors/debugging/apport/). diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index ac0987ca..588e220d 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -8,9 +8,8 @@ Upstream: [R188](../backlog/R188-console-group0-authority.md). Goal: make Group 0 the shared CLI/Web authority while retaining only process and launch inputs locally. -Status: paused at the verified bootstrap-publication checkpoint by user request -to prioritize the single-node image and merge preparation. Remaining tasks below -are retained for resumption; this requirement is not complete. +Status: active after the single-node CI repair. The remaining authority, +configuration, documentation and crash-diagnostics tasks below are pending. ## Registration and acceptance failures @@ -91,7 +90,7 @@ are retained for resumption; this requirement is not complete. Three-service lifecycle regressions and the complete CLI suite, fmt and clippy pass. Logs: `/tmp/crowdb-chunk-launch-*.log`. Removal of legacy startup/restore paths remains coupled to bootstrap cutover below. -- [ ] **Remove mixed persistence**: remove the unreleased `ConsoleConfig` +- [~] **Remove mixed persistence**: remove the unreleased `ConsoleConfig` parser/writer, inline SSH secrets, topology restoration and fixtures after the launch lifecycle and replay-safe bootstrap paths are wired. Preserve bootstrap intent independently until verified cutover. Update CLI commands, @@ -146,6 +145,10 @@ is paused; it does not block the single-node image requirement. `container/single-node-container/{Dockerfile,entrypoint.sh,tests/**}`, `container/crowdb-monitor/src/**`, `container/single-node-container/README.md`. + The single-node README now states the host collector boundary and identifies + Apport, systemd-coredump and Docker Desktop lookup paths without promising a + volume dump. This host reports an Apport pipe pattern and core ulimit 0. + Volume retention, exact-build symbols and disposable-host acceptance remain. ## Documentation and completion - [ ] **Bare-metal documentation**: migrate verified KV, chunk and access From 0608b8459855cf4d5b4cf78bfb872c241a9d5940 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:03:05 +0800 Subject: [PATCH 09/74] Record unresolved CI and requirement issues --- doc/backlog/backlog.md | 34 ++++++++++++++++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index 4ff88de4..b8dc1f51 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -248,3 +248,37 @@ must be deleted — see the workflow's Post-merge cleanup section. --- + +## Open Issues + +- **R188 crash diagnostics acceptance:** This host routes `core_pattern` to + Apport, so a container-local directory and core ulimit cannot guarantee a + dump in `/opt/crowdb/data`. End-to-end acceptance needs a disposable host + with file-based collection or a verified host-collector export workflow. + Exact-build source-line symbols also need a distribution choice: compressed + line tables in the image with a measured size increase, or separate + exact-build debug symbols. Retention and symbolization remain unverified. +- **Server Tests intermittent strip lookup:** The CI log for + `small_object_writer_e2e::eight_closed_mirror_strips_become_one_durable_ec_strip_without_reread` + failed at `location strip`, while the exact test and its 16-test suite pass + locally. Preserve the failing run's runtime-server logs and artifact to + identify the first divergent state before changing the assertion or retry + policy. +- **R188 hardware display data:** Group 0 rack and node values hold IDs and + status but not the console's rack name, node host or SSH settings. The + authority cutover must define where shared display names live and keep + machine-local launch inputs in the launch registry. Until that split is + implemented, two consoles cannot reconstruct identical physical views from + Group 0 alone. +- **R167 shared multipart core:** The referenced R190 requirement is no + longer present in the backlog. Both adapters now use shared phase names, + selected-part validation, accounting and metadata-only location composition. + Iceberg's remaining session/part recovery is bound to its catalog identity + and store. Extract the remaining protocol-neutral transition decisions before + wiring S3; keep keys, authorization and responses in the protocol adapters. +- **R167 unreachable part cleanup:** S3 now preserves immutable part + generations so completion can publish a selected generation across a + concurrent part-number replacement. Losing replacement candidates and old + generations can remain unreachable. Abort, expiry, and replacement cleanup + need durable bounded records and reader-pin protection before the HTTP path + is enabled. From 6248dac043e34706cf698044410ea4a727a3f31e Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:07:01 +0800 Subject: [PATCH 10/74] Add bounded S3 multipart part listing --- doc/working/plan-s3-multipart.md | 4 +- lib/crowdb-access-s3/src/metadata.rs | 4 +- .../src/metadata/multipart_repository.rs | 2 + .../metadata/multipart_repository/listing.rs | 97 +++++++++++++++ lib/crowdb-access-s3/src/metadata/store.rs | 30 ++++- .../tests/multipart_repository_test.rs | 113 +++++++++++++++--- 6 files changed, 229 insertions(+), 21 deletions(-) create mode 100644 lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index a543687a..06e85752 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -40,7 +40,9 @@ Iceberg multipart path. state remain. - [ ] **S3 routes and wire**: classify create/upload/list/complete/abort/list uploads, parse bounded completion XML, emit compatible responses and errors. - Preserve SigV4 authentication and existing basic routes. + Preserve SigV4 authentication and existing basic routes. The repository now + provides bounded, ordered ListParts pagination over current generations; + upload listing and HTTP dispatch remain. - [ ] **Part ingestion**: reuse the bounded streaming writer and admission budget, persist part location/integrity before success, reconcile lost replies. - [ ] **Atomic completion**: fence selected part generations, validate order, diff --git a/lib/crowdb-access-s3/src/metadata.rs b/lib/crowdb-access-s3/src/metadata.rs index edee280f..0fc10cad 100644 --- a/lib/crowdb-access-s3/src/metadata.rs +++ b/lib/crowdb-access-s3/src/metadata.rs @@ -24,7 +24,9 @@ mod generated { pub use key::{BucketId, MetadataKey, MetadataKeyError, TenantId}; pub use multipart::{MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord}; -pub use multipart_repository::{CompletionPart, MultipartRepository, MultipartRepositoryError}; +pub use multipart_repository::{ + CompletionPart, MultipartPartPage, MultipartRepository, MultipartRepositoryError, +}; pub use namespace::{BucketDeleteOutcome, BucketNamespace, BucketNamespaceError}; pub use record::{BucketNameRecord, MetadataRecordError, ObjectRecord}; pub use store::{ChunkKvMetadataStore, MetadataStoreError, PutIfAbsentOutcome}; diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index e509c8ef..ef94f1b0 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -11,10 +11,12 @@ use super::{ }; mod completion; +mod listing; mod publication; mod terminal; pub use completion::CompletionPart; +pub use listing::MultipartPartPage; #[derive(Debug, thiserror::Error)] pub enum MultipartRepositoryError { diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs new file mode 100644 index 00000000..d1a08121 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs @@ -0,0 +1,97 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Bounded listing of current part-number generations. + +use super::{ + MetadataKey, MultipartPartRecord, MultipartPhase, MultipartRepository, MultipartRepositoryError, + MultipartSessionRecord, +}; + +const MAX_LIST_PARTS: usize = 1_000; +const MAX_SCAN_BYTES: usize = 4 * 1024 * 1024; + +pub struct MultipartPartPage { + pub parts: Vec, + pub next_part_number_marker: Option, +} + +impl MultipartRepository { + /// Lists current, visible part generations in ascending part-number order. + /// + /// # Errors + /// Rejects a closed upload, invalid marker/limit or malformed scan data. + pub async fn list_parts( + &self, + session: &MultipartSessionRecord, + after_number: u16, + max_parts: usize, + ) -> Result { + if max_parts == 0 || max_parts > MAX_LIST_PARTS || after_number > 10_000 { + return Err(MultipartRepositoryError::Conflict); + } + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.phase != MultipartPhase::Open { + return Err(MultipartRepositoryError::Conflict); + } + if after_number == 10_000 { + return Ok(MultipartPartPage { + parts: Vec::new(), + next_part_number_marker: None, + }); + } + let prefix = MetadataKey::multipart_part_prefix(&self.tenant, current.bucket_id, ¤t.upload_id); + let start = if after_number == 0 { + prefix.clone() + } else { + MetadataKey::multipart_part( + &self.tenant, + current.bucket_id, + ¤t.upload_id, + after_number + 1, + )? + }; + let end = MetadataKey::multipart_part_end(&self.tenant, current.bucket_id, ¤t.upload_id); + let page = self + .store + .scan_page(start, end, max_parts + 1, MAX_SCAN_BYTES, None) + .await?; + let mut parts = Vec::with_capacity(page.items.len().min(max_parts)); + let has_more = page.continuation.is_some() || page.items.len() > max_parts; + for item in page.items.into_iter().take(max_parts) { + let suffix = item + .key + .strip_prefix(prefix.as_slice()) + .ok_or(MultipartRepositoryError::InvalidPart)?; + let number = match suffix { + [high, low] => u16::from_be_bytes([*high, *low]), + _ => return Err(MultipartRepositoryError::InvalidPart), + }; + if number <= after_number + || parts + .last() + .is_some_and(|prior: &MultipartPartRecord| prior.number >= number) + { + return Err(MultipartRepositoryError::InvalidPart); + } + parts.push(MultipartPartRecord::decode( + &item.value, + current.bucket_id, + ¤t.upload_id, + number, + )?); + } + let next_part_number_marker = if has_more { + Some(parts.last().ok_or(MultipartRepositoryError::InvalidPart)?.number) + } else { + None + }; + Ok(MultipartPartPage { + parts, + next_part_number_marker, + }) + } +} diff --git a/lib/crowdb-access-s3/src/metadata/store.rs b/lib/crowdb-access-s3/src/metadata/store.rs index 4e8fcc82..df583ac4 100644 --- a/lib/crowdb-access-s3/src/metadata/store.rs +++ b/lib/crowdb-access-s3/src/metadata/store.rs @@ -3,7 +3,9 @@ use std::sync::Arc; -use crowdb_chunk_kv_client::{ChunkKvClient, ClientError, MultiScanPage, MultiScanRequest}; +use crowdb_chunk_kv_client::{ + ChunkKvClient, ClientError, MultiScanContinuation, MultiScanPage, MultiScanRequest, +}; use crowdb_protocol::chunk_kv::{ ChunkKvResponse, OperationResult, RpcCompareCondition, RpcFailure, RpcValue, ScanDirection, }; @@ -182,6 +184,24 @@ impl ChunkKvMetadataStore { max_items: usize, max_bytes: usize, ) -> Result, MetadataStoreError> { + Ok(self + .scan_page(start, end, max_items, max_bytes, None) + .await? + .items) + } + + /// Scans a bounded page while preserving its exact continuation fence. + /// + /// # Errors + /// Returns routing, storage or terminal scan failures. + pub async fn scan_page( + &self, + start: Vec, + end: Vec, + max_items: usize, + max_bytes: usize, + continuation: Option, + ) -> Result { let page = self .client .scan(MultiScanRequest { @@ -190,13 +210,13 @@ impl ChunkKvMetadataStore { direction: ScanDirection::Forward, max_items, max_bytes, - continuation: None, + continuation, }) .await?; - if let Some(terminal_failure) = page.terminal_failure { - return Err(failure(&terminal_failure)); + if let Some(terminal_failure) = &page.terminal_failure { + return Err(failure(terminal_failure)); } - Ok(page.items) + Ok(page) } } diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs index 49777526..730d20b7 100644 --- a/lib/crowdb-access-s3/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -18,7 +18,8 @@ use crowdb_chunk_kv_client::{ use crowdb_protocol::chunk_kv::{ ChunkKvRangeCatalogEntry, ChunkKvRangeCatalogHead, ChunkKvRangeCatalogPage, ChunkKvRangeCatalogPageRef, ChunkKvRangeCatalogPartitionState, ChunkKvResponse, Id128, KeyRange, OperationResult, OwnerDescriptor, - PartitionArtifact, PointOperation, PointRequest, RpcCompareCondition, RpcValue, + PartitionArtifact, PointOperation, PointRequest, RpcCompareCondition, RpcValue, ScanContinuation, + ScanRequest, }; use crowdb_protocol::chunk_stream::StreamName; use crowdb_protocol::chunkdb::rpc::Location; @@ -34,14 +35,19 @@ impl ChunkKvRangeCatalogSource for Catalog { } } -struct ActorTransport(mpsc::UnboundedSender<(PointOperation, oneshot::Sender>)>); +enum ActorRequest { + Point(PointOperation, oneshot::Sender>), + Scan(ScanRequest, oneshot::Sender>), +} + +struct ActorTransport(mpsc::UnboundedSender); #[async_trait] impl ChunkKvTransport for ActorTransport { async fn point(&self, _: &str, request: &PointRequest) -> Result { let (response, receiver) = oneshot::channel(); self.0 - .send((request.operation.clone(), response)) + .send(ActorRequest::Point(request.operation.clone(), response)) .expect("actor is running"); receiver.await.expect("actor replies") } @@ -50,8 +56,12 @@ impl ChunkKvTransport for ActorTransport { unreachable!() } - async fn scan(&self, _: &str, _: &crowdb_protocol::chunk_kv::ScanRequest) -> Result { - unreachable!() + async fn scan(&self, _: &str, request: &ScanRequest) -> Result { + let (response, receiver) = oneshot::channel(); + self.0 + .send(ActorRequest::Scan(request.clone(), response)) + .expect("actor is running"); + receiver.await.expect("actor replies") } } @@ -105,20 +115,64 @@ fn reply(operation: PointOperation, values: &mut HashMap, RpcValue>) -> } } +fn scan_reply(request: &ScanRequest, values: &HashMap, RpcValue>) -> ChunkKvResponse { + let mut ordered: Vec = values + .values() + .filter(|entry| { + request.start.as_ref().map_or(true, |start| entry.key >= *start) + && request.end.as_ref().map_or(true, |end| entry.key < *end) + && request + .continuation + .as_ref() + .map_or(true, |token| entry.key > token.last_key) + }) + .cloned() + .collect(); + ordered.sort_by(|left, right| left.key.cmp(&right.key)); + let limit = usize::try_from(request.limit).unwrap(); + let truncated = ordered.len() > limit; + ordered.truncate(limit); + let continuation = if truncated { + Some(ScanContinuation { + direction: request.direction, + last_key: ordered.last().unwrap().key.clone(), + partition_id: request.routing.partition_id, + owner_epoch: request.routing.owner_epoch, + map_revision: request.routing.map_revision, + }) + } else { + None + }; + ChunkKvResponse { + map_revision: 1, + journal_position: None, + result: Ok(OperationResult::Scan { + items: ordered, + continuation, + }), + } +} + async fn repository() -> (MultipartRepository, Arc, Arc) { - let (sender, mut receiver) = - mpsc::unbounded_channel::<(PointOperation, oneshot::Sender>)>(); + let (sender, mut receiver) = mpsc::unbounded_channel::(); let lose_reply = Arc::new(AtomicBool::new(false)); let actor_lose_reply = Arc::clone(&lose_reply); tokio::spawn(async move { let mut values = HashMap::new(); - while let Some((operation, response)) = receiver.recv().await { - let is_mutation = !matches!(operation, PointOperation::Get { .. }); - let result = reply(operation, &mut values); - if is_mutation && actor_lose_reply.swap(false, Ordering::SeqCst) { - let _ = response.send(Err(ClientError::Transport("committed reply lost".into()))); - } else { - let _ = response.send(Ok(result)); + while let Some(request) = receiver.recv().await { + match request { + ActorRequest::Point(operation, response) => { + let is_mutation = !matches!(operation, PointOperation::Get { .. }); + let result = reply(operation, &mut values); + if is_mutation && actor_lose_reply.swap(false, Ordering::SeqCst) { + let _ = response.send(Err(ClientError::Transport("committed reply lost".into()))); + } else { + let _ = response.send(Ok(result)); + } + } + ActorRequest::Scan(request, response) => { + let _ = response.send(Ok(scan_reply(&request, &values))); + } } } }); @@ -479,3 +533,34 @@ async fn abort_confirms_lost_reply_and_rejects_part_publication() { 5 ); } + +#[tokio::test] +async fn part_listing_paginates_current_generations_in_number_order() { + let (repository, _, _) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + for number in [3, 1, 2] { + let mut value = part(); + value.number = number; + repository.put_stream_part(&session, &value, 110).await.unwrap(); + } + let mut replacement = part(); + replacement.number = 2; + repository + .put_stream_part(&session, &replacement, 111) + .await + .unwrap(); + let first = repository.list_parts(&session, 0, 2).await.unwrap(); + assert_eq!( + first.parts.iter().map(|part| part.number).collect::>(), + [1, 2] + ); + assert_eq!(first.parts[1].revision, 2); + assert_eq!(first.next_part_number_marker, Some(2)); + let second = repository.list_parts(&session, 2, 2).await.unwrap(); + assert_eq!( + second.parts.iter().map(|part| part.number).collect::>(), + [3] + ); + assert_eq!(second.next_part_number_marker, None); +} From a943c879c6a1db223b21b3e8d0b55264521361a0 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:09:30 +0800 Subject: [PATCH 11/74] Classify S3 multipart request shapes --- doc/working/plan-s3-multipart.md | 3 +- lib/crowdb-access-s3/src/route.rs | 4 + lib/crowdb-access-s3/src/route/multipart.rs | 100 ++++++++++++++++++++ lib/crowdb-access-s3/tests/route_test.rs | 67 ++++++++++++- 4 files changed, 172 insertions(+), 2 deletions(-) create mode 100644 lib/crowdb-access-s3/src/route/multipart.rs diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 06e85752..170db9f6 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -42,7 +42,8 @@ Iceberg multipart path. uploads, parse bounded completion XML, emit compatible responses and errors. Preserve SigV4 authentication and existing basic routes. The repository now provides bounded, ordered ListParts pagination over current generations; - upload listing and HTTP dispatch remain. + multipart query shapes are parsed separately. Upload listing and HTTP dispatch + remain pending. - [ ] **Part ingestion**: reuse the bounded streaming writer and admission budget, persist part location/integrity before success, reconcile lost replies. - [ ] **Atomic completion**: fence selected part generations, validate order, diff --git a/lib/crowdb-access-s3/src/route.rs b/lib/crowdb-access-s3/src/route.rs index 9562d159..970dcea6 100644 --- a/lib/crowdb-access-s3/src/route.rs +++ b/lib/crowdb-access-s3/src/route.rs @@ -6,6 +6,10 @@ use hyper::{HeaderMap, Method, Uri}; use percent_encoding::percent_decode_str; +mod multipart; + +pub use multipart::{classify_multipart, MultipartOperation, MultipartRoute}; + #[repr(usize)] #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub enum S3Operation { diff --git a/lib/crowdb-access-s3/src/route/multipart.rs b/lib/crowdb-access-s3/src/route/multipart.rs new file mode 100644 index 00000000..31243522 --- /dev/null +++ b/lib/crowdb-access-s3/src/route/multipart.rs @@ -0,0 +1,100 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Parse the S3 multipart query surface without enabling HTTP dispatch yet. + +use hyper::{Method, Uri}; + +use super::{decode, RouteError}; + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum MultipartOperation { + Create, + UploadPart, + ListParts, + Complete, + Abort, + ListUploads, +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct MultipartRoute { + pub operation: MultipartOperation, + pub bucket: Vec, + pub key: Option>, + pub upload_id: Option<[u8; 16]>, + pub part_number: Option, +} + +/// Classifies a multipart query after the common `SigV4` authentication step. +/// +/// # Errors +/// Rejects duplicate selectors, malformed identities and unsupported method +/// or path combinations. Returns `None` for ordinary basic S3 requests. +pub fn classify_multipart(method: &Method, uri: &Uri) -> Result, RouteError> { + let mut uploads = false; + let mut upload_id = None; + let mut part_number = None; + for pair in uri.query().unwrap_or_default().split('&') { + let (name, value) = pair.split_once('=').unwrap_or((pair, "")); + match name { + "uploads" if !uploads && value.is_empty() => uploads = true, + "uploadId" if upload_id.is_none() => upload_id = Some(parse_upload_id(value)?), + "partNumber" if part_number.is_none() => { + let number = value.parse::().map_err(|_| RouteError::Invalid)?; + if number == 0 || number > 10_000 { + return Err(RouteError::Invalid); + } + part_number = Some(number); + } + "uploads" | "uploadId" | "partNumber" => return Err(RouteError::Invalid), + _ => {} + } + } + if !uploads && upload_id.is_none() { + return Ok(None); + } + if uploads && (upload_id.is_some() || part_number.is_some()) { + return Err(RouteError::Invalid); + } + let path = uri.path().strip_prefix('/').ok_or(RouteError::Invalid)?; + let (bucket, key) = path + .split_once('/') + .map_or((path, None), |(bucket, key)| (bucket, Some(key))); + let bucket = decode(bucket); + if bucket.is_empty() { + return Err(RouteError::Invalid); + } + let key = key.map(decode); + let operation = match (method, key.as_deref(), uploads, upload_id, part_number) { + (&Method::POST, Some(_), true, None, None) => MultipartOperation::Create, + (&Method::GET, None, true, None, None) => MultipartOperation::ListUploads, + (&Method::PUT, Some(_), false, Some(_), Some(_)) => MultipartOperation::UploadPart, + (&Method::GET, Some(_), false, Some(_), None) => MultipartOperation::ListParts, + (&Method::POST, Some(_), false, Some(_), None) => MultipartOperation::Complete, + (&Method::DELETE, Some(_), false, Some(_), None) => MultipartOperation::Abort, + _ => return Err(RouteError::Invalid), + }; + Ok(Some(MultipartRoute { + operation, + bucket, + key, + upload_id, + part_number, + })) +} + +fn parse_upload_id(value: &str) -> Result<[u8; 16], RouteError> { + if value.len() != 32 || !value.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(RouteError::Invalid); + } + let mut id = [0_u8; 16]; + for (output, pair) in id.iter_mut().zip(value.as_bytes().chunks_exact(2)) { + *output = u8::from_str_radix(std::str::from_utf8(pair).map_err(|_| RouteError::Invalid)?, 16) + .map_err(|_| RouteError::Invalid)?; + } + if id == [0; 16] { + return Err(RouteError::Invalid); + } + Ok(id) +} diff --git a/lib/crowdb-access-s3/tests/route_test.rs b/lib/crowdb-access-s3/tests/route_test.rs index 68a88a45..a1917c50 100644 --- a/lib/crowdb-access-s3/tests/route_test.rs +++ b/lib/crowdb-access-s3/tests/route_test.rs @@ -1,7 +1,9 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crowdb_access_s3::route::{classify, classify_request, RouteError, S3Operation}; +use crowdb_access_s3::route::{ + classify, classify_multipart, classify_request, MultipartOperation, RouteError, S3Operation, +}; use hyper::header::HeaderValue; use hyper::{HeaderMap, Method, Uri}; @@ -43,3 +45,66 @@ fn rejects_headers_that_select_excluded_behavior() { Err(RouteError::NotImplemented) ); } + +#[test] +fn multipart_queries_have_unambiguous_paths_and_identities() { + let id = "ab".repeat(16); + let cases = [ + ( + Method::POST, + "/bucket/key?uploads".to_string(), + MultipartOperation::Create, + ), + ( + Method::GET, + "/bucket?uploads".to_string(), + MultipartOperation::ListUploads, + ), + ( + Method::PUT, + format!("/bucket/key?partNumber=7&uploadId={id}"), + MultipartOperation::UploadPart, + ), + ( + Method::GET, + format!("/bucket/key?uploadId={id}"), + MultipartOperation::ListParts, + ), + ( + Method::POST, + format!("/bucket/key?uploadId={id}"), + MultipartOperation::Complete, + ), + ( + Method::DELETE, + format!("/bucket/key?uploadId={id}"), + MultipartOperation::Abort, + ), + ]; + for (method, uri, expected) in cases { + let route = classify_multipart(&method, &uri.parse().unwrap()) + .unwrap() + .unwrap(); + assert_eq!(route.operation, expected); + assert_eq!(route.bucket, b"bucket"); + } + assert_eq!( + classify_multipart(&Method::GET, &"/bucket/key".parse().unwrap()), + Ok(None) + ); + assert_eq!( + classify_multipart( + &Method::PUT, + &format!("/bucket/key?uploadId={id}").parse().unwrap() + ), + Err(RouteError::Invalid) + ); + assert_eq!( + classify_multipart(&Method::POST, &"/bucket/key?uploadId=bad".parse().unwrap()), + Err(RouteError::Invalid) + ); + assert_eq!( + classify_multipart(&Method::GET, &"/bucket?uploads&uploads".parse().unwrap()), + Err(RouteError::Invalid) + ); +} From 2662031240de802b27c7ea55c6a29103479b2476 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:23:14 +0800 Subject: [PATCH 12/74] Add S3 multipart error and XML responses --- doc/working/plan-s3-multipart.md | 3 +- lib/crowdb-access-s3/src/error.rs | 17 +++++- lib/crowdb-access-s3/src/wire.rs | 75 +++++++++++++++++++++++- lib/crowdb-access-s3/tests/error_test.rs | 12 ++++ lib/crowdb-access-s3/tests/wire_test.rs | 35 ++++++++++- 5 files changed, 138 insertions(+), 4 deletions(-) diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 170db9f6..c0457715 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -43,7 +43,8 @@ Iceberg multipart path. Preserve SigV4 authentication and existing basic routes. The repository now provides bounded, ordered ListParts pagination over current generations; multipart query shapes are parsed separately. Upload listing and HTTP dispatch - remain pending. + remain pending. S3-compatible multipart error codes and the create, complete + and ListParts XML response builders have focused tests. - [ ] **Part ingestion**: reuse the bounded streaming writer and admission budget, persist part location/integrity before success, reconcile lost replies. - [ ] **Atomic completion**: fence selected part generations, validate order, diff --git a/lib/crowdb-access-s3/src/error.rs b/lib/crowdb-access-s3/src/error.rs index 00dcbf3a..69a75046 100644 --- a/lib/crowdb-access-s3/src/error.rs +++ b/lib/crowdb-access-s3/src/error.rs @@ -13,8 +13,12 @@ pub enum S3ErrorCode { NotImplemented, NoSuchBucket, NoSuchKey, + NoSuchUpload, BucketNotEmpty, InvalidRequest, + InvalidPart, + InvalidPartOrder, + EntityTooSmall, InvalidRange, PreconditionFailed, SlowDown, @@ -35,8 +39,12 @@ impl S3ErrorCode { Self::NotImplemented => "NotImplemented", Self::NoSuchBucket => "NoSuchBucket", Self::NoSuchKey => "NoSuchKey", + Self::NoSuchUpload => "NoSuchUpload", Self::BucketNotEmpty => "BucketNotEmpty", Self::InvalidRequest => "InvalidRequest", + Self::InvalidPart => "InvalidPart", + Self::InvalidPartOrder => "InvalidPartOrder", + Self::EntityTooSmall => "EntityTooSmall", Self::InvalidRange => "InvalidRange", Self::PreconditionFailed => "PreconditionFailed", Self::SlowDown => "SlowDown", @@ -57,8 +65,12 @@ impl S3ErrorCode { Self::NotImplemented => NOT_IMPLEMENTED_MESSAGE, Self::NoSuchBucket => "The specified bucket does not exist.", Self::NoSuchKey => "The specified key does not exist.", + Self::NoSuchUpload => "The specified multipart upload does not exist.", Self::BucketNotEmpty => "The bucket you tried to delete is not empty.", Self::InvalidRequest => "The request is not valid for this service.", + Self::InvalidPart => "One or more of the specified parts could not be found or matched.", + Self::InvalidPartOrder => "The list of parts was not in ascending order.", + Self::EntityTooSmall => "A nonfinal multipart part is smaller than the minimum size.", Self::InvalidRange => "The requested range is not satisfiable.", Self::PreconditionFailed => "At least one precondition failed.", Self::SlowDown => "Please reduce your request rate.", @@ -83,9 +95,12 @@ impl S3ErrorCode { const fn status(self) -> StatusCode { match self { Self::NotImplemented => StatusCode::NOT_IMPLEMENTED, - Self::NoSuchBucket | Self::NoSuchKey => StatusCode::NOT_FOUND, + Self::NoSuchBucket | Self::NoSuchKey | Self::NoSuchUpload => StatusCode::NOT_FOUND, Self::BucketNotEmpty => StatusCode::CONFLICT, Self::InvalidRequest + | Self::InvalidPart + | Self::InvalidPartOrder + | Self::EntityTooSmall | Self::RequestTimeTooSkewed | Self::InvalidDigest | Self::BadDigest diff --git a/lib/crowdb-access-s3/src/wire.rs b/lib/crowdb-access-s3/src/wire.rs index 532f7e1a..c8ac08ad 100644 --- a/lib/crowdb-access-s3/src/wire.rs +++ b/lib/crowdb-access-s3/src/wire.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crate::metadata::{BucketNameRecord, ObjectRecord}; +use crate::metadata::{BucketNameRecord, MultipartPartPage, ObjectRecord}; use crate::object::ListObjectsV2Page; #[must_use] @@ -22,6 +22,79 @@ pub fn list_buckets(tenant: &[u8], buckets: &[BucketNameRecord]) -> String { output } +#[must_use] +pub fn create_multipart_upload(bucket: &[u8], key: &[u8], upload_id: &[u8; 16]) -> String { + let mut output = xml_start("InitiateMultipartUploadResult"); + element(&mut output, "Bucket", &String::from_utf8_lossy(bucket)); + element(&mut output, "Key", &String::from_utf8_lossy(key)); + element(&mut output, "UploadId", &hex_upload_id(upload_id)); + output.push_str(""); + output +} + +#[must_use] +pub fn complete_multipart_upload(location: &str, bucket: &[u8], key: &[u8], etag: &str) -> String { + let mut output = xml_start("CompleteMultipartUploadResult"); + element(&mut output, "Location", location); + element(&mut output, "Bucket", &String::from_utf8_lossy(bucket)); + element(&mut output, "Key", &String::from_utf8_lossy(key)); + element(&mut output, "ETag", &format!("\"{etag}\"")); + output.push_str(""); + output +} + +#[must_use] +pub fn list_multipart_parts( + bucket: &[u8], + key: &[u8], + upload_id: &[u8; 16], + marker: u16, + max_parts: usize, + page: &MultipartPartPage, +) -> String { + let mut output = xml_start("ListPartsResult"); + element(&mut output, "Bucket", &String::from_utf8_lossy(bucket)); + element(&mut output, "Key", &String::from_utf8_lossy(key)); + element(&mut output, "UploadId", &hex_upload_id(upload_id)); + element(&mut output, "PartNumberMarker", &marker.to_string()); + if let Some(next) = page.next_part_number_marker { + element(&mut output, "NextPartNumberMarker", &next.to_string()); + } + element(&mut output, "MaxParts", &max_parts.to_string()); + element( + &mut output, + "IsTruncated", + if page.next_part_number_marker.is_some() { + "true" + } else { + "false" + }, + ); + for part in &page.parts { + output.push_str(""); + element(&mut output, "PartNumber", &part.number.to_string()); + element(&mut output, "LastModified", &iso8601(part.modified_ms)); + element( + &mut output, + "ETag", + &format!("\"{:x}\"", md5::Digest(part.raw_md5)), + ); + element(&mut output, "Size", &part.length.to_string()); + output.push_str(""); + } + output.push_str(""); + output +} + +fn hex_upload_id(upload_id: &[u8; 16]) -> String { + use std::fmt::Write as _; + let mut result = String::with_capacity(32); + for byte in upload_id { + write!(&mut result, "{byte:02x}").expect("string write cannot fail"); + } + result +} + #[must_use] pub fn list_objects( bucket: &[u8], diff --git a/lib/crowdb-access-s3/tests/error_test.rs b/lib/crowdb-access-s3/tests/error_test.rs index 1a49bc3b..1a12dd1f 100644 --- a/lib/crowdb-access-s3/tests/error_test.rs +++ b/lib/crowdb-access-s3/tests/error_test.rs @@ -52,6 +52,18 @@ fn every_public_error_class_has_a_stable_status_and_code() { ), (S3ErrorCode::NoSuchBucket, StatusCode::NOT_FOUND, "NoSuchBucket"), (S3ErrorCode::NoSuchKey, StatusCode::NOT_FOUND, "NoSuchKey"), + (S3ErrorCode::NoSuchUpload, StatusCode::NOT_FOUND, "NoSuchUpload"), + (S3ErrorCode::InvalidPart, StatusCode::BAD_REQUEST, "InvalidPart"), + ( + S3ErrorCode::InvalidPartOrder, + StatusCode::BAD_REQUEST, + "InvalidPartOrder", + ), + ( + S3ErrorCode::EntityTooSmall, + StatusCode::BAD_REQUEST, + "EntityTooSmall", + ), ( S3ErrorCode::BucketNotEmpty, StatusCode::CONFLICT, diff --git a/lib/crowdb-access-s3/tests/wire_test.rs b/lib/crowdb-access-s3/tests/wire_test.rs index eef424bd..b0dff769 100644 --- a/lib/crowdb-access-s3/tests/wire_test.rs +++ b/lib/crowdb-access-s3/tests/wire_test.rs @@ -1,7 +1,9 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crowdb_access_s3::metadata::{BucketId, BucketNameRecord, ObjectRecord, TenantId}; +use crowdb_access_s3::metadata::{ + BucketId, BucketNameRecord, MultipartPartPage, MultipartPartRecord, ObjectRecord, TenantId, +}; use crowdb_access_s3::object::ListObjectsV2Page; use crowdb_access_s3::wire; @@ -46,3 +48,34 @@ fn list_objects_serializes_stable_etag_time_and_continuation() { assert!(xml.contains("true")); assert!(xml.contains("opaque")); } + +#[test] +fn multipart_xml_escapes_names_and_reports_selected_part_metadata() { + let id = [0xab; 16]; + let created = wire::create_multipart_upload(b"b&", b"k<", &id); + assert!(created.contains("b&")); + assert!(created.contains("k<")); + assert!(created.contains(&format!("{}", "ab".repeat(16)))); + + let page = MultipartPartPage { + parts: vec![MultipartPartRecord { + bucket_id: BucketId::new([1; 16]), + upload_id: id, + number: 4, + revision: 1, + modified_ms: 1_000, + length: 8, + raw_md5: [9; 16], + locations: Vec::new(), + }], + next_part_number_marker: Some(4), + }; + let listed = wire::list_multipart_parts(b"b&", b"k<", &id, 2, 1, &page); + assert!(listed.contains("2")); + assert!(listed.contains("4")); + assert!(listed.contains(""09090909090909090909090909090909"")); + assert!(listed.contains("8")); + let completed = wire::complete_multipart_upload("http://host/b&/k<", b"b&", b"k<", "abc-1"); + assert!(completed.contains("http://host/b&/k<")); + assert!(completed.contains(""abc-1"")); +} From 883884c241f4a4fcf0a1a0b22c613254460a3806 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:24:17 +0800 Subject: [PATCH 13/74] Stop hardware cascades on child deletion failure --- doc/working/plan-console-authority.md | 5 +++- .../src/hardware/hierarchy.rs | 26 +++++-------------- 2 files changed, 10 insertions(+), 21 deletions(-) diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 588e220d..f6165317 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -97,7 +97,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. Web persistence and S3 mini-cluster callers together; no compatibility reader. - [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through shared Group 0 hardware operations; preserve conflicts and uncertain writes - without local-first commits. Docker keeps its hardware restrictions. + without local-first commits. Docker keeps its hardware restrictions. Hardware + client cascades now stop at a failed child deletion instead of deleting its + parent while a descendant may survive; exact-value confirmation and the + CLI/Web cutover remain. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. diff --git a/lib/crowdb-kv-client/src/hardware/hierarchy.rs b/lib/crowdb-kv-client/src/hardware/hierarchy.rs index 74d383c9..0178ffdb 100644 --- a/lib/crowdb-kv-client/src/hardware/hierarchy.rs +++ b/lib/crowdb-kv-client/src/hardware/hierarchy.rs @@ -18,8 +18,6 @@ use std::sync::Arc; use std::time::{SystemTime, UNIX_EPOCH}; -use tracing::warn; - use crowdb_protocol::common::{HwStatus, NodeValue, RackValue}; use crowdb_protocol::common_type::{DiskGroupId, NodeId, RackId}; use crowdb_protocol::diskdb::rpc::{DiskGroupValue, DiskValue}; @@ -951,22 +949,14 @@ impl HardwareClient { // Remove child disks. let disks = self.list_disks_in_group(rack_id, node_id, dg_id).await?; for (disk_id, _) in &disks { - if let Err(e) = self.remove_disk(rack_id, node_id, dg_id, disk_id).await { - warn!(error = %e, dg_id, "cascade: remove_disk failed; continuing"); - } + self.remove_disk(rack_id, node_id, dg_id, disk_id).await?; } // Remove owner map entry. - if let Err(e) = self.remove_owner(rack_id, node_id, dg_id).await { - warn!(error = %e, dg_id, "cascade: remove_owner failed; continuing"); - } + self.remove_owner(rack_id, node_id, dg_id).await?; // Remove bind map entry. - if let Err(e) = self.remove_bind(rack_id, node_id, dg_id).await { - warn!(error = %e, dg_id, "cascade: remove_bind failed; continuing"); - } + self.remove_bind(rack_id, node_id, dg_id).await?; // Remove usage summary. - if let Err(e) = self.remove_disk_group_usage(dg_id).await { - warn!(error = %e, dg_id, "cascade: remove_disk_group_usage failed; continuing"); - } + self.remove_disk_group_usage(dg_id).await?; // Remove the disk-group record itself. self.remove_disk_group(rack_id, node_id, dg_id).await } @@ -976,9 +966,7 @@ impl HardwareClient { pub async fn remove_node_cascade(&self, rack_id: RackId, node_id: NodeId) -> Result<()> { let dgs = self.list_disk_groups_on_node(rack_id, node_id).await?; for dg in &dgs { - if let Err(e) = self.remove_disk_group_cascade(rack_id, node_id, dg.dg_id).await { - warn!(error = %e, node_id, dg_id = dg.dg_id, "cascade: remove_disk_group_cascade failed; continuing"); - } + self.remove_disk_group_cascade(rack_id, node_id, dg.dg_id).await?; } self.remove_node(rack_id, node_id).await } @@ -988,9 +976,7 @@ impl HardwareClient { pub async fn remove_rack_cascade(&self, rack_id: RackId) -> Result<()> { let nodes = self.list_nodes_in_rack(rack_id).await?; for (node_id, _) in &nodes { - if let Err(e) = self.remove_node_cascade(rack_id, *node_id).await { - warn!(error = %e, rack_id, node_id, "cascade: remove_node_cascade failed; continuing"); - } + self.remove_node_cascade(rack_id, *node_id).await?; } self.remove_rack(rack_id).await } From 918a04283f40d3c2133745bed1ad38bed58c8d0e Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:27:59 +0800 Subject: [PATCH 14/74] Share bounded multipart completion XML parsing --- .../src/iceberg/file_selection.rs | 185 ++---------------- app/crowdb-access-server/src/lib.rs | 1 + .../src/multipart_complete.rs | 174 ++++++++++++++++ app/crowdb-access-server/src/s3.rs | 1 + .../tests/s3_multipart_xml_test.rs | 16 ++ doc/working/plan-s3-multipart.md | 4 +- 6 files changed, 206 insertions(+), 175 deletions(-) create mode 100644 app/crowdb-access-server/src/multipart_complete.rs create mode 100644 app/crowdb-access-server/tests/s3_multipart_xml_test.rs diff --git a/app/crowdb-access-server/src/iceberg/file_selection.rs b/app/crowdb-access-server/src/iceberg/file_selection.rs index 2bb6c4cf..26525fff 100644 --- a/app/crowdb-access-server/src/iceberg/file_selection.rs +++ b/app/crowdb-access-server/src/iceberg/file_selection.rs @@ -1,28 +1,11 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +pub use crate::multipart_complete::{CompletePart, CompleteRequestError, CompleteSelection}; use crowdb_access_iceberg::catalog::CatalogError; use crowdb_access_iceberg::file::{ MultipartPhase, MultipartRepository, MultipartSelection, MultipartSession, SelectedPart, }; -use quick_xml::events::Event; -use quick_xml::Reader; - -const MAX_COMPLETE_XML_BYTES: usize = 2 * 1024 * 1024; -const MAX_COMPLETE_PARTS: usize = 10_000; -const S3_NAMESPACE: &[u8] = b"http://s3.amazonaws.com/doc/2006-03-01/"; - -#[derive(Clone, Debug, Eq, PartialEq)] -pub struct CompletePart { - pub number: u16, - pub etag: String, -} - -#[derive(Clone, Debug, Eq, PartialEq)] -pub struct CompleteSelection { - parts: Vec, -} - -#[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] -#[error("invalid multipart completion XML")] -pub struct CompleteRequestError; #[derive(Debug, thiserror::Error)] pub enum CompleteResolveError { @@ -35,96 +18,6 @@ pub enum CompleteResolveError { } impl CompleteSelection { - /// Parses a bounded S3 `CompleteMultipartUpload` body. The caller must match - /// each selected digest to the current durable part revision before freezing. - /// # Errors - /// Rejects malformed XML, extra fields and unordered or duplicate parts. - pub fn parse(bytes: &[u8]) -> Result { - if bytes.is_empty() || bytes.len() > MAX_COMPLETE_XML_BYTES { - return Err(CompleteRequestError); - } - let mut reader = Reader::from_reader(bytes); - let mut state = State::Start; - let mut parts = Vec::new(); - let mut number = None; - let mut digest = None; - let mut etag = Vec::new(); - loop { - match reader.read_event().map_err(|_| CompleteRequestError)? { - Event::Decl(_) if state == State::Start => {} - Event::Start(event) if valid_attributes(state, &event)? => { - state = match (state, event.name().as_ref()) { - (State::Start, b"CompleteMultipartUpload") => State::Root, - (State::Root, b"Part") if parts.len() < MAX_COMPLETE_PARTS => State::Part, - (State::Part, b"PartNumber") if number.is_none() => State::Number, - (State::Part, b"ETag") if digest.is_none() => State::Etag, - _ => return Err(CompleteRequestError), - }; - } - Event::Text(event) => match state { - State::Number if number.is_none() => { - let value: &[u8] = event.as_ref(); - if value.is_empty() || !value.iter().all(u8::is_ascii_digit) { - return Err(CompleteRequestError); - } - number = Some( - std::str::from_utf8(value) - .map_err(|_| CompleteRequestError)? - .parse::() - .map_err(|_| CompleteRequestError)?, - ); - } - State::Etag => append_etag(&mut etag, &event)?, - State::Start | State::Root | State::Part | State::Done - if event.iter().all(u8::is_ascii_whitespace) => {} - _ => return Err(CompleteRequestError), - }, - Event::GeneralRef(event) if state == State::Etag => { - if event.len() > 16 { - return Err(CompleteRequestError); - } - let name = std::str::from_utf8(&event).map_err(|_| CompleteRequestError)?; - let encoded = format!("&{name};"); - let decoded = quick_xml::escape::unescape(&encoded).map_err(|_| CompleteRequestError)?; - append_etag(&mut etag, decoded.as_bytes())?; - } - Event::End(event) => { - state = match (state, event.name().as_ref()) { - (State::Number, b"PartNumber") if number.is_some() => State::Part, - (State::Etag, b"ETag") => { - digest = Some(parse_etag(&etag)?); - etag.clear(); - State::Part - } - (State::Part, b"Part") => { - let number = number.take().ok_or(CompleteRequestError)?; - let etag = digest.take().ok_or(CompleteRequestError)?; - if number == 0 - || number > 10_000 - || parts - .last() - .is_some_and(|part: &CompletePart| part.number >= number) - { - return Err(CompleteRequestError); - } - parts.push(CompletePart { number, etag }); - State::Root - } - (State::Root, b"CompleteMultipartUpload") if !parts.is_empty() => State::Done, - _ => return Err(CompleteRequestError), - }; - } - Event::Eof if state == State::Done => return Ok(Self { parts }), - _ => return Err(CompleteRequestError), - } - } - } - - #[must_use] - pub fn parts(&self) -> &[CompletePart] { - &self.parts - } - /// Resolves the selected parts against one current durable session snapshot. /// # Errors /// Rejects missing, replaced or differently hashed parts and storage failures. @@ -133,14 +26,14 @@ impl CompleteSelection { repository: &MultipartRepository, session: &MultipartSession, ) -> Result { - if self.parts.len() > usize::from(session.limits.max_parts) { + if self.parts().len() > usize::from(session.limits.max_parts) { return Err(CompleteResolveError::InvalidPart); } if session.completion.is_some() { let frozen = repository.load_selection(session).await?; if let Some(snapshots) = frozen.snapshots() { - if self.parts.len() != snapshots.len() - || self.parts.iter().zip(frozen.parts().iter().zip(snapshots)).any( + if self.parts().len() != snapshots.len() + || self.parts().iter().zip(frozen.parts().iter().zip(snapshots)).any( |(requested, (selected, snapshot))| { requested.number != selected.number || requested.etag != snapshot.etag }, @@ -151,9 +44,9 @@ impl CompleteSelection { return Ok(frozen); } } - let mut selected = Vec::with_capacity(self.parts.len()); - let mut parts = Vec::with_capacity(self.parts.len()); - for (index, requested) in self.parts.iter().enumerate() { + let mut selected = Vec::with_capacity(self.parts().len()); + let mut parts = Vec::with_capacity(self.parts().len()); + for (index, requested) in self.parts().iter().enumerate() { let part = if session.phase == MultipartPhase::Open { repository.part_for_upload(session, requested.number).await? } else { @@ -163,7 +56,7 @@ impl CompleteSelection { if part.etag() != requested.etag { return Err(CompleteResolveError::InvalidPart); } - if index + 1 < self.parts.len() && part.length() < 5 * 1024 * 1024 { + if index + 1 < self.parts().len() && part.length() < 5 * 1024 * 1024 { return Err(CompleteResolveError::EntityTooSmall); } selected.push(SelectedPart { @@ -180,59 +73,3 @@ impl CompleteSelection { } } } - -#[derive(Clone, Copy, Eq, PartialEq)] -enum State { - Start, - Root, - Part, - Number, - Etag, - Done, -} - -fn parse_etag(bytes: &[u8]) -> Result { - let hex = bytes - .strip_prefix(b"\"") - .and_then(|bytes| bytes.strip_suffix(b"\"")) - .ok_or(CompleteRequestError)?; - if hex.len() != 32 && hex.len() != 64 { - return Err(CompleteRequestError); - } - for pair in hex.chunks_exact(2) { - let _ = (hex_digit(pair[0])? << 4) | hex_digit(pair[1])?; - } - String::from_utf8(hex.to_vec()).map_err(|_| CompleteRequestError) -} - -fn append_etag(etag: &mut Vec, bytes: &[u8]) -> Result<(), CompleteRequestError> { - if etag.len().saturating_add(bytes.len()) > 66 { - return Err(CompleteRequestError); - } - etag.extend_from_slice(bytes); - Ok(()) -} - -fn hex_digit(byte: u8) -> Result { - match byte { - b'0'..=b'9' => Ok(byte - b'0'), - b'a'..=b'f' => Ok(byte - b'a' + 10), - _ => Err(CompleteRequestError), - } -} - -fn valid_attributes( - state: State, - event: &quick_xml::events::BytesStart<'_>, -) -> Result { - let mut attributes = event.attributes(); - let Some(attribute) = attributes.next() else { - return Ok(true); - }; - let attribute = attribute.map_err(|_| CompleteRequestError)?; - Ok(state == State::Start - && event.name().as_ref() == b"CompleteMultipartUpload" - && attribute.key.as_ref() == b"xmlns" - && attribute.value.as_ref() == S3_NAMESPACE - && attributes.next().is_none()) -} diff --git a/app/crowdb-access-server/src/lib.rs b/app/crowdb-access-server/src/lib.rs index d7b637f7..3029821e 100644 --- a/app/crowdb-access-server/src/lib.rs +++ b/app/crowdb-access-server/src/lib.rs @@ -6,6 +6,7 @@ pub mod config; mod http_receive; pub mod iceberg; +mod multipart_complete; #[cfg(feature = "s3")] pub mod credentials; diff --git a/app/crowdb-access-server/src/multipart_complete.rs b/app/crowdb-access-server/src/multipart_complete.rs new file mode 100644 index 00000000..46f89f84 --- /dev/null +++ b/app/crowdb-access-server/src/multipart_complete.rs @@ -0,0 +1,174 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Bounded S3 completion XML shared by the Iceberg and S3 protocol modules. + +use quick_xml::events::Event; +use quick_xml::Reader; + +const MAX_COMPLETE_XML_BYTES: usize = 2 * 1024 * 1024; +const MAX_COMPLETE_PARTS: usize = 10_000; +const S3_NAMESPACE: &[u8] = b"http://s3.amazonaws.com/doc/2006-03-01/"; + +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct CompletePart { + pub number: u16, + pub etag: String, +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct CompleteSelection { + parts: Vec, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] +#[error("invalid multipart completion XML")] +pub struct CompleteRequestError; + +impl CompleteSelection { + /// Parses a bounded S3 `CompleteMultipartUpload` body. The caller must match + /// each selected digest to the current durable part revision before freezing. + /// # Errors + /// Rejects malformed XML, extra fields and unordered or duplicate parts. + pub fn parse(bytes: &[u8]) -> Result { + if bytes.is_empty() || bytes.len() > MAX_COMPLETE_XML_BYTES { + return Err(CompleteRequestError); + } + let mut reader = Reader::from_reader(bytes); + let mut state = State::Start; + let mut parts = Vec::new(); + let mut number = None; + let mut digest = None; + let mut etag = Vec::new(); + loop { + match reader.read_event().map_err(|_| CompleteRequestError)? { + Event::Decl(_) if state == State::Start => {} + Event::Start(event) if valid_attributes(state, &event)? => { + state = match (state, event.name().as_ref()) { + (State::Start, b"CompleteMultipartUpload") => State::Root, + (State::Root, b"Part") if parts.len() < MAX_COMPLETE_PARTS => State::Part, + (State::Part, b"PartNumber") if number.is_none() => State::Number, + (State::Part, b"ETag") if digest.is_none() => State::Etag, + _ => return Err(CompleteRequestError), + }; + } + Event::Text(event) => match state { + State::Number if number.is_none() => { + let value: &[u8] = event.as_ref(); + if value.is_empty() || !value.iter().all(u8::is_ascii_digit) { + return Err(CompleteRequestError); + } + number = Some( + std::str::from_utf8(value) + .map_err(|_| CompleteRequestError)? + .parse::() + .map_err(|_| CompleteRequestError)?, + ); + } + State::Etag => append_etag(&mut etag, &event)?, + State::Start | State::Root | State::Part | State::Done + if event.iter().all(u8::is_ascii_whitespace) => {} + _ => return Err(CompleteRequestError), + }, + Event::GeneralRef(event) if state == State::Etag => { + if event.len() > 16 { + return Err(CompleteRequestError); + } + let name = std::str::from_utf8(&event).map_err(|_| CompleteRequestError)?; + let encoded = format!("&{name};"); + let decoded = quick_xml::escape::unescape(&encoded).map_err(|_| CompleteRequestError)?; + append_etag(&mut etag, decoded.as_bytes())?; + } + Event::End(event) => { + state = match (state, event.name().as_ref()) { + (State::Number, b"PartNumber") if number.is_some() => State::Part, + (State::Etag, b"ETag") => { + digest = Some(parse_etag(&etag)?); + etag.clear(); + State::Part + } + (State::Part, b"Part") => { + let number = number.take().ok_or(CompleteRequestError)?; + let etag = digest.take().ok_or(CompleteRequestError)?; + if number == 0 + || number > 10_000 + || parts + .last() + .is_some_and(|part: &CompletePart| part.number >= number) + { + return Err(CompleteRequestError); + } + parts.push(CompletePart { number, etag }); + State::Root + } + (State::Root, b"CompleteMultipartUpload") if !parts.is_empty() => State::Done, + _ => return Err(CompleteRequestError), + }; + } + Event::Eof if state == State::Done => return Ok(Self { parts }), + _ => return Err(CompleteRequestError), + } + } + } + + #[must_use] + pub fn parts(&self) -> &[CompletePart] { + &self.parts + } +} + +#[derive(Clone, Copy, Eq, PartialEq)] +enum State { + Start, + Root, + Part, + Number, + Etag, + Done, +} + +fn parse_etag(bytes: &[u8]) -> Result { + let hex = bytes + .strip_prefix(b"\"") + .and_then(|bytes| bytes.strip_suffix(b"\"")) + .ok_or(CompleteRequestError)?; + if hex.len() != 32 && hex.len() != 64 { + return Err(CompleteRequestError); + } + for pair in hex.chunks_exact(2) { + let _ = (hex_digit(pair[0])? << 4) | hex_digit(pair[1])?; + } + String::from_utf8(hex.to_vec()).map_err(|_| CompleteRequestError) +} + +fn append_etag(etag: &mut Vec, bytes: &[u8]) -> Result<(), CompleteRequestError> { + if etag.len().saturating_add(bytes.len()) > 66 { + return Err(CompleteRequestError); + } + etag.extend_from_slice(bytes); + Ok(()) +} + +fn hex_digit(byte: u8) -> Result { + match byte { + b'0'..=b'9' => Ok(byte - b'0'), + b'a'..=b'f' => Ok(byte - b'a' + 10), + _ => Err(CompleteRequestError), + } +} + +fn valid_attributes( + state: State, + event: &quick_xml::events::BytesStart<'_>, +) -> Result { + let mut attributes = event.attributes(); + let Some(attribute) = attributes.next() else { + return Ok(true); + }; + let attribute = attribute.map_err(|_| CompleteRequestError)?; + Ok(state == State::Start + && event.name().as_ref() == b"CompleteMultipartUpload" + && attribute.key.as_ref() == b"xmlns" + && attribute.value.as_ref() == S3_NAMESPACE + && attributes.next().is_none()) +} diff --git a/app/crowdb-access-server/src/s3.rs b/app/crowdb-access-server/src/s3.rs index 8c820102..13e2a656 100644 --- a/app/crowdb-access-server/src/s3.rs +++ b/app/crowdb-access-server/src/s3.rs @@ -23,6 +23,7 @@ mod dispatcher; mod operations; pub use crate::http_receive::install_body_receive_provider; +pub use crate::multipart_complete::{CompletePart, CompleteRequestError, CompleteSelection}; pub use dispatcher::S3Dispatcher; pub use operations::{ProductionS3Operations, S3Operations, S3OperationsFuture, S3ServiceConfig}; diff --git a/app/crowdb-access-server/tests/s3_multipart_xml_test.rs b/app/crowdb-access-server/tests/s3_multipart_xml_test.rs new file mode 100644 index 00000000..0299186d --- /dev/null +++ b/app/crowdb-access-server/tests/s3_multipart_xml_test.rs @@ -0,0 +1,16 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_server::s3::CompleteSelection; + +#[test] +fn s3_and_iceberg_use_the_same_bounded_completion_parser() { + let digest = "ab".repeat(16); + let body = format!( + "1\"{digest}\"" + ); + let s3 = CompleteSelection::parse(body.as_bytes()).unwrap(); + let iceberg = crowdb_access_server::iceberg::CompleteSelection::parse(body.as_bytes()).unwrap(); + assert_eq!(s3.parts(), iceberg.parts()); + assert_eq!(s3.parts()[0].etag, digest); +} diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index c0457715..6cf2d5fd 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -44,7 +44,9 @@ Iceberg multipart path. provides bounded, ordered ListParts pagination over current generations; multipart query shapes are parsed separately. Upload listing and HTTP dispatch remain pending. S3-compatible multipart error codes and the create, complete - and ListParts XML response builders have focused tests. + and ListParts XML response builders have focused tests. The bounded completion + XML parser now has one implementation in access-server and is exposed by both + the Iceberg and S3 protocol modules. The S3 HTTP path still needs wiring. - [ ] **Part ingestion**: reuse the bounded streaming writer and admission budget, persist part location/integrity before success, reconcile lost replies. - [ ] **Atomic completion**: fence selected part generations, validate order, From fa8e54a18d14ecd48ba1375f2391ebe9484329fe Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:32:40 +0800 Subject: [PATCH 15/74] Add bounded S3 multipart upload listing --- doc/working/plan-s3-multipart.md | 6 +- lib/crowdb-access-s3/src/metadata.rs | 2 +- lib/crowdb-access-s3/src/metadata/key.rs | 34 +++++++ .../src/metadata/multipart.rs | 6 ++ .../src/metadata/multipart_repository.rs | 4 +- .../metadata/multipart_repository/listing.rs | 98 +++++++++++++++++++ .../tests/multipart_repository_test.rs | 45 +++++++++ 7 files changed, 191 insertions(+), 4 deletions(-) diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 6cf2d5fd..7392dc76 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -42,8 +42,10 @@ Iceberg multipart path. uploads, parse bounded completion XML, emit compatible responses and errors. Preserve SigV4 authentication and existing basic routes. The repository now provides bounded, ordered ListParts pagination over current generations; - multipart query shapes are parsed separately. Upload listing and HTTP dispatch - remain pending. S3-compatible multipart error codes and the create, complete + multipart query shapes are parsed separately. Bounded upload listing now + paginates active sessions by key and upload ID, skipping terminal/expired + records and failing on scan-budget exhaustion; HTTP dispatch remains pending. + S3-compatible multipart error codes and the create, complete and ListParts XML response builders have focused tests. The bounded completion XML parser now has one implementation in access-server and is exposed by both the Iceberg and S3 protocol modules. The S3 HTTP path still needs wiring. diff --git a/lib/crowdb-access-s3/src/metadata.rs b/lib/crowdb-access-s3/src/metadata.rs index 0fc10cad..83ba4163 100644 --- a/lib/crowdb-access-s3/src/metadata.rs +++ b/lib/crowdb-access-s3/src/metadata.rs @@ -25,7 +25,7 @@ mod generated { pub use key::{BucketId, MetadataKey, MetadataKeyError, TenantId}; pub use multipart::{MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord}; pub use multipart_repository::{ - CompletionPart, MultipartPartPage, MultipartRepository, MultipartRepositoryError, + CompletionPart, MultipartPartPage, MultipartRepository, MultipartRepositoryError, MultipartUploadPage, }; pub use namespace::{BucketDeleteOutcome, BucketNamespace, BucketNamespaceError}; pub use record::{BucketNameRecord, MetadataRecordError, ObjectRecord}; diff --git a/lib/crowdb-access-s3/src/metadata/key.rs b/lib/crowdb-access-s3/src/metadata/key.rs index 8942ed44..73ef0ca0 100644 --- a/lib/crowdb-access-s3/src/metadata/key.rs +++ b/lib/crowdb-access-s3/src/metadata/key.rs @@ -172,6 +172,40 @@ impl MetadataKey { key } + /// Starts the upload interval whose object names share a byte prefix. + /// + /// # Errors + /// Rejects a prefix longer than an object key. + pub fn multipart_session_key_prefix( + tenant: &TenantId, + bucket: BucketId, + object_prefix: &[u8], + ) -> Result, MetadataKeyError> { + if object_prefix.len() > MAX_KEY_BYTES { + return Err(MetadataKeyError::TooLong("object key prefix")); + } + let mut key = Self::multipart_session_prefix(tenant, bucket); + append_ordered_bytes_prefix(&mut key, object_prefix); + Ok(key) + } + + /// Ends the upload interval whose object names share a byte prefix. + /// + /// # Errors + /// Rejects a prefix longer than an object key. + pub fn multipart_session_key_prefix_end( + tenant: &TenantId, + bucket: BucketId, + object_prefix: &[u8], + ) -> Result, MetadataKeyError> { + if object_prefix.is_empty() { + return Ok(Self::multipart_session_end(tenant, bucket)); + } + let mut end = Self::multipart_session_key_prefix(tenant, bucket, object_prefix)?; + increment_lexicographic(&mut end); + Ok(end) + } + /// Identifies one upload by object key and stable upload ID. /// /// # Errors diff --git a/lib/crowdb-access-s3/src/metadata/multipart.rs b/lib/crowdb-access-s3/src/metadata/multipart.rs index 559e5753..0dcd2bfa 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart.rs @@ -92,6 +92,12 @@ impl MultipartSessionRecord { Ok(record) } + pub(crate) fn decode_unbound(bytes: &[u8]) -> Result { + let record: Self = decode(SESSION_MAGIC, bytes)?; + record.validate()?; + Ok(record) + } + fn validate(&self) -> Result<(), MultipartRecordError> { if self.object_key.is_empty() || self.object_key.len() > MAX_OBJECT_KEY_BYTES diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index ef94f1b0..ce5bbbbb 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -16,7 +16,7 @@ mod publication; mod terminal; pub use completion::CompletionPart; -pub use listing::MultipartPartPage; +pub use listing::{MultipartPartPage, MultipartUploadPage}; #[derive(Debug, thiserror::Error)] pub enum MultipartRepositoryError { @@ -32,6 +32,8 @@ pub enum MultipartRepositoryError { InvalidPart, #[error("a nonfinal multipart part is smaller than 5 MiB")] EntityTooSmall, + #[error("multipart listing exhausted its bounded scan budget")] + ScanBudgetExhausted, } pub struct MultipartRepository { diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs index d1a08121..5926247d 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs @@ -7,16 +7,114 @@ use super::{ MetadataKey, MultipartPartRecord, MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, }; +use crate::metadata::BucketId; const MAX_LIST_PARTS: usize = 1_000; const MAX_SCAN_BYTES: usize = 4 * 1024 * 1024; +const MAX_UPLOAD_SCAN_PAGES: usize = 8; +const MAX_UPLOAD_SCAN_ITEMS: usize = 4_096; pub struct MultipartPartPage { pub parts: Vec, pub next_part_number_marker: Option, } +pub struct MultipartUploadPage { + pub uploads: Vec, + pub next: Option<(Vec, [u8; 16])>, +} + impl MultipartRepository { + /// Lists active uploads in object-key and upload-ID order with bounded scans. + /// + /// Callers must create upload IDs in initiation-time order for the same key. + /// A dense interval of terminal records returns a scan-budget error rather + /// than an incomplete success page. + /// + /// # Errors + /// Rejects invalid markers, corrupt records and exhausted scan budgets. + pub async fn list_uploads( + &self, + bucket: BucketId, + prefix: &[u8], + key_marker: Option<&[u8]>, + upload_marker: Option<&[u8; 16]>, + max_uploads: usize, + now_ms: u64, + ) -> Result { + if max_uploads == 0 + || max_uploads > MAX_LIST_PARTS + || (upload_marker.is_some() && key_marker.is_none()) + { + return Err(MultipartRepositoryError::Conflict); + } + let mut start = MetadataKey::multipart_session_key_prefix(&self.tenant, bucket, prefix)?; + let end = MetadataKey::multipart_session_key_prefix_end(&self.tenant, bucket, prefix)?; + if let Some(key) = key_marker { + let mut after = MetadataKey::multipart_session( + &self.tenant, + bucket, + key, + upload_marker.unwrap_or(&[u8::MAX; 16]), + )?; + after.push(0); + start = start.max(after); + } + if start >= end { + return Ok(MultipartUploadPage { + uploads: Vec::new(), + next: None, + }); + } + let mut uploads = Vec::with_capacity(max_uploads + 1); + let mut continuation = None; + let mut scanned = 0; + for _ in 0..MAX_UPLOAD_SCAN_PAGES { + let page = self + .store + .scan_page( + start.clone(), + end.clone(), + (MAX_UPLOAD_SCAN_ITEMS - scanned).min(MAX_LIST_PARTS), + MAX_SCAN_BYTES, + continuation, + ) + .await?; + scanned += page.items.len(); + for item in page.items { + let session = MultipartSessionRecord::decode_unbound(&item.value)?; + if session.bucket_id != bucket + || MetadataKey::multipart_session( + &self.tenant, + bucket, + &session.object_key, + &session.upload_id, + )? != item.key + { + return Err(MultipartRepositoryError::Conflict); + } + if session.phase == MultipartPhase::Open + && session.created_ms <= now_ms + && now_ms < session.expires_ms + { + uploads.push(session); + if uploads.len() > max_uploads { + uploads.pop(); + let last = uploads.last().ok_or(MultipartRepositoryError::Conflict)?; + let next = Some((last.object_key.clone(), last.upload_id)); + return Ok(MultipartUploadPage { uploads, next }); + } + } + } + match page.continuation { + None => return Ok(MultipartUploadPage { uploads, next: None }), + Some(next) if scanned < MAX_UPLOAD_SCAN_ITEMS => continuation = Some(next), + Some(_) => return Err(MultipartRepositoryError::ScanBudgetExhausted), + } + } + Err(MultipartRepositoryError::ScanBudgetExhausted) + } + /// Lists current, visible part generations in ascending part-number order. /// /// # Errors diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs index 730d20b7..e2c3077b 100644 --- a/lib/crowdb-access-s3/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -564,3 +564,48 @@ async fn part_listing_paginates_current_generations_in_number_order() { ); assert_eq!(second.next_part_number_marker, None); } + +#[tokio::test] +async fn upload_listing_filters_terminal_records_and_resumes_same_key() { + let (repository, _, _) = repository().await; + let mut first = session(); + first.object_key = b"pre/a".to_vec(); + first.upload_id = [1; 16]; + let mut second = first.clone(); + second.upload_id = [2; 16]; + second.created_ms = 101; + let mut third = first.clone(); + third.object_key = b"pre/b".to_vec(); + third.upload_id = [3; 16]; + let mut terminal = first.clone(); + terminal.object_key = b"pre/aborted".to_vec(); + terminal.upload_id = [4; 16]; + for upload in [&first, &second, &third, &terminal] { + repository.begin(upload).await.unwrap(); + } + repository.abort(&terminal).await.unwrap(); + + let page = repository + .list_uploads(first.bucket_id, b"pre/", None, None, 1, 110) + .await + .unwrap(); + assert_eq!(page.uploads, [first.clone()]); + assert_eq!(page.next, Some((first.object_key.clone(), first.upload_id))); + let (key, id) = page.next.unwrap(); + let page = repository + .list_uploads(first.bucket_id, b"pre/", Some(&key), Some(&id), 1, 110) + .await + .unwrap(); + assert_eq!(page.uploads, [second.clone()]); + assert_eq!(page.next, Some((second.object_key.clone(), second.upload_id))); + let page = repository + .list_uploads(first.bucket_id, b"pre/", Some(b"pre/a"), None, 10, 110) + .await + .unwrap(); + assert_eq!(page.uploads, [third]); + assert!(page.next.is_none()); + assert!(repository + .list_uploads(first.bucket_id, b"pre/", None, Some(&id), 1, 110) + .await + .is_err()); +} From 16f1b91c9d195e2f4a407b2d97fdd60e41332e33 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:34:44 +0800 Subject: [PATCH 16/74] Add ordered multipart IDs and upload list XML --- doc/working/plan-s3-multipart.md | 5 +- lib/crowdb-access-s3/src/metadata.rs | 4 +- .../src/metadata/multipart.rs | 9 ++++ lib/crowdb-access-s3/src/wire.rs | 52 ++++++++++++++++++- .../tests/metadata_multipart_test.rs | 12 ++++- lib/crowdb-access-s3/tests/wire_test.rs | 43 ++++++++++++++- 6 files changed, 119 insertions(+), 6 deletions(-) diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 7392dc76..e24ffb82 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -45,8 +45,9 @@ Iceberg multipart path. multipart query shapes are parsed separately. Bounded upload listing now paginates active sessions by key and upload ID, skipping terminal/expired records and failing on scan-budget exhaustion; HTTP dispatch remains pending. - S3-compatible multipart error codes and the create, complete - and ListParts XML response builders have focused tests. The bounded completion + Upload IDs now sort by initiation millisecond. S3-compatible multipart error + codes and the create, complete, ListParts, and ListMultipartUploads XML + response builders have focused tests. The bounded completion XML parser now has one implementation in access-server and is exposed by both the Iceberg and S3 protocol modules. The S3 HTTP path still needs wiring. - [ ] **Part ingestion**: reuse the bounded streaming writer and admission diff --git a/lib/crowdb-access-s3/src/metadata.rs b/lib/crowdb-access-s3/src/metadata.rs index 83ba4163..0f56a4e7 100644 --- a/lib/crowdb-access-s3/src/metadata.rs +++ b/lib/crowdb-access-s3/src/metadata.rs @@ -23,7 +23,9 @@ mod generated { } pub use key::{BucketId, MetadataKey, MetadataKeyError, TenantId}; -pub use multipart::{MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord}; +pub use multipart::{ + new_upload_id, MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord, +}; pub use multipart_repository::{ CompletionPart, MultipartPartPage, MultipartRepository, MultipartRepositoryError, MultipartUploadPage, }; diff --git a/lib/crowdb-access-s3/src/metadata/multipart.rs b/lib/crowdb-access-s3/src/metadata/multipart.rs index 0dcd2bfa..3042bb18 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart.rs @@ -19,6 +19,15 @@ const MAX_CONTENT_TYPE_BYTES: usize = 1024; pub use crowdb_access_multipart::MultipartPhase; +/// Creates a random upload identity whose byte order follows initiation time. +/// Uploads created in the same millisecond have an unspecified relative order. +#[must_use] +pub fn new_upload_id(now_ms: u64) -> [u8; 16] { + let mut id = *uuid::Uuid::new_v4().as_bytes(); + id[..8].copy_from_slice(&now_ms.to_be_bytes()); + id +} + #[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] pub struct MultipartSessionRecord { pub bucket_id: BucketId, diff --git a/lib/crowdb-access-s3/src/wire.rs b/lib/crowdb-access-s3/src/wire.rs index c8ac08ad..0fcffd59 100644 --- a/lib/crowdb-access-s3/src/wire.rs +++ b/lib/crowdb-access-s3/src/wire.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crate::metadata::{BucketNameRecord, MultipartPartPage, ObjectRecord}; +use crate::metadata::{BucketNameRecord, MultipartPartPage, MultipartUploadPage, ObjectRecord}; use crate::object::ListObjectsV2Page; #[must_use] @@ -86,6 +86,56 @@ pub fn list_multipart_parts( output } +#[must_use] +pub fn list_multipart_uploads( + bucket: &[u8], + prefix: &[u8], + key_marker: Option<&[u8]>, + upload_marker: Option<&[u8; 16]>, + max_uploads: usize, + owner_id: &str, + page: &MultipartUploadPage, +) -> String { + let mut output = xml_start("ListMultipartUploadsResult"); + element(&mut output, "Bucket", &String::from_utf8_lossy(bucket)); + element( + &mut output, + "KeyMarker", + &String::from_utf8_lossy(key_marker.unwrap_or_default()), + ); + element( + &mut output, + "UploadIdMarker", + &upload_marker.map(hex_upload_id).unwrap_or_default(), + ); + if let Some((key, id)) = &page.next { + element(&mut output, "NextKeyMarker", &String::from_utf8_lossy(key)); + element(&mut output, "NextUploadIdMarker", &hex_upload_id(id)); + } + element(&mut output, "Prefix", &String::from_utf8_lossy(prefix)); + element(&mut output, "MaxUploads", &max_uploads.to_string()); + element( + &mut output, + "IsTruncated", + if page.next.is_some() { "true" } else { "false" }, + ); + for session in &page.uploads { + output.push_str(""); + element(&mut output, "Key", &String::from_utf8_lossy(&session.object_key)); + element(&mut output, "UploadId", &hex_upload_id(&session.upload_id)); + output.push_str(""); + element(&mut output, "ID", owner_id); + output.push_str(""); + element(&mut output, "ID", owner_id); + output.push_str(""); + element(&mut output, "StorageClass", "STANDARD"); + element(&mut output, "Initiated", &iso8601(session.created_ms)); + output.push_str(""); + } + output.push_str(""); + output +} + fn hex_upload_id(upload_id: &[u8; 16]) -> String { use std::fmt::Write as _; let mut result = String::with_capacity(32); diff --git a/lib/crowdb-access-s3/tests/metadata_multipart_test.rs b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs index fe46fa54..c5a03bf8 100644 --- a/lib/crowdb-access-s3/tests/metadata_multipart_test.rs +++ b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs @@ -3,7 +3,8 @@ use crowdb_access_multipart::SelectedPart; use crowdb_access_s3::metadata::{ - BucketId, MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord, + new_upload_id, BucketId, MultipartPartRecord, MultipartPhase, MultipartRecordError, + MultipartSessionRecord, }; use crowdb_protocol::chunkdb::rpc::Location; use crowdb_protocol::common::ChunkId; @@ -51,6 +52,15 @@ fn part() -> MultipartPartRecord { } } +#[test] +fn generated_upload_ids_sort_by_initiation_millisecond() { + let first = new_upload_id(100); + let second = new_upload_id(101); + assert!(first < second); + assert_ne!(first, [0; 16]); + assert_ne!(second, [0; 16]); +} + #[test] fn session_and_part_records_round_trip_only_under_their_own_keys() { let session = session(); diff --git a/lib/crowdb-access-s3/tests/wire_test.rs b/lib/crowdb-access-s3/tests/wire_test.rs index b0dff769..e04f3481 100644 --- a/lib/crowdb-access-s3/tests/wire_test.rs +++ b/lib/crowdb-access-s3/tests/wire_test.rs @@ -2,7 +2,8 @@ // Licensed under the Apache License, Version 2.0. use crowdb_access_s3::metadata::{ - BucketId, BucketNameRecord, MultipartPartPage, MultipartPartRecord, ObjectRecord, TenantId, + BucketId, BucketNameRecord, MultipartPartPage, MultipartPartRecord, MultipartPhase, + MultipartSessionRecord, MultipartUploadPage, ObjectRecord, TenantId, }; use crowdb_access_s3::object::ListObjectsV2Page; use crowdb_access_s3::wire; @@ -79,3 +80,43 @@ fn multipart_xml_escapes_names_and_reports_selected_part_metadata() { assert!(completed.contains("http://host/b&/k<")); assert!(completed.contains(""abc-1"")); } + +#[test] +fn multipart_upload_listing_emits_stable_markers_and_initiation_time() { + let id = [0xab; 16]; + let session = MultipartSessionRecord { + bucket_id: BucketId::new([1; 16]), + object_key: b"prefix/k&".to_vec(), + upload_id: id, + revision: 1, + phase: MultipartPhase::Open, + created_ms: 1_000, + expires_ms: 2_000, + content_type: "application/octet-stream".into(), + max_parts: 1, + max_part_bytes: 5, + max_object_bytes: 5, + max_staged_bytes: 5, + part_count: 0, + staged_bytes: 0, + selection: None, + completion_request_digest: None, + publication_ms: None, + object_predecessor: None, + etag: None, + }; + let page = MultipartUploadPage { + uploads: vec![session], + next: Some((b"prefix/k&".to_vec(), id)), + }; + let xml = wire::list_multipart_uploads(b"bucket", b"prefix/", None, None, 1, "owner&", &page); + assert!(xml.contains("prefix/k&")); + assert!(xml.contains("prefix/k&")); + assert!(xml.contains(&format!( + "{}", + "ab".repeat(16) + ))); + assert!(xml.contains("1970-01-01T00:00:01Z")); + assert!(xml.contains("owner&")); + assert!(xml.contains("true")); +} From b69ee84c0634bd9a895a97275f2d9c2ed794727d Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:35:30 +0800 Subject: [PATCH 17/74] Keep crash diagnostics decisions in backlog issues --- doc/backlog/backlog.md | 4 +++- doc/working/plan-console-authority.md | 14 -------------- 2 files changed, 3 insertions(+), 15 deletions(-) diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index b8dc1f51..8157d807 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -257,7 +257,9 @@ must be deleted — see the workflow's Post-merge cleanup section. with file-based collection or a verified host-collector export workflow. Exact-build source-line symbols also need a distribution choice: compressed line tables in the image with a measured size increase, or separate - exact-build debug symbols. Retention and symbolization remain unverified. + exact-build debug symbols. The all-dependency symbol experiment enlarged the + monitor substantially; a complete-image measurement remains pending. + Bounded volume retention and source-line symbolization remain unverified. - **Server Tests intermittent strip lookup:** The CI log for `small_object_writer_e2e::eight_closed_mirror_strips_become_one_durable_ec_strip_without_reread` failed at `location strip`, while the exact test and its 16-test suite pass diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index f6165317..9aefa5a1 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -277,17 +277,3 @@ is paused; it does not block the single-node image requirement. collector/export and exact-build source-line symbolization acceptance through `pixi run test-monitor` and `pixi run test-single-node-container`. - Style: `pixi run rs-fmt-check` and `pixi run rs-lint`. - -## Open Questions - -- **Crash collection and symbols:** the current host routes `core_pattern` to - Apport. A container-local file directory/ulimit cannot override that policy, - and changing the host-wide collector is outside container implementation. - Choose acceptance on a disposable Linux host with file-based core collection, - or certify and document a host-collector export workflow. Source-line symbol - distribution also needs a choice: bundle compressed CROWDB line tables and - adjust the measured image-size ceiling, or publish exact-build debug symbols - separately while retaining runtime function names. The existing all-dependency - experiment increased monitor size substantially; neither complete-image option - has yet been measured. Bounded volume retention and end-to-end source-line - symbolization remain incomplete, not claimed acceptance. From 51c66303a1182564ec6298a57024abda5ac0b129 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:38:18 +0800 Subject: [PATCH 18/74] Record missing server failure artifact --- doc/backlog/backlog.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index 8157d807..7fb10795 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -263,9 +263,11 @@ must be deleted — see the workflow's Post-merge cleanup section. - **Server Tests intermittent strip lookup:** The CI log for `small_object_writer_e2e::eight_closed_mirror_strips_become_one_durable_ec_strip_without_reread` failed at `location strip`, while the exact test and its 16-test suite pass - locally. Preserve the failing run's runtime-server logs and artifact to - identify the first divergent state before changing the assertion or retry - policy. + locally. CI run `36449749925` completed its "Upload test logs on failure" + step, but the run artifact list contains only `docker-preview-1`; the + `runtime-server` artifact is absent. Preserve runtime logs on the next + failure to identify the first divergent state before changing the assertion + or retry policy. - **R188 hardware display data:** Group 0 rack and node values hold IDs and status but not the console's rack name, node host or SSH settings. The authority cutover must define where shared display names live and keep From f759b256cef73faba7b24e7a6003a50d1b0d979f Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 02:58:43 +0800 Subject: [PATCH 19/74] Record full server test reproduction result --- doc/backlog/backlog.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index 7fb10795..ce7a3da9 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -262,8 +262,9 @@ must be deleted — see the workflow's Post-merge cleanup section. Bounded volume retention and source-line symbolization remain unverified. - **Server Tests intermittent strip lookup:** The CI log for `small_object_writer_e2e::eight_closed_mirror_strips_become_one_durable_ec_strip_without_reread` - failed at `location strip`, while the exact test and its 16-test suite pass - locally. CI run `36449749925` completed its "Upload test logs on failure" + failed at `location strip`, while the exact test, its 16-test suite, and the + full `test-server` task all pass locally at default test concurrency. CI run + `36449749925` completed its "Upload test logs on failure" step, but the run artifact list contains only `docker-preview-1`; the `runtime-server` artifact is absent. Preserve runtime logs on the next failure to identify the first divergent state before changing the assertion From a5e1bb55b07d869e0a27992d06439e70d1586736 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 03:02:52 +0800 Subject: [PATCH 20/74] Reject orphan multipart part selectors --- lib/crowdb-access-s3/src/route.rs | 1 + lib/crowdb-access-s3/src/route/multipart.rs | 3 +++ lib/crowdb-access-s3/tests/route_test.rs | 8 ++++++++ 3 files changed, 12 insertions(+) diff --git a/lib/crowdb-access-s3/src/route.rs b/lib/crowdb-access-s3/src/route.rs index 970dcea6..de710f28 100644 --- a/lib/crowdb-access-s3/src/route.rs +++ b/lib/crowdb-access-s3/src/route.rs @@ -126,6 +126,7 @@ fn selects_extension(query: Option<&str>) -> bool { name, "uploads" | "uploadId" + | "partNumber" | "versionId" | "tagging" | "lifecycle" diff --git a/lib/crowdb-access-s3/src/route/multipart.rs b/lib/crowdb-access-s3/src/route/multipart.rs index 31243522..3968b4c9 100644 --- a/lib/crowdb-access-s3/src/route/multipart.rs +++ b/lib/crowdb-access-s3/src/route/multipart.rs @@ -51,6 +51,9 @@ pub fn classify_multipart(method: &Method, uri: &Uri) -> Result {} } } + if !uploads && upload_id.is_none() && part_number.is_some() { + return Err(RouteError::Invalid); + } if !uploads && upload_id.is_none() { return Ok(None); } diff --git a/lib/crowdb-access-s3/tests/route_test.rs b/lib/crowdb-access-s3/tests/route_test.rs index a1917c50..0db9b5b9 100644 --- a/lib/crowdb-access-s3/tests/route_test.rs +++ b/lib/crowdb-access-s3/tests/route_test.rs @@ -34,6 +34,10 @@ fn rejects_extensions_before_dispatch() { classify(&Method::POST, &"/bucket/key?uploads".parse().unwrap()), Err(RouteError::NotImplemented) ); + assert_eq!( + classify(&Method::PUT, &"/bucket/key?partNumber=7".parse().unwrap()), + Err(RouteError::NotImplemented) + ); } #[test] @@ -107,4 +111,8 @@ fn multipart_queries_have_unambiguous_paths_and_identities() { classify_multipart(&Method::GET, &"/bucket?uploads&uploads".parse().unwrap()), Err(RouteError::Invalid) ); + assert_eq!( + classify_multipart(&Method::PUT, &"/bucket/key?partNumber=7".parse().unwrap()), + Err(RouteError::Invalid) + ); } From 24ce63d5280f2c17c98aa8cdedfbb1e8c5123076 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 03:05:04 +0800 Subject: [PATCH 21/74] Confirm identical durable multipart part retries --- doc/working/plan-s3-multipart.md | 6 ++++-- .../src/metadata/multipart_repository.rs | 8 ++++++++ .../tests/multipart_repository_test.rs | 15 +++++++++++++-- 3 files changed, 25 insertions(+), 4 deletions(-) diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index e24ffb82..1072407f 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -35,8 +35,10 @@ Iceberg multipart path. as namespace scope; preserve immutable part data after replacement. Versioned session/part records and ordered, binary-safe keys are in place; CAS-backed begin, phase transition and part replacement now use exact-value - confirmation after lost replies. Completion snapshots and a predecessor-fenced - metadata-only object publication path are in place. HTTP wiring and cleanup + confirmation after lost replies. An identical part record retry returns the + existing revision; a new location remains a replacement. Completion + snapshots and a predecessor-fenced metadata-only object publication path are + in place. HTTP wiring and cleanup state remain. - [ ] **S3 routes and wire**: classify create/upload/list/complete/abort/list uploads, parse bounded completion XML, emit compatible responses and errors. diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index ce5bbbbb..a9c5372b 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -216,6 +216,14 @@ impl MultipartRepository { return Err(MultipartRepositoryError::Conflict); } let before = self.part(¤t, part.number).await?; + if let Some(existing) = &before { + if existing.length == part.length + && existing.raw_md5 == part.raw_md5 + && existing.locations == part.locations + { + return Ok(Some(existing.clone())); + } + } let mut after = part.clone(); after.revision = before .as_ref() diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs index e2c3077b..d8987b1d 100644 --- a/lib/crowdb-access-s3/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -296,11 +296,19 @@ async fn session_cas_and_independent_part_replacement_obey_the_freeze() { .unwrap() .unwrap(); assert_eq!(first.revision, 1); - let second = repository + let replay = repository .put_stream_part(&session, &part(), 111) .await .unwrap() .unwrap(); + assert_eq!(replay, first); + let mut replacement = part(); + replacement.locations[0].offset += 39; + let second = repository + .put_stream_part(&session, &replacement, 112) + .await + .unwrap() + .unwrap(); assert_eq!(second.revision, 2); assert_eq!(repository.part(&session, 1).await.unwrap(), Some(second)); assert_eq!( @@ -470,8 +478,10 @@ async fn frozen_part_generation_survives_a_late_pointer_change() { let session = session(); repository.begin(&session).await.unwrap(); repository.put_stream_part(&session, &part(), 110).await.unwrap(); + let mut replacement = part(); + replacement.locations[0].offset += 39; let selected = repository - .put_stream_part(&session, &part(), 111) + .put_stream_part(&session, &replacement, 111) .await .unwrap() .unwrap(); @@ -546,6 +556,7 @@ async fn part_listing_paginates_current_generations_in_number_order() { } let mut replacement = part(); replacement.number = 2; + replacement.locations[0].offset += 39; repository .put_stream_part(&session, &replacement, 111) .await From 65b60b6e2a8371c29b0c0c0006cbc50adf83d2c5 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 03:05:25 +0800 Subject: [PATCH 22/74] Track multipart replay cleanup gap --- doc/backlog/backlog.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index ce7a3da9..334f5153 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -283,7 +283,9 @@ must be deleted — see the workflow's Post-merge cleanup section. wiring S3; keep keys, authorization and responses in the protocol adapters. - **R167 unreachable part cleanup:** S3 now preserves immutable part generations so completion can publish a selected generation across a - concurrent part-number replacement. Losing replacement candidates and old + concurrent part-number replacement. An identical durable part record retry + keeps its revision, but an HTTP retry may stream the same bytes to a new + location. Losing replacement candidates, replayed stream locations and old generations can remain unreachable. Abort, expiry, and replacement cleanup need durable bounded records and reader-pin protection before the HTTP path is enabled. From f0e62a700354f9aa46995b90eaa2b2bb9a4cd712 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 06:30:08 +0800 Subject: [PATCH 23/74] Place open issues in owning requirements --- doc/backlog/R148-chunk-stream-scale-out.md | 11 ++++++ doc/backlog/R167-s3-multipart-upload.md | 16 ++++++++ doc/backlog/R188-console-group0-authority.md | 17 ++++++++ doc/backlog/backlog.md | 41 -------------------- 4 files changed, 44 insertions(+), 41 deletions(-) diff --git a/doc/backlog/R148-chunk-stream-scale-out.md b/doc/backlog/R148-chunk-stream-scale-out.md index c3341634..92b96fe5 100644 --- a/doc/backlog/R148-chunk-stream-scale-out.md +++ b/doc/backlog/R148-chunk-stream-scale-out.md @@ -140,3 +140,14 @@ Required gates: - `pixi run -- cargo test -p crowdb-chunk-kv --all-targets` - `pixi run -- cargo test -p crowdb-chunkdb --all-targets` - `pixi run clean-env && pixi run test-server` + +## Open Issues + +- The CI log for + `small_object_writer_e2e::eight_closed_mirror_strips_become_one_durable_ec_strip_without_reread` + failed at `location strip`, while the exact test, its 16-test suite and the + full `test-server` task pass locally at default test concurrency. CI run + `36449749925` completed its "Upload test logs on failure" step, but the run + artifact list contains only `docker-preview-1`; the `runtime-server` artifact + is absent. Preserve runtime logs on the next failure to identify the first + divergent state before changing the assertion or retry policy. diff --git a/doc/backlog/R167-s3-multipart-upload.md b/doc/backlog/R167-s3-multipart-upload.md index c7927bd9..e5bd52d8 100644 --- a/doc/backlog/R167-s3-multipart-upload.md +++ b/doc/backlog/R167-s3-multipart-upload.md @@ -78,3 +78,19 @@ Required gates: - `pixi run -- cargo test -p crowdb-access-server --all-targets` - `pixi run -- cargo fmt --all -- --check` - `pixi run rs-lint` + +## Open Issues + +- The referenced R190 requirement is no longer present in the backlog. Both + adapters now use shared phase names, selected-part validation, accounting and + metadata-only location composition. Iceberg's remaining session and part + recovery is bound to its catalog identity and store. Extract the remaining + protocol-neutral transition decisions while keeping keys, authorization and + responses in the protocol adapters. +- S3 preserves immutable part generations so completion can publish a selected + generation across concurrent part-number replacement. An identical durable + part record retry keeps its revision, but an HTTP retry may stream the same + bytes to a new location. Losing replacement candidates, replayed stream + locations and old generations can remain unreachable. Abort, expiry and + replacement cleanup need durable bounded records and reader-pin protection + before the HTTP path is enabled. diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md index ee422080..58075a71 100644 --- a/doc/backlog/R188-console-group0-authority.md +++ b/doc/backlog/R188-console-group0-authority.md @@ -153,3 +153,20 @@ Required gates: - `pixi run test-single-node-container` - `pixi run rs-fmt-check` - `pixi run rs-lint` + +## Open Issues + +- This host routes `core_pattern` to Apport, so a container-local directory and + core ulimit cannot guarantee a dump in `/opt/crowdb/data`. End-to-end + acceptance needs a disposable host with file-based collection or a verified + host-collector export workflow. Exact-build source-line symbols also need a + distribution choice: compressed line tables in the image with a measured + size increase, or separate exact-build debug symbols. The all-dependency + symbol experiment enlarged the monitor substantially; a complete-image + measurement remains pending. Bounded volume retention and source-line + symbolization remain unverified. +- Group 0 rack and node values hold IDs and status but not the console's rack + name, node host or SSH settings. The authority cutover must define where + shared display names live and keep machine-local launch inputs in the launch + registry. Until that split is implemented, two consoles cannot reconstruct + identical physical views from Group 0 alone. diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index 334f5153..4ff88de4 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -248,44 +248,3 @@ must be deleted — see the workflow's Post-merge cleanup section. --- - -## Open Issues - -- **R188 crash diagnostics acceptance:** This host routes `core_pattern` to - Apport, so a container-local directory and core ulimit cannot guarantee a - dump in `/opt/crowdb/data`. End-to-end acceptance needs a disposable host - with file-based collection or a verified host-collector export workflow. - Exact-build source-line symbols also need a distribution choice: compressed - line tables in the image with a measured size increase, or separate - exact-build debug symbols. The all-dependency symbol experiment enlarged the - monitor substantially; a complete-image measurement remains pending. - Bounded volume retention and source-line symbolization remain unverified. -- **Server Tests intermittent strip lookup:** The CI log for - `small_object_writer_e2e::eight_closed_mirror_strips_become_one_durable_ec_strip_without_reread` - failed at `location strip`, while the exact test, its 16-test suite, and the - full `test-server` task all pass locally at default test concurrency. CI run - `36449749925` completed its "Upload test logs on failure" - step, but the run artifact list contains only `docker-preview-1`; the - `runtime-server` artifact is absent. Preserve runtime logs on the next - failure to identify the first divergent state before changing the assertion - or retry policy. -- **R188 hardware display data:** Group 0 rack and node values hold IDs and - status but not the console's rack name, node host or SSH settings. The - authority cutover must define where shared display names live and keep - machine-local launch inputs in the launch registry. Until that split is - implemented, two consoles cannot reconstruct identical physical views from - Group 0 alone. -- **R167 shared multipart core:** The referenced R190 requirement is no - longer present in the backlog. Both adapters now use shared phase names, - selected-part validation, accounting and metadata-only location composition. - Iceberg's remaining session/part recovery is bound to its catalog identity - and store. Extract the remaining protocol-neutral transition decisions before - wiring S3; keep keys, authorization and responses in the protocol adapters. -- **R167 unreachable part cleanup:** S3 now preserves immutable part - generations so completion can publish a selected generation across a - concurrent part-number replacement. An identical durable part record retry - keeps its revision, but an HTTP retry may stream the same bytes to a new - location. Losing replacement candidates, replayed stream locations and old - generations can remain unreachable. Abort, expiry, and replacement cleanup - need durable bounded records and reader-pin protection before the HTTP path - is enabled. From 5e668d04ab02666b9154535af9fa942238215598 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 06:39:02 +0800 Subject: [PATCH 24/74] Route authenticated S3 multipart requests --- app/crowdb-access-server/src/s3/dispatcher.rs | 5 +++- app/crowdb-access-server/src/s3/operations.rs | 6 ++++ lib/crowdb-access-s3/src/metrics.rs | 8 ++++- lib/crowdb-access-s3/src/route.rs | 29 +++++++++++++++++++ lib/crowdb-access-s3/tests/metrics_test.rs | 2 +- lib/crowdb-access-s3/tests/route_test.rs | 9 ++++-- 6 files changed, 53 insertions(+), 6 deletions(-) diff --git a/app/crowdb-access-server/src/s3/dispatcher.rs b/app/crowdb-access-server/src/s3/dispatcher.rs index 90cd1458..5459b8cd 100644 --- a/app/crowdb-access-server/src/s3/dispatcher.rs +++ b/app/crowdb-access-server/src/s3/dispatcher.rs @@ -296,7 +296,10 @@ fn defer_body_provider( factory: Option DeferredBodyReceiveProvider + Send + Sync>>, request: &mut Request, ) { - if operation == crowdb_access_s3::route::S3Operation::PutObject { + if matches!( + operation, + crowdb_access_s3::route::S3Operation::PutObject | crowdb_access_s3::route::S3Operation::UploadPart + ) { if let Some(factory) = factory { request.extensions_mut().insert(factory()); } diff --git a/app/crowdb-access-server/src/s3/operations.rs b/app/crowdb-access-server/src/s3/operations.rs index 8e661662..b86313d2 100644 --- a/app/crowdb-access-server/src/s3/operations.rs +++ b/app/crowdb-access-server/src/s3/operations.rs @@ -143,6 +143,12 @@ impl ProductionS3Operations { S3Operation::GetObject => self.get_object(route, &request, &request_id).await, S3Operation::ListObjectsV2 => self.list_objects(route, &request).await, S3Operation::DeleteObject => self.delete_object(route).await, + S3Operation::CreateMultipartUpload + | S3Operation::UploadPart + | S3Operation::ListParts + | S3Operation::CompleteMultipartUpload + | S3Operation::AbortMultipartUpload + | S3Operation::ListMultipartUploads => Err(S3ErrorCode::NotImplemented), }; self.record_dependency_outcome(operation, &result); result.unwrap_or_else(|code| { diff --git a/lib/crowdb-access-s3/src/metrics.rs b/lib/crowdb-access-s3/src/metrics.rs index 4084ab3c..473430e6 100644 --- a/lib/crowdb-access-s3/src/metrics.rs +++ b/lib/crowdb-access-s3/src/metrics.rs @@ -13,7 +13,7 @@ use crowdb_chunk_client::{ use crate::native_buffer::{NativeBodyAllocator, NativeBufferMetricsSnapshot}; -const OPERATION_COUNT: usize = 9; +const OPERATION_COUNT: usize = 15; const OUTCOME_COUNT: usize = 6; const OPERATION_NAMES: [&str; OPERATION_COUNT] = [ "create_bucket", @@ -25,6 +25,12 @@ const OPERATION_NAMES: [&str; OPERATION_COUNT] = [ "get_object", "list_objects_v2", "delete_object", + "create_multipart_upload", + "upload_part", + "list_parts", + "complete_multipart_upload", + "abort_multipart_upload", + "list_multipart_uploads", ]; const OUTCOME_NAMES: [&str; OUTCOME_COUNT] = [ "success", diff --git a/lib/crowdb-access-s3/src/route.rs b/lib/crowdb-access-s3/src/route.rs index de710f28..afe295a3 100644 --- a/lib/crowdb-access-s3/src/route.rs +++ b/lib/crowdb-access-s3/src/route.rs @@ -22,6 +22,12 @@ pub enum S3Operation { GetObject, ListObjectsV2, DeleteObject, + CreateMultipartUpload, + UploadPart, + ListParts, + CompleteMultipartUpload, + AbortMultipartUpload, + ListMultipartUploads, } #[derive(Clone, Debug, Eq, PartialEq)] @@ -29,6 +35,8 @@ pub struct S3Route { pub operation: S3Operation, pub bucket: Option>, pub key: Option>, + pub upload_id: Option<[u8; 16]>, + pub part_number: Option, } #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -54,6 +62,8 @@ pub fn classify(method: &Method, uri: &Uri) -> Result { operation: S3Operation::ListBuckets, bucket: None, key: None, + upload_id: None, + part_number: None, }) .ok_or(RouteError::Invalid); } @@ -80,6 +90,8 @@ pub fn classify(method: &Method, uri: &Uri) -> Result { operation, bucket: Some(bucket), key, + upload_id: None, + part_number: None, }) } @@ -111,6 +123,23 @@ pub fn classify_request(method: &Method, uri: &Uri, headers: &HeaderMap) -> Resu }) { return Err(RouteError::NotImplemented); } + if let Some(multipart) = classify_multipart(method, uri)? { + let operation = match multipart.operation { + MultipartOperation::Create => S3Operation::CreateMultipartUpload, + MultipartOperation::UploadPart => S3Operation::UploadPart, + MultipartOperation::ListParts => S3Operation::ListParts, + MultipartOperation::Complete => S3Operation::CompleteMultipartUpload, + MultipartOperation::Abort => S3Operation::AbortMultipartUpload, + MultipartOperation::ListUploads => S3Operation::ListMultipartUploads, + }; + return Ok(S3Route { + operation, + bucket: Some(multipart.bucket), + key: multipart.key, + upload_id: multipart.upload_id, + part_number: multipart.part_number, + }); + } classify(method, uri) } diff --git a/lib/crowdb-access-s3/tests/metrics_test.rs b/lib/crowdb-access-s3/tests/metrics_test.rs index d74ae8d8..15c00eb2 100644 --- a/lib/crowdb-access-s3/tests/metrics_test.rs +++ b/lib/crowdb-access-s3/tests/metrics_test.rs @@ -93,7 +93,7 @@ fn exported_request_series_have_fixed_cardinality_and_no_namespace_labels() { .lines() .filter(|line| line.starts_with("crowdb_s3_requests_total{")) .count(), - 9 * 6 + 15 * 6 ); assert!(!rendered.contains("bucket=")); assert!(!rendered.contains("key=")); diff --git a/lib/crowdb-access-s3/tests/route_test.rs b/lib/crowdb-access-s3/tests/route_test.rs index 0db9b5b9..ab51f2b9 100644 --- a/lib/crowdb-access-s3/tests/route_test.rs +++ b/lib/crowdb-access-s3/tests/route_test.rs @@ -86,11 +86,14 @@ fn multipart_queries_have_unambiguous_paths_and_identities() { ), ]; for (method, uri, expected) in cases { - let route = classify_multipart(&method, &uri.parse().unwrap()) - .unwrap() - .unwrap(); + let uri = uri.parse().unwrap(); + let route = classify_multipart(&method, &uri).unwrap().unwrap(); assert_eq!(route.operation, expected); assert_eq!(route.bucket, b"bucket"); + let authenticated = classify_request(&method, &uri, &HeaderMap::new()).unwrap(); + assert_eq!(authenticated.bucket.as_deref(), Some(b"bucket".as_slice())); + assert_eq!(authenticated.upload_id, route.upload_id); + assert_eq!(authenticated.part_number, route.part_number); } assert_eq!( classify_multipart(&Method::GET, &"/bucket/key".parse().unwrap()), From 5311fff5f7c9eb97712f4ad0eba44ce31b39f68d Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 06:39:21 +0800 Subject: [PATCH 25/74] Track authenticated multipart dispatch boundary --- doc/working/plan-s3-multipart.md | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 1072407f..11506502 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -44,14 +44,16 @@ Iceberg multipart path. uploads, parse bounded completion XML, emit compatible responses and errors. Preserve SigV4 authentication and existing basic routes. The repository now provides bounded, ordered ListParts pagination over current generations; - multipart query shapes are parsed separately. Bounded upload listing now + multipart query shapes now enter the authenticated dispatcher with distinct + metrics, but production operations still return `NotImplemented`. Bounded + upload listing now paginates active sessions by key and upload ID, skipping terminal/expired records and failing on scan-budget exhaustion; HTTP dispatch remains pending. Upload IDs now sort by initiation millisecond. S3-compatible multipart error codes and the create, complete, ListParts, and ListMultipartUploads XML response builders have focused tests. The bounded completion XML parser now has one implementation in access-server and is exposed by both - the Iceberg and S3 protocol modules. The S3 HTTP path still needs wiring. + the Iceberg and S3 protocol modules. The S3 HTTP operations still need wiring. - [ ] **Part ingestion**: reuse the bounded streaming writer and admission budget, persist part location/integrity before success, reconcile lost replies. - [ ] **Atomic completion**: fence selected part generations, validate order, From 62dac685fb7833a97b992fcfb596278a2ec06fa9 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 07:29:07 +0800 Subject: [PATCH 26/74] Share access template in single-node profile --- container/crowdb-monitor/src/preview.rs | 13 ++++++++----- container/crowdb-monitor/src/render.rs | 18 ++++++++++++++---- container/crowdb-monitor/tests/render_test.rs | 9 ++++++++- .../tests/single_node_profile_test.rs | 17 +++++++++++++---- .../tests/storage_bootstrap_test.rs | 10 ++++++++-- container/single-node-container/profile.toml | 8 ++++---- 6 files changed, 55 insertions(+), 20 deletions(-) diff --git a/container/crowdb-monitor/src/preview.rs b/container/crowdb-monitor/src/preview.rs index ff72738d..67686a68 100644 --- a/container/crowdb-monitor/src/preview.rs +++ b/container/crowdb-monitor/src/preview.rs @@ -1,4 +1,4 @@ -use std::collections::{BTreeMap, BTreeSet}; +use std::collections::BTreeMap; use std::fs; use std::future::Future; use std::path::Path; @@ -478,20 +478,23 @@ fn kv_root(profile: &DeploymentProfile) -> Result Result, PreviewError> { - let mut files = BTreeSet::new(); + let mut files = BTreeMap::new(); for service in &profile.services { if let Some(path) = &service.config_template { let name = path .file_name() .ok_or(PreviewError::Invalid("template has no name"))?; - if !files.insert(name.to_os_string()) { + if files + .insert(name.to_os_string(), path) + .is_some_and(|existing| existing != path) + { return Err(PreviewError::Invalid("template name is duplicated")); } } } let mut input = Vec::new(); - for name in files { - let path = profile.paths.template_root.join(&name); + for name in files.keys() { + let path = profile.paths.template_root.join(name); let metadata = fs::symlink_metadata(&path)?; if !metadata.file_type().is_file() || metadata.len() > MAX_TEMPLATE_BYTES { return Err(PreviewError::Invalid("template is not a bounded regular file")); diff --git a/container/crowdb-monitor/src/render.rs b/container/crowdb-monitor/src/render.rs index 5ccc0503..afd02716 100644 --- a/container/crowdb-monitor/src/render.rs +++ b/container/crowdb-monitor/src/render.rs @@ -1,4 +1,4 @@ -use std::collections::{BTreeMap, BTreeSet}; +use std::collections::BTreeMap; use std::fs::{self, File, OpenOptions}; use std::io::Write; use std::os::unix::fs::OpenOptionsExt; @@ -45,7 +45,7 @@ pub fn render_configs( Err(error) => return Err(error.into()), } let mut outputs = Vec::new(); - let mut file_names = BTreeSet::new(); + let mut file_names = BTreeMap::new(); for service in &profile.services { let Some(template) = &service.config_template else { continue; @@ -53,9 +53,20 @@ pub fn render_configs( let name = template .file_name() .ok_or_else(|| RenderError::Invalid("template path has no file name".into()))?; - if !file_names.insert(name.to_os_string()) { + if file_names + .insert(name.to_os_string(), template) + .is_some_and(|existing| existing != template) + { return invalid("multiple services render to the same file name"); } + let path = destination.join(name); + if outputs.iter().any(|output: &RenderedConfig| output.path == path) { + outputs.push(RenderedConfig { + service_id: service.id.clone(), + path, + }); + continue; + } let source = template_root.join(name); let metadata = fs::symlink_metadata(&source)?; if !metadata.file_type().is_file() || metadata.len() > MAX_TEMPLATE_BYTES { @@ -69,7 +80,6 @@ pub fn render_configs( toml::from_str::(&rendered).map_err(|error| { RenderError::Invalid(format!("template for {} is not TOML: {error}", service.id)) })?; - let path = destination.join(name); atomic_write(&path, rendered.as_bytes())?; outputs.push(RenderedConfig { service_id: service.id.clone(), diff --git a/container/crowdb-monitor/tests/render_test.rs b/container/crowdb-monitor/tests/render_test.rs index 644a1867..1dcf470c 100644 --- a/container/crowdb-monitor/tests/render_test.rs +++ b/container/crowdb-monitor/tests/render_test.rs @@ -52,7 +52,14 @@ fn profile() -> DeploymentProfile { fn renders_profile_paths_and_topology_without_secrets() { let dirs = TestDirs::new(); let outputs = render_configs(&profile(), &dirs.templates(), &dirs.run()).unwrap(); - assert_eq!(outputs.len(), 6); + assert_eq!(outputs.len(), 8); + assert_eq!( + outputs + .iter() + .filter(|output| output.path == dirs.run().join("config/access.toml")) + .count(), + 2 + ); let diskio = fs::read_to_string(dirs.run().join("config/diskio.toml")).unwrap(); assert!(diskio.contains("path = \"/opt/crowdb/data/disks/disk-0004.img\"")); assert!(diskio.contains("zone_capacity = 17179869184")); diff --git a/container/crowdb-monitor/tests/single_node_profile_test.rs b/container/crowdb-monitor/tests/single_node_profile_test.rs index 302d2931..e651e82f 100644 --- a/container/crowdb-monitor/tests/single_node_profile_test.rs +++ b/container/crowdb-monitor/tests/single_node_profile_test.rs @@ -48,18 +48,27 @@ fn single_node_preview_has_exact_topology_and_endpoints() { iceberg.env.get("CROWDB_ICEBERG_PUBLIC_URI"), Some(&"http://localhost".to_owned()) ); + assert_eq!(iceberg.probe.target, "http://127.0.0.1:80/v1/config"); assert_eq!( - iceberg.env.get("CROWDB_ICEBERG_LISTEN"), - Some(&"0.0.0.0:80".to_owned()) + iceberg.env.get("CROWDB_MANAGEMENT_SEEDS"), + Some(&"http://127.0.0.1:10000".to_owned()) ); - assert_eq!(iceberg.probe.target, "http://127.0.0.1:80/v1/config"); let s3 = profile .services .iter() .find(|service| service.id == "s3") .unwrap(); - assert_eq!(s3.env.get("CROWDB_S3_LISTEN"), Some(&"0.0.0.0:81".to_owned())); assert_eq!(s3.probe.target, "http://127.0.0.1:81/_crowdb/health/ready"); + assert_eq!(s3.args[1], "/opt/crowdb/run/config/access.toml"); + assert_eq!(iceberg.args[2], s3.args[1]); + assert_eq!(iceberg.config_template, s3.config_template); + let access = std::fs::read_to_string( + Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/templates/access.toml"), + ) + .unwrap(); + let access: toml::Value = toml::from_str(&access).unwrap(); + assert_eq!(access["iceberg"]["listen"].as_str(), Some("0.0.0.0:80")); + assert_eq!(access["s3"]["listen"].as_str(), Some("0.0.0.0:81")); let web = profile .services .iter() diff --git a/container/crowdb-monitor/tests/storage_bootstrap_test.rs b/container/crowdb-monitor/tests/storage_bootstrap_test.rs index 3b57a9b0..33880603 100644 --- a/container/crowdb-monitor/tests/storage_bootstrap_test.rs +++ b/container/crowdb-monitor/tests/storage_bootstrap_test.rs @@ -84,7 +84,11 @@ impl TestRoot { .to_string_lossy() .into_owned(), ], - "iceberg" => vec!["serve".into()], + "iceberg" => vec![ + "serve".into(), + "--config".into(), + format!("{}/run/config/access.toml", self.0.display()), + ], _ => unreachable!(), }; if service.id == "iceberg" { @@ -135,6 +139,7 @@ impl TestRoot { "diskio.toml", "chunkdb.toml", "chunk-kv.toml", + "access.toml", ] { let body = fs::read_to_string(source.join(name)).unwrap(); let body = body @@ -149,7 +154,8 @@ impl TestRoot { .replace("127.0.0.1:12100", &format!("127.0.0.1:{}", ports.chunkdb_http)) .replace("127.0.0.1:12200", &format!("127.0.0.1:{}", ports.chunkdb_rpc)) .replace("127.0.0.1:15100", &format!("127.0.0.1:{}", ports.chunk_kv_http)) - .replace("127.0.0.1:15200", &format!("127.0.0.1:{}", ports.chunk_kv_rpc)); + .replace("127.0.0.1:15200", &format!("127.0.0.1:{}", ports.chunk_kv_rpc)) + .replace("0.0.0.0:80", &format!("127.0.0.1:{}", ports.iceberg)); fs::write(self.0.join("templates").join(name), body).unwrap(); } } diff --git a/container/single-node-container/profile.toml b/container/single-node-container/profile.toml index b3bf07dd..de9b299a 100644 --- a/container/single-node-container/profile.toml +++ b/container/single-node-container/profile.toml @@ -173,8 +173,8 @@ backoff_max_ms = 5000 [[services]] id = "s3" program = "/opt/crowdb/bin/crowdb-access-server" -args = ["--config", "/opt/crowdb/run/config/s3.toml"] -env = { CROWDB_S3_PUBLIC_URI = "http://localhost:81", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000" } +args = ["--config", "/opt/crowdb/run/config/access.toml"] +env = { CROWDB_S3_PUBLIC_URI = "http://localhost:81", CROWDB_S3_REGION = "us-east-1", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000" } dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] fence_listeners = ["127.0.0.1:81"] config_template = "/opt/crowdb/etc/templates/access.toml" @@ -191,8 +191,8 @@ backoff_max_ms = 5000 [[services]] id = "iceberg" program = "/opt/crowdb/bin/crowdb-iceberg" -args = ["serve", "--config", "/opt/crowdb/run/config/iceberg.toml"] -env = { CROWDB_ICEBERG_PUBLIC_URI = "http://localhost", CROWDB_ICEBERG_GC_ENABLED = "0" } +args = ["serve", "--config", "/opt/crowdb/run/config/access.toml"] +env = { CROWDB_ICEBERG_PUBLIC_URI = "http://localhost", CROWDB_ICEBERG_GC_ENABLED = "0", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000" } dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] fence_listeners = ["127.0.0.1:80"] config_template = "/opt/crowdb/etc/templates/access.toml" From 9f21125a2a534c6bfe1fa8fd6678a1944ca6fe83 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 07:29:11 +0800 Subject: [PATCH 27/74] Capture chunk strip lookup diagnostics --- doc/backlog/R148-chunk-stream-scale-out.md | 3 ++- lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs | 7 ++++++- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/doc/backlog/R148-chunk-stream-scale-out.md b/doc/backlog/R148-chunk-stream-scale-out.md index 92b96fe5..b309fe77 100644 --- a/doc/backlog/R148-chunk-stream-scale-out.md +++ b/doc/backlog/R148-chunk-stream-scale-out.md @@ -149,5 +149,6 @@ Required gates: full `test-server` task pass locally at default test concurrency. CI run `36449749925` completed its "Upload test logs on failure" step, but the run artifact list contains only `docker-preview-1`; the `runtime-server` artifact - is absent. Preserve runtime logs on the next failure to identify the first + is absent. The test now prints the queried chunk and location on this failure. + Wait for a recurrence and use those diagnostics to identify the first divergent state before changing the assertion or retry policy. diff --git a/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs b/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs index 458452bf..7a9ef553 100644 --- a/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs +++ b/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs @@ -216,7 +216,12 @@ async fn read_data_image(stack: &E2eStack, chunk: &Chunk, location: &Location) - let end = start + u64::from(strip.capacity) * KIB as u64; start <= location.offset && location.offset < end }) - .expect("location strip"); + .unwrap_or_else(|| { + panic!( + "location strip missing while reading image: chunk_id={:?}, offset={}, length={}, chunk={chunk:#?}", + location.chunk_id, location.offset, location.length + ) + }); let strip_start = u64::from(strip.chunk_offset) * KIB as u64; let unit_bytes = u64::from(strip.unit_kb) * KIB as u64; let segment = match strip.strip.as_ref().expect("strip body") { From 99cd7dd6a6390b9eca28eee74783fa6c61656a38 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 07:29:16 +0800 Subject: [PATCH 28/74] Record cleanup and console authority decisions --- doc/backlog/R167-s3-multipart-upload.md | 30 ++++--- doc/backlog/R188-console-group0-authority.md | 47 +++++++---- doc/backlog/R92-chunkdb-in-chunk-gc.md | 17 +--- doc/backlog/R95-chunkdb-chunk-range-delete.md | 82 +++++++++++++++++-- doc/working/plan-console-authority.md | 14 +++- doc/working/plan-s3-multipart.md | 16 ++-- 6 files changed, 140 insertions(+), 66 deletions(-) diff --git a/doc/backlog/R167-s3-multipart-upload.md b/doc/backlog/R167-s3-multipart-upload.md index e5bd52d8..1e6bd873 100644 --- a/doc/backlog/R167-s3-multipart-upload.md +++ b/doc/backlog/R167-s3-multipart-upload.md @@ -24,9 +24,10 @@ The scope boundary is 1. Add create, upload-part, list-parts, complete, abort, and required upload listing operations through S3-owned API and namespace adapters over the protocol-neutral multipart session, part, and completion core from R190. -2. Store immutable part identities and integrity records durably; a retried - part number replaces only that part's selected generation and schedules old - private data for cleanup. +2. Store immutable part identities and integrity records durably under a common + upload key prefix; a retried part number replaces only that part's selected + generation. Keep old generations discoverable for R95's chunk-centered scan + without writing MPU cleanup records. 3. Complete with one fenced metadata transaction that validates ordered part identities, sizes, checksums, and expected upload state before publishing one immutable object generation. Compose the selected parts' chunk-location @@ -35,7 +36,8 @@ The scope boundary is 4. Persist each uploaded part's raw 16-byte MD5. The multipart ETag is the lowercase hexadecimal MD5 of the selected parts' raw MD5 bytes in order, followed by `-`. Keep the basic single-part ETag rule unchanged. - Abort and expiry create bounded, idempotent cleanup records. + Abort and expiry mark the upload terminal or expired in durable metadata; + R95 later qualifies unreachable bytes after its age and reference checks. 5. Preserve the basic admission bounds for parallel part traffic and wire compatibility for all retry and conflict outcomes. @@ -46,6 +48,8 @@ The scope boundary is deletion, and integrity contracts. - Reuses R190's protocol-neutral multipart core; S3 retains its own authorization, namespace, wire errors, ETag response, and object publication. +- R95 owns eventual chunk-centered reclamation, including unused MPU part + locations; R167 does not add a per-upload cleanup queue. - R170 owns any accelerated multipart transfer and additionally depends on this requirement before enabling that operation. @@ -65,12 +69,13 @@ The scope boundary is Integration test. - Given response loss during part upload, complete, and abort, when identities retry after restart, assert the durable outcome is returned without duplicate - generations or cleanup. Invariant: every multipart transition is idempotent. + generations. Invariant: every multipart transition is idempotent. E2E test. -- Given abort/expiry with readers or cleanup failures, when reconciliation - runs, assert no active upload is reclaimed and unreachable part data is - eventually queued within bounds. Invariant: cleanup follows durable upload - state. Integration test. +- Given abort, expiry and part replacement, when upload metadata is scanned, + assert current and immutable part generations remain discoverable under the + upload prefix for R95 and no MPU cleanup record is created. Invariant: R167 + preserves reference evidence without deciding physical reclamation. + Integration test. Required gates: @@ -87,10 +92,3 @@ Required gates: recovery is bound to its catalog identity and store. Extract the remaining protocol-neutral transition decisions while keeping keys, authorization and responses in the protocol adapters. -- S3 preserves immutable part generations so completion can publish a selected - generation across concurrent part-number replacement. An identical durable - part record retry keeps its revision, but an HTTP retry may stream the same - bytes to a new location. Losing replacement candidates, replayed stream - locations and old generations can remain unreachable. Abort, expiry and - replacement cleanup need durable bounded records and reader-pin protection - before the HTTP path is enabled. diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md index 58075a71..01a3b79d 100644 --- a/doc/backlog/R188-console-group0-authority.md +++ b/doc/backlog/R188-console-group0-authority.md @@ -28,9 +28,11 @@ not block R187 completion. ## Solution 1. Keep Group 0 as the durable authority for CROWDB hardware hierarchy, - ownership and binding maps, KV store/group/replica metadata, and service - registration. Do not add Docker or bare-metal process deployment records to - Group 0. Docker process state comes from `crowdb-monitor`; bare-metal launch + including rack names, node management hosts, nonsecret SSH connection + settings and credential reference IDs, ownership and binding maps, KV + store/group/replica metadata, and service registration. Do not store SSH + private keys or passwords in Group 0. Do not add process deployment records + to Group 0. Docker process state comes from `crowdb-monitor`; bare-metal launch policy remains local. Deployment mode changes which lifecycle and hardware controls are allowed, not the meaning of Group 0 records. 2. Replace the mixed `ConsoleConfig` persistence in @@ -39,9 +41,10 @@ not block R187 completion. registry. Finish wiring the existing `LaunchRegistry` parser to actual bare-metal deploy/restart operations; remove the unreleased mixed parser/writer, topology fields, restore path, fixtures, and fallback rather - than adding a compatibility reader. Retain SSH credential references, - binary/config paths, workspace, and auto-start policy locally; never persist - inline secrets or runtime PID as topology. Docker Web rejects a launch + than adding a compatibility reader. Resolve Group 0 credential reference IDs + through each console's local secret store. Retain binary/config paths, + workspace, and auto-start policy locally; never persist inline secrets or + runtime PID as topology. Docker Web rejects a launch registry and keeps its monitor-owned process path. 3. Unify CLI and bare-metal Web hardware mutations through Group 0-backed operations in `crowdb-console-shared::ops::hardware`. Confirm writes before @@ -79,8 +82,12 @@ not block R187 completion. private data-volume location. Provide an exact-build source-line symbolization workflow for child and monitor crashes. Dumps can contain secrets and user data; diagnostics must not expose them in ordinary logs. - Host acceptance and symbol-distribution choices remain open in the execution - plan; no image-size increase or host configuration change is assumed. + Ship exact-build debug symbols as a separate GitHub Release asset generated + from the same staged runtime as the image, indexed by version and source + revision. The release preparation script in `tools/` runs manually, shows a + dry-run plan, updates versions, creates the tag and GitHub Release, then + dispatches the existing verified DockerHub publication workflow. Host + acceptance remains open; no host configuration change is assumed. ## Dependencies @@ -109,6 +116,12 @@ not block R187 completion. nodes, disk groups, or disks and a write conflicts or loses its response, assert both read one confirmed result and neither commits a local-first topology change. Invariant: hardware authority. Integration test. +- Given two consoles with different local launch registries, when both read the + same rack and node, assert Group 0 supplies identical names, management hosts, + SSH connection settings and credential reference IDs while each console + resolves secret material only from its local secret store. Invariant: shared + hardware display and connection identity never depend on local topology. + Integration test. - Given CLI, Docker Web, and bare-metal Web with the same Group 0, when each performs authenticated logical store/group/replica operations, assert one shared result, correct fan-out/rollback, and no local logical copy. @@ -144,6 +157,13 @@ not block R187 completion. locates the dump or explicitly reports unsupported collection, without claiming an absent data-volume core. Invariant: truthful collector boundary. Integration test. +- Given a clean main checkout and a version bump, when the release tool runs in + dry-run mode, assert it shows every version change and no file or remote is + modified. When run for a release, assert the tag and GitHub Release identify + the same verified revision, the symbol asset contains source-line information, + GNU debuglink CRCs and SHA-256 hashes match the image's stripped binaries. + Invariant: released symbols come from the image build and + remain available after a build host changes. E2E test. Required gates: @@ -159,14 +179,5 @@ Required gates: - This host routes `core_pattern` to Apport, so a container-local directory and core ulimit cannot guarantee a dump in `/opt/crowdb/data`. End-to-end acceptance needs a disposable host with file-based collection or a verified - host-collector export workflow. Exact-build source-line symbols also need a - distribution choice: compressed line tables in the image with a measured - size increase, or separate exact-build debug symbols. The all-dependency - symbol experiment enlarged the monitor substantially; a complete-image - measurement remains pending. Bounded volume retention and source-line + host-collector export workflow. Bounded volume retention and source-line symbolization remain unverified. -- Group 0 rack and node values hold IDs and status but not the console's rack - name, node host or SSH settings. The authority cutover must define where - shared display names live and keep machine-local launch inputs in the launch - registry. Until that split is implemented, two consoles cannot reconstruct - identical physical views from Group 0 alone. diff --git a/doc/backlog/R92-chunkdb-in-chunk-gc.md b/doc/backlog/R92-chunkdb-in-chunk-gc.md index 933ff05b..08b377e4 100644 --- a/doc/backlog/R92-chunkdb-in-chunk-gc.md +++ b/doc/backlog/R92-chunkdb-in-chunk-gc.md @@ -10,19 +10,8 @@ to avoid global merge overhead. **Solution**: Implement in-chunk GC operations (ReclaimStrip, CollapseStrip, MergeStrips) for shared chunks. Add logical-to-physical offset mapping -to support GC while keeping chunk IDs stable. Add a ChunkDB orphan scanner for -chunks and shared ranges allocated by access uploads that crash or fail before -their complete file/object descriptor is published. R190 intentionally does not -write per-chunk catalog intents or upload-owner records on its write hot path. -The scanner must compare candidates with authoritative published S3 and Iceberg -references and reader protection before reclaiming, and must not infer orphan -status merely from age or a missing intermediate upload record. Account for -in-flight writers and delayed publication so physical ranges are never reused -while a writer or reader can still own them. Report candidate and reclaimed -bytes separately. Include Iceberg MPU Complete's frozen selection payload as -an authoritative reference while completion is in progress: it stores the -selected parts' exact chunk locations, which remain live even if a concurrent -UploadPart replaces the same part number before publication. After publication, -the immutable file descriptor is the authoritative reference. +to support GC while keeping chunk IDs stable. R95 owns the chunk-centered +orphan scan and qualified range deletion; this requirement provides the +in-chunk reclamation operations after R95 proves a range unreachable. **Scope**: Placeholder - detailed design to be refined before implementation. diff --git a/doc/backlog/R95-chunkdb-chunk-range-delete.md b/doc/backlog/R95-chunkdb-chunk-range-delete.md index 6d980c7a..c03a7b76 100644 --- a/doc/backlog/R95-chunkdb-chunk-range-delete.md +++ b/doc/backlog/R95-chunkdb-chunk-range-delete.md @@ -1,14 +1,80 @@ -### R95: chunkdb — Chunk Range Delete +### R95: chunkdb — Qualified chunk range deletion and orphan scan -**Problem**: Shared chunks need partial deletion capability for individual object deletion. Without range delete, entire shared chunks cannot be reclaimed efficiently. +## Problem -**Solution**: Define `DeleteChunkRange(chunk_id, offset, size)` in the chunkdb -protocol, client, and server dispatch now. The initial server implementation -returns an explicit not-implemented result without mutation. The full R95 -implementation adds range validation, used-bitmap management, idempotency, and -in-chunk GC integration before any caller may treat success as reclamation. +Shared chunks contain ranges owned by different objects or multipart parts. +Deleting a whole chunk for one unreachable range can erase live neighbors. +Uploads may also write a chunk and fail before their part or final object +reference is recorded. A cleanup queue populated by each MPU mutation would +add metadata writes to the upload path and still miss those pre-record crashes. -**Scope**: Placeholder - detailed design to be refined before implementation. +## Solution + +1. Complete `DeleteChunkRange(chunk_id, offset, size)` in the chunkdb protocol, + client and server. The existing stub remains explicitly not implemented + until range validation, used-bitmap updates, idempotency and in-chunk GC can + prove that the exact physical range is safe to retire. +2. Use a bounded, chunk-centered scanner to find old chunks and ranges whose + bytes have no live owner. Do not create per-MPU cleanup intents, candidate + records or a durable cleanup queue. Start with a configurable age threshold + of one day; a candidate must be older than the threshold after its last + write. Age alone never authorizes deletion. +3. Compare each candidate against published S3 objects, Iceberg file + descriptors, active multipart sessions and parts, frozen completion + selections, in-flight writers and reader protection. A lost part-publication + response or missing intermediate MPU record is not proof that a chunk is + unused. Refuse deletion when reference or reader state cannot be confirmed. +4. Keep S3 MPU session, current-part and immutable part-generation keys under + a common upload prefix in Chunk-KV so the scanner can enumerate one upload's + references with a bounded prefix scan. R167 owns that key layout and the + metadata-only Abort/expiry transition; old part generations remain available + until the scanner proves their bytes are unreachable. An aborted or expired + upload becomes a candidate only after the age gate and reference checks. +5. Recheck chunk identity, layout generation and exact range against current + authority immediately before reclaim. Treat lost delete replies + idempotently and keep an in-memory scan cursor and bounded work budget. + Restart may rescan old chunks. Report examined, + eligible, deferred and reclaimed bytes without per-upload metric labels. + +## Dependencies + +- R92 supplies in-chunk strip reclamation after R95 qualifies dead ranges. +- R167 supplies grouped MPU keys and durable session, part and completion + references. The scanner also recognizes Iceberg's frozen MPU selection. +- Reader protection and published generation references must be queryable + before physical deletion is enabled; R168 may use the qualified range-delete + interface for ordinary S3 object deletion. + +## Acceptance + +- Given a shared chunk with live and unreachable ranges, when the scanner + evaluates a candidate older than one day, assert only the exact unreachable + range is passed to `DeleteChunkRange`. Invariant: a live neighbor is never + reclaimed. E2E test. +- Given an MPU part that was replaced, aborted or expired, when its grouped KV + prefix is scanned, assert old locations remain protected by any active or + frozen selection and become eligible only after the age and reference + checks. Invariant: no extra MPU cleanup record is required. Integration test. +- Given a write that crashed before part metadata publication, when the chunk + scan runs, assert it defers the chunk until the age gate and in-flight writer + proof settle, then discovers the orphan without a part record. Invariant: + missing upload metadata alone cannot reclaim data. Integration test. +- Given a current reader, changed layout generation or uncertain authority, + when range deletion is attempted, assert no bytes are reused. After the + reader releases and identity is confirmed, repeat the same delete through a + lost reply and assert one idempotent result. Invariant: reclamation is fenced + by current chunk and reader authority. Integration test. +- Given a large backlog and foreground load, when scanning runs, assert it + respects item, byte and time budgets and reports deferred/reclaimed progress. + Invariant: cleanup cannot monopolize the data path. Integration test. + +Required gates: + +- `pixi run -- cargo test -p crowdb-chunkdb --all-targets` +- `pixi run -- cargo test -p crowdb-chunk-client --all-targets` +- `pixi run -- cargo test -p crowdb-access-s3 --all-targets` +- `pixi run -- cargo fmt --all -- --check` +- `pixi run rs-lint` diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 9aefa5a1..7016c106 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -99,8 +99,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. shared Group 0 hardware operations; preserve conflicts and uncertain writes without local-first commits. Docker keeps its hardware restrictions. Hardware client cascades now stop at a failed child deletion instead of deleting its - parent while a descendant may survive; exact-value confirmation and the - CLI/Web cutover remain. + parent while a descendant may survive. Extend Group 0 hardware values with + rack names, node management hosts, nonsecret SSH connection settings and + credential reference IDs; resolve secret material locally. Exact-value + confirmation and the CLI/Web cutover remain. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. @@ -152,6 +154,14 @@ is paused; it does not block the single-node image requirement. Apport, systemd-coredump and Docker Desktop lookup paths without promising a volume dump. This host reports an Apport pipe pattern and core ulimit 0. Volume retention, exact-build symbols and disposable-host acceptance remain. +- [ ] **Manual release and symbols**: add a `tools/` release script with a + read-only dry run, consistent version updates, tag and GitHub Release + creation, and dispatch of the existing publication workflow. Extract debug + symbols from the same staged ELF files as the image, keep them out of the + image, and upload the version/revision-named archive to that release. + Verify symbol identity and source-line lookup. Files: + `tools/release.py`, `container/single-node-container/{build.sh,collect-libs.sh}`, + `.github/workflows/release-container.yml`. ## Documentation and completion - [ ] **Bare-metal documentation**: migrate verified KV, chunk and access diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 11506502..378cfcb6 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -31,15 +31,15 @@ Iceberg multipart path. ## S3 adapter and HTTP - [ ] **Durable S3 records**: add upload and part keys/records with raw 16-byte - MD5, selected revision and cleanup state. Use bucket identity and object key - as namespace scope; preserve immutable part data after replacement. + MD5 and selected revision under one upload prefix. Use bucket identity and + object key as namespace scope; preserve immutable part data after replacement. Versioned session/part records and ordered, binary-safe keys are in place; CAS-backed begin, phase transition and part replacement now use exact-value confirmation after lost replies. An identical part record retry returns the existing revision; a new location remains a replacement. Completion snapshots and a predecessor-fenced metadata-only object publication path are - in place. HTTP wiring and cleanup - state remain. + in place. The current session, part and generation key families still need + grouping under one upload prefix for R95. HTTP wiring remains. - [ ] **S3 routes and wire**: classify create/upload/list/complete/abort/list uploads, parse bounded completion XML, emit compatible responses and errors. Preserve SigV4 authentication and existing basic routes. The repository now @@ -63,15 +63,15 @@ Iceberg multipart path. object-key CAS, and confirms exact publication after response loss. An immutable generation records preserve selected bytes across a concurrent part-number replacement. The HTTP path and end-to-end publication test remain. -- [ ] **Abort and expiry**: make terminal states idempotent, queue unreachable - private part data for bounded cleanup, and protect active/read-pinned data. +- [ ] **Abort and expiry**: make terminal states idempotent and preserve the + part generations that R95's chunk-centered scanner needs for reference checks. The S3 adapter now has an idempotent, response-loss-safe logical abort; the - durable cleanup queue, expiry scan and read-pin protection remain. + expiry scan and common-prefix key layout remain. R95 owns physical cleanup. ## Acceptance and cleanup - [ ] **Focused and E2E tests**: known MD5 vectors, out-of-order/replaced parts, - invalid completion, response loss and restart, abort/expiry cleanup, and + invalid completion, response loss and restart, abort/expiry metadata, and ordinary single-part compatibility. - [ ] **Gates and docs**: run both access crate suites, access-server E2E, Rust fmt and clippy separately; update S3 design and remove R167 plus this From e398b12cb7e16a34ee7eb2f4e26c6873de62e1cf Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 07:29:23 +0800 Subject: [PATCH 29/74] Prepare verified container releases with symbols --- .github/workflows/release-container.yml | 32 +++- app/crowdb-diskio/CMakeLists.txt | 2 +- container/single-node-container/README.md | 33 ++-- container/single-node-container/build.sh | 18 ++- .../single-node-container/collect-libs.sh | 13 ++ .../tests/release-policy.sh | 13 ++ tools/ci-checks/check-container-symbols.py | 66 ++++++++ tools/release.py | 144 ++++++++++++++++++ 8 files changed, 307 insertions(+), 14 deletions(-) create mode 100644 tools/ci-checks/check-container-symbols.py create mode 100644 tools/release.py diff --git a/.github/workflows/release-container.yml b/.github/workflows/release-container.yml index cc4efcbd..257ee1cb 100644 --- a/.github/workflows/release-container.yml +++ b/.github/workflows/release-container.yml @@ -19,6 +19,7 @@ jobs: CARGO_INCREMENTAL: "0" CARGO_PROFILE_DEV_DEBUG: line-tables-only CARGO_PROFILE_TEST_DEBUG: line-tables-only + CROWDB_PACKAGE_SYMBOLS: "1" permissions: contents: read outputs: @@ -47,7 +48,7 @@ jobs: [[ "$RELEASE_TAG" == "v$(cat VERSION)" ]] revision=$(git rev-parse --verify "refs/tags/$RELEASE_TAG^{commit}") [[ "$revision" == "$(git rev-parse HEAD)" ]] - [[ "$(gh release view "$RELEASE_TAG" --json isDraft --jq .isDraft)" == false ]] + [[ "$(gh release view "$RELEASE_TAG" --json isDraft --jq .isDraft)" == true ]] printf "version=%s\nrevision=%s\n" "${RELEASE_TAG#v}" "$revision" >> "$GITHUB_OUTPUT" ' - name: Free disk space @@ -84,14 +85,24 @@ jobs: run: pixi run clean-env && pixi run test-console-ui - name: Check Rust formatting and lint run: pixi run rs-fmt-check && pixi run rs-lint + - name: Verify exact-build source-line symbols + run: pixi run -- python tools/ci-checks/check-container-symbols.py - name: Archive verified runtime files - run: pixi run tar -C target/container-runtime -czf target/container-runtime.tar.gz . + run: | + pixi run tar -C target/container-runtime -czf target/container-runtime.tar.gz . + pixi run tar -C target/container-symbols -I zstd -cf "target/crowdb-symbols-${{ inputs.tag }}-git-${{ steps.source.outputs.revision }}-linux-amd64.tar.zst" . - uses: actions/upload-artifact@v4 with: name: verified-container-runtime path: target/container-runtime.tar.gz compression-level: 0 retention-days: 7 + - uses: actions/upload-artifact@v4 + with: + name: verified-container-symbols + path: target/crowdb-symbols-*.tar.zst + compression-level: 0 + retention-days: 7 - name: Upload preview failure logs if: failure() uses: actions/upload-artifact@v4 @@ -106,7 +117,7 @@ jobs: runs-on: ubuntu-24.04 environment: DockerHub permissions: - contents: read + contents: write id-token: write steps: - uses: actions/checkout@v4 @@ -138,6 +149,10 @@ jobs: with: name: verified-container-runtime path: target + - uses: actions/download-artifact@v4 + with: + name: verified-container-symbols + path: target - name: Extract verified runtime files run: | pixi run bash -euc 'mkdir -p target/container-runtime && tar -C target/container-runtime -xzf target/container-runtime.tar.gz' @@ -167,3 +182,14 @@ jobs: env: DIGEST: ${{ steps.build.outputs.digest }} run: pixi run cosign sign --yes "docker.io/crowdb/crowdb-iceberg@$DIGEST" + - name: Attach exact-build symbols to GitHub Release + env: + GH_TOKEN: ${{ github.token }} + RELEASE_TAG: ${{ inputs.tag }} + REVISION: ${{ needs.verify.outputs.revision }} + run: pixi run gh release upload "$RELEASE_TAG" "target/crowdb-symbols-$RELEASE_TAG-git-$REVISION-linux-amd64.tar.zst" --repo "$GITHUB_REPOSITORY" + - name: Publish verified GitHub Release + env: + GH_TOKEN: ${{ github.token }} + RELEASE_TAG: ${{ inputs.tag }} + run: pixi run gh release edit "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --draft=false diff --git a/app/crowdb-diskio/CMakeLists.txt b/app/crowdb-diskio/CMakeLists.txt index b9898ef0..8b067630 100644 --- a/app/crowdb-diskio/CMakeLists.txt +++ b/app/crowdb-diskio/CMakeLists.txt @@ -25,7 +25,7 @@ add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/../../lib/crowdb-rpc crowdb-rpc-bui # when built with the `ffi` feature. set(CROWDB_KV_CLIENT_DIR ${CMAKE_CURRENT_SOURCE_DIR}/../../lib/crowdb-kv-client) set(CROWDB_ROOT_DIR ${CMAKE_CURRENT_SOURCE_DIR}/../..) -if(CMAKE_BUILD_TYPE STREQUAL "Release") +if(CMAKE_BUILD_TYPE STREQUAL "Release" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") set(CARGO_PROFILE release) else() set(CARGO_PROFILE debug) diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index 3b5b2029..b7767c16 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -25,10 +25,25 @@ pixi run test-single-node-container `CROWDB_CONTAINER_IMAGE` to build and test a separate candidate tag. `pixi run stage-single-node-container` produces the runtime directory without -building a Docker image. The release workflow archives the verified directory -and packages those same files in its publish job, without recompiling them. -Docker Hub publication is manual; actual publication verification is deferred -until administrator preparation is complete. +building a Docker image. To prepare a release from a clean, current `main` +checkout, preview the patch bump and then run it explicitly: + +```sh +pixi run -- python tools/release.py --dry-run +pixi run -- python tools/release.py --execute +``` + +`--bump minor` and `--bump major` select larger version changes. The script +updates every version manifest, commits and tags the release, atomically pushes +`main` and the tag, creates a draft GitHub Release, then dispatches the existing +verified DockerHub workflow. Execution requires authenticated `gh` and GitHub +permission to push `main`; the dry run changes no files or remote state. The +workflow archives the verified runtime and its separate exact-build symbol +package from one build, then packages those same runtime files in its publish +job without recompiling them. The symbol archive is attached to the GitHub +Release as `crowdb-symbols--git--linux-amd64.tar.zst`. +The workflow publishes the GitHub Release after the Docker image, signature +and symbol upload succeed. ## Crash collection boundary @@ -36,9 +51,9 @@ The image does not configure the host's Linux core collector. Inspect `/proc/sys/kernel/core_pattern` on the Docker host before expecting a dump in the mounted data volume. A leading `|` sends a crash to a host-side collector; relative file patterns write in the crashing process's working directory. -The container does not currently set a private core working directory or a -core size limit, and it does not provide dump retention or exact-build debug -symbols. Do not assume `/opt/crowdb/data` contains a core after a crash. +The container does not currently set a private core working directory, a +core size limit, or dump retention. Release debug symbols are provided +separately. Do not assume `/opt/crowdb/data` contains a core after a crash. - On a systemd-coredump host, use `coredumpctl list` and `coredumpctl dump` on the host to locate and export a captured dump. @@ -49,8 +64,8 @@ symbols. Do not assume `/opt/crowdb/data` contains a core after a crash. Core dumps can contain credentials and user data. Store exports privately, apply host retention policy, and match the exact image revision and binary -build when symbolizing. The required bounded volume collection and symbol -distribution remain tracked by R188. +build when symbolizing. The required bounded volume collection and source-line +symbolization check remain tracked by R188. Collector behavior follows the [Linux core pattern documentation](https://docs.kernel.org/admin-guide/sysctl/kernel.html), [systemd-coredump manual](https://www.freedesktop.org/software/systemd/man/250/systemd-coredump.socket.html), diff --git a/container/single-node-container/build.sh b/container/single-node-container/build.sh index 663ad44c..78d0bd5a 100644 --- a/container/single-node-container/build.sh +++ b/container/single-node-container/build.sh @@ -10,13 +10,22 @@ cd "$(git rev-parse --show-toplevel)" for tool in patchelf strip ldd; do command -v "$tool" >/dev/null || { echo "Missing packaging tool: $tool" >&2; exit 1; } done +if [[ "${CROWDB_PACKAGE_SYMBOLS:-0}" == 1 ]]; then + for tool in objcopy readelf; do + command -v "$tool" >/dev/null || { echo "Missing packaging tool: $tool" >&2; exit 1; } + done + export CARGO_PROFILE_RELEASE_DEBUG=line-tables-only + cmake_build_type=RelWithDebInfo +else + cmake_build_type=Release +fi if [[ "$mode" == image ]]; then docker info >/dev/null fi # Build on the host, reusing the existing Cargo, CMake and npm artifacts. cargo build --locked --release -p crowdb-kv-client --features ffi -cmake -S app/crowdb-diskio -B app/crowdb-diskio/build -DCMAKE_BUILD_TYPE=Release +cmake -S app/crowdb-diskio -B app/crowdb-diskio/build -DCMAKE_BUILD_TYPE="$cmake_build_type" cmake --build app/crowdb-diskio/build -j 4 --target crowdb-diskio cargo build --locked --release \ -p crowdb-monitor -p crowdb-kv-server -p crowdb-diskdb \ @@ -33,6 +42,13 @@ cp -a container/single-node-container/templates "$staging/templates" cp container/single-node-container/{Dockerfile,profile.toml,entrypoint.sh} "$staging/" git rev-parse HEAD > "$staging/SOURCE_REVISION" cp VERSION "$staging/VERSION" +if [[ "${CROWDB_PACKAGE_SYMBOLS:-0}" == 1 ]]; then + cp VERSION "$staging/symbols/VERSION" + git rev-parse HEAD > "$staging/symbols/SOURCE_REVISION" + (cd "$staging" && sha256sum bin/* lib/libcrowdb*.so) > "$staging/symbols/RUNTIME_SHA256SUMS" + rm -rf target/container-symbols + mv "$staging/symbols" target/container-symbols +fi rm -rf target/container-runtime mv "$staging" target/container-runtime trap - EXIT diff --git a/container/single-node-container/collect-libs.sh b/container/single-node-container/collect-libs.sh index 4276ad68..72b950d9 100644 --- a/container/single-node-container/collect-libs.sh +++ b/container/single-node-container/collect-libs.sh @@ -56,6 +56,19 @@ for library in "$output"/lib/*; do done rm "$output/dependencies.txt" +if [[ "${CROWDB_PACKAGE_SYMBOLS:-0}" == 1 ]]; then + for artifact in "$output"/bin/* "$output"/lib/libcrowdb*.so; do + [[ -f "$artifact" ]] || continue + if ! readelf -W -S "$artifact" | grep -E '[[:space:]]\.debug_line[[:space:]]' >/dev/null; then + echo "Missing source-line symbols: $artifact" >&2 + exit 1 + fi + symbol="$output/symbols/$(basename "$(dirname "$artifact")")/$(basename "$artifact").debug" + mkdir -p "$(dirname "$symbol")" + objcopy --only-keep-debug "$artifact" "$symbol" + objcopy --add-gnu-debuglink="$symbol" "$artifact" + done +fi for artifact in "$output"/bin/* "$output"/lib/*; do strip --strip-debug "$artifact" done diff --git a/container/single-node-container/tests/release-policy.sh b/container/single-node-container/tests/release-policy.sh index cc75258f..9cd98909 100644 --- a/container/single-node-container/tests/release-policy.sh +++ b/container/single-node-container/tests/release-policy.sh @@ -15,6 +15,7 @@ for required in \ 'git rev-parse --verify "refs/tags/$RELEASE_TAG^{commit}"' \ '[[ "$revision" == "$(git rev-parse HEAD)" ]]' \ 'gh release view "$RELEASE_TAG"' \ + '== true ]]' \ '[[ "$status" == 404 ]]' \ 'needs: verify' \ 'docker.io/crowdb/crowdb-iceberg:${{ inputs.tag }}' \ @@ -33,6 +34,12 @@ verify_job=$(sed -n '/^ verify:/,/^ publish:/p' "$release") publish_job=$(sed -n '/^ publish:/,$p' "$release") [[ "$verify_job" == *'name: verified-container-runtime'* ]] [[ "$publish_job" == *'name: verified-container-runtime'* ]] +[[ "$verify_job" == *'name: verified-container-symbols'* ]] +[[ "$publish_job" == *'name: verified-container-symbols'* ]] +[[ "$verify_job" == *'CROWDB_PACKAGE_SYMBOLS: "1"'* ]] +[[ "$verify_job" == *'pixi run -- python tools/ci-checks/check-container-symbols.py'* ]] +[[ "$publish_job" == *'gh release upload "$RELEASE_TAG"'* ]] +[[ "$publish_job" == *'gh release edit "$RELEASE_TAG"'* ]] [[ "$publish_job" == *'context: target/container-runtime'* ]] for gate in 'pixi run test-single-node-container' 'test-boto3-e2e' 'test-pyiceberg-e2e' \ 'pixi run test-console' 'pixi run test-console-ui' 'pixi run rs-fmt-check && pixi run rs-lint'; do @@ -47,3 +54,9 @@ ci_job=$(sed -n '/^ DockerPreview:/,$p' "$ci") [[ "$ci_job" == *'contents: read'* && "$ci_job" == *'pixi run test-single-node-container'* ]] [[ "$ci_job" == *'Upload preview failure logs'* && "$ci_job" == *'CROWDB_PREVIEW_TEST_ARTIFACTS'* ]] ! grep -Eq 'secrets\.|docker/login-action|docker/build-push-action' <<<"$ci_job" + +release_tool=tools/release.py +for required in '--dry-run' '--execute' 'git", "push", "--atomic"' \ + '"release", "create"' '"workflow", "run"'; do + grep -Fq -- "$required" "$release_tool" +done diff --git a/tools/ci-checks/check-container-symbols.py b/tools/ci-checks/check-container-symbols.py new file mode 100644 index 00000000..c9e28ecf --- /dev/null +++ b/tools/ci-checks/check-container-symbols.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +# Copyright 2026-present Gian +# Licensed under the Apache License, Version 2.0. +"""Verify the symbol archive matches the staged release runtime exactly.""" + +import re +import struct +import subprocess +import tempfile +import zlib +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +RUNTIME = ROOT / "target/container-runtime" +SYMBOLS = ROOT / "target/container-symbols" + + +def sections(path: Path) -> str: + return subprocess.run( + ["readelf", "-W", "-S", str(path)], check=True, text=True, + capture_output=True, + ).stdout + + +def check_debuglink(binary: Path, symbol: Path) -> None: + with tempfile.TemporaryDirectory() as temp_dir: + section = Path(temp_dir) / "debuglink" + subprocess.run( + ["objcopy", f"--dump-section=.gnu_debuglink={section}", str(binary)], + check=True, + ) + data = section.read_bytes() + name, _, _ = data.partition(b"\0") + offset = (len(name) + 4) & ~3 + if name.decode() != symbol.name or len(data) < offset + 4: + raise ValueError(f"Wrong debuglink name or size: {binary}") + expected_crc = struct.unpack_from(" None: + for metadata in ("VERSION", "SOURCE_REVISION"): + if (RUNTIME / metadata).read_bytes() != (SYMBOLS / metadata).read_bytes(): + raise ValueError(f"Runtime and symbol {metadata} differ") + subprocess.run( + ["sha256sum", "--check", str(SYMBOLS / "RUNTIME_SHA256SUMS")], + cwd=RUNTIME, check=True, + ) + binaries = sorted((RUNTIME / "bin").iterdir()) + libraries = sorted((RUNTIME / "lib").glob("libcrowdb*.so")) + for binary in binaries + libraries: + symbol = SYMBOLS / binary.parent.name / f"{binary.name}.debug" + if not symbol.is_file(): + raise ValueError(f"Missing debug symbols: {symbol}") + if not re.search(r"\s\.debug_line\s", sections(symbol)): + raise ValueError(f"Missing source lines: {symbol}") + if re.search(r"\s\.debug_line\s", sections(binary)): + raise ValueError(f"Runtime still has source lines: {binary}") + check_debuglink(binary, symbol) + print(f"Verified symbols for {binary.relative_to(RUNTIME)}") + + +if __name__ == "__main__": + main() diff --git a/tools/release.py b/tools/release.py new file mode 100644 index 00000000..3a2dfe94 --- /dev/null +++ b/tools/release.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +# Copyright 2026-present Gian +# Licensed under the Apache License, Version 2.0. +"""Prepare a versioned release and dispatch the verified container workflow. + +Run through Pixi: pixi run -- python tools/release.py --dry-run + pixi run -- python tools/release.py --execute +""" + +import argparse +import difflib +import re +import subprocess +import sys +import tomllib +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +VERSION_RE = re.compile(r"^(\d+)\.(\d+)\.(\d+)(-dev)?$") +REPO = "buzzcrow/crowdb" + + +def command(*args: str, capture: bool = False) -> str: + result = subprocess.run( + args, cwd=ROOT, check=True, text=True, + stdout=subprocess.PIPE if capture else None, + timeout=60, + ) + return result.stdout.strip() if capture else "" + + +def next_version(current: str, bump: str) -> str: + match = VERSION_RE.fullmatch(current) + if match is None: + raise ValueError(f"Unsupported VERSION: {current}") + major, minor, patch = (int(part) for part in match.group(1, 2, 3)) + if bump == "major": + return f"{major + 1}.0.0" + if bump == "minor": + return f"{major}.{minor + 1}.0" + if match.group(4) is None: + patch += 1 + return f"{major}.{minor}.{patch}" + + +def changes(current: str, target: str) -> dict[Path, str]: + workspace = tomllib.loads((ROOT / "Cargo.toml").read_text(encoding="utf-8")) + workspace_count = len(workspace["workspace"]["members"]) + files = [ + "Cargo.toml", "Cargo.lock", "pixi.toml", + "app/crowdb-access-server/tests/common/iceberg_rust/Cargo.toml", + "app/crowdb-access-server/tests/common/iceberg_rust/Cargo.lock", + "app/crowdb-web/ui/package.json", "app/crowdb-web/ui/package-lock.json", + ] + updates = {ROOT / "VERSION": target + "\n"} + for name in files: + path = ROOT / name + original = path.read_text(encoding="utf-8") + old = f'version = "{current}"' if path.suffix in (".toml", ".lock") else f'"version": "{current}"' + new = old.replace(current, target) + count = original.count(old) + expected = { + "Cargo.lock": workspace_count, + "app/crowdb-web/ui/package-lock.json": 2, + }.get(name, 1) + if count != expected: + raise ValueError(f"Expected {expected} version entries in {name}, found {count}") + updates[path] = original.replace(old, new) + return updates + + +def preflight(tag: str) -> None: + remote_url = command("git", "remote", "get-url", "origin", capture=True) + if remote_url not in ( + "git@github.com:buzzcrow/crowdb.git", + "https://github.com/buzzcrow/crowdb.git", + ): + raise ValueError(f"Unexpected origin: {remote_url}") + if command("git", "branch", "--show-current", capture=True) != "main": + raise ValueError("Run --execute from the clean main branch") + if command("git", "status", "--porcelain", capture=True): + raise ValueError("Working tree must be clean before release") + head = command("git", "rev-parse", "HEAD", capture=True) + remote_lines = command("git", "ls-remote", "origin", "refs/heads/main", capture=True).split() + if not remote_lines: + raise ValueError("origin/main is unavailable") + remote = remote_lines[0] + if head != remote: + raise ValueError("Local main does not match origin/main") + if command("git", "tag", "--list", tag, capture=True): + raise ValueError(f"Local tag already exists: {tag}") + if command("git", "ls-remote", "--tags", "origin", f"refs/tags/{tag}", capture=True): + raise ValueError(f"Remote tag already exists: {tag}") + command("gh", "auth", "status") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + mode = parser.add_mutually_exclusive_group(required=True) + mode.add_argument("--dry-run", action="store_true", help="Print the plan without writing or contacting GitHub") + mode.add_argument("--execute", action="store_true", help="Update, tag, push, release and dispatch") + parser.add_argument("--bump", choices=("patch", "minor", "major"), default="patch") + args = parser.parse_args() + + current = (ROOT / "VERSION").read_text(encoding="utf-8").strip() + target = next_version(current, args.bump) + tag = f"v{target}" + updates = changes(current, target) + print(f"Release {current} -> {target} ({tag})", flush=True) + for path in updates: + print(f" update {path.relative_to(ROOT)}", flush=True) + print(" check versions and diff; commit; tag; atomically push main + tag", flush=True) + print(" create draft GitHub Release; dispatch release-container.yml", flush=True) + if args.dry_run: + for path, updated in updates.items(): + original = path.read_text(encoding="utf-8") + sys.stdout.writelines(difflib.unified_diff( + original.splitlines(keepends=True), updated.splitlines(keepends=True), + fromfile=str(path.relative_to(ROOT)), + tofile=str(path.relative_to(ROOT)), + )) + return + + preflight(tag) + for path, content in updates.items(): + path.write_text(content, encoding="utf-8") + command("pixi", "run", "--", "python", "tools/ci-checks/check-version.py") + command("git", "diff", "--check") + command("git", "add", *(str(path.relative_to(ROOT)) for path in updates)) + command("git", "commit", "-m", f"Release {target}") + command("git", "tag", "-a", tag, "-m", tag) + command("git", "push", "--atomic", "origin", "HEAD:refs/heads/main", f"refs/tags/{tag}") + command("gh", "release", "create", tag, "--repo", REPO, "--verify-tag", "--generate-notes", "--draft") + command("gh", "workflow", "run", "release-container.yml", "--repo", REPO, + "--ref", tag, "-f", f"tag={tag}") + print(f"Started verified publication for {tag}") + + +if __name__ == "__main__": + try: + main() + except (OSError, ValueError, subprocess.CalledProcessError, subprocess.TimeoutExpired) as error: + print(f"Release stopped: {error}", file=sys.stderr) + raise SystemExit(1) from error From 1bb40b2ca29056b4c8686a7fa1f87abb5cbad978 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 08:38:31 +0800 Subject: [PATCH 30/74] Make release symbols optional --- .github/workflows/release-container.yml | 37 ++++++++++++++----- container/single-node-container/README.md | 22 ++++++----- .../tests/release-policy.sh | 9 ++++- doc/backlog/R188-console-group0-authority.md | 16 ++++---- doc/working/plan-console-authority.md | 12 +++--- tools/release.py | 5 ++- 6 files changed, 66 insertions(+), 35 deletions(-) diff --git a/.github/workflows/release-container.yml b/.github/workflows/release-container.yml index 257ee1cb..dc3e9e05 100644 --- a/.github/workflows/release-container.yml +++ b/.github/workflows/release-container.yml @@ -7,6 +7,11 @@ on: description: Existing Git release tag to publish required: true type: string + include_symbols: + description: Build and attach the large exact-build symbol archive + required: false + default: false + type: boolean concurrency: group: crowdb-iceberg-single-node-preview-release @@ -19,7 +24,7 @@ jobs: CARGO_INCREMENTAL: "0" CARGO_PROFILE_DEV_DEBUG: line-tables-only CARGO_PROFILE_TEST_DEBUG: line-tables-only - CROWDB_PACKAGE_SYMBOLS: "1" + CROWDB_PACKAGE_SYMBOLS: ${{ inputs.include_symbols && '1' || '0' }} permissions: contents: read outputs: @@ -86,11 +91,13 @@ jobs: - name: Check Rust formatting and lint run: pixi run rs-fmt-check && pixi run rs-lint - name: Verify exact-build source-line symbols + if: inputs.include_symbols run: pixi run -- python tools/ci-checks/check-container-symbols.py - name: Archive verified runtime files - run: | - pixi run tar -C target/container-runtime -czf target/container-runtime.tar.gz . - pixi run tar -C target/container-symbols -I zstd -cf "target/crowdb-symbols-${{ inputs.tag }}-git-${{ steps.source.outputs.revision }}-linux-amd64.tar.zst" . + run: pixi run tar -C target/container-runtime -czf target/container-runtime.tar.gz . + - name: Archive exact-build symbols + if: inputs.include_symbols + run: pixi run tar -C target/container-symbols -I zstd -cf "target/crowdb-symbols-${{ inputs.tag }}-git-${{ steps.source.outputs.revision }}-linux-amd64.tar.zst" . - uses: actions/upload-artifact@v4 with: name: verified-container-runtime @@ -98,6 +105,9 @@ jobs: compression-level: 0 retention-days: 7 - uses: actions/upload-artifact@v4 + id: symbols_artifact + if: inputs.include_symbols + continue-on-error: true with: name: verified-container-symbols path: target/crowdb-symbols-*.tar.zst @@ -150,6 +160,9 @@ jobs: name: verified-container-runtime path: target - uses: actions/download-artifact@v4 + id: symbols_download + if: inputs.include_symbols + continue-on-error: true with: name: verified-container-symbols path: target @@ -182,14 +195,20 @@ jobs: env: DIGEST: ${{ steps.build.outputs.digest }} run: pixi run cosign sign --yes "docker.io/crowdb/crowdb-iceberg@$DIGEST" - - name: Attach exact-build symbols to GitHub Release + - name: Publish verified GitHub Release env: GH_TOKEN: ${{ github.token }} RELEASE_TAG: ${{ inputs.tag }} - REVISION: ${{ needs.verify.outputs.revision }} - run: pixi run gh release upload "$RELEASE_TAG" "target/crowdb-symbols-$RELEASE_TAG-git-$REVISION-linux-amd64.tar.zst" --repo "$GITHUB_REPOSITORY" - - name: Publish verified GitHub Release + run: pixi run gh release edit "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --draft=false + - name: Attach exact-build symbols to GitHub Release + id: symbol_upload + if: inputs.include_symbols && steps.symbols_download.outcome == 'success' + continue-on-error: true env: GH_TOKEN: ${{ github.token }} RELEASE_TAG: ${{ inputs.tag }} - run: pixi run gh release edit "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --draft=false + REVISION: ${{ needs.verify.outputs.revision }} + run: pixi run gh release upload "$RELEASE_TAG" "target/crowdb-symbols-$RELEASE_TAG-git-$REVISION-linux-amd64.tar.zst" --repo "$GITHUB_REPOSITORY" + - name: Report optional symbol upload failure + if: inputs.include_symbols && (steps.symbols_download.outcome == 'failure' || steps.symbol_upload.outcome == 'failure') + run: echo '::warning::The image and release were published, but the optional symbol archive could not be uploaded.' diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index b7767c16..09d4264f 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -36,14 +36,17 @@ pixi run -- python tools/release.py --execute `--bump minor` and `--bump major` select larger version changes. The script updates every version manifest, commits and tags the release, atomically pushes `main` and the tag, creates a draft GitHub Release, then dispatches the existing -verified DockerHub workflow. Execution requires authenticated `gh` and GitHub +verified DockerHub workflow. Add `--symbols` to either command to include the +large exact-build symbol archive; the default release skips it. Execution +requires authenticated `gh` and GitHub permission to push `main`; the dry run changes no files or remote state. The -workflow archives the verified runtime and its separate exact-build symbol -package from one build, then packages those same runtime files in its publish -job without recompiling them. The symbol archive is attached to the GitHub -Release as `crowdb-symbols--git--linux-amd64.tar.zst`. -The workflow publishes the GitHub Release after the Docker image, signature -and symbol upload succeed. +workflow archives the verified runtime, then packages those same files in its +publish job without recompiling them. With `--symbols`, it also archives +exact-build symbols from that build and attaches +`crowdb-symbols--git--linux-amd64.tar.zst` to the GitHub +Release. The workflow publishes the GitHub Release after the Docker image and +signature succeed. If the optional symbol upload fails, the published release +remains available and the workflow reports a warning. ## Crash collection boundary @@ -52,8 +55,9 @@ The image does not configure the host's Linux core collector. Inspect the mounted data volume. A leading `|` sends a crash to a host-side collector; relative file patterns write in the crashing process's working directory. The container does not currently set a private core working directory, a -core size limit, or dump retention. Release debug symbols are provided -separately. Do not assume `/opt/crowdb/data` contains a core after a crash. +core size limit, or dump retention. Release debug symbols are available when +the release was run with `--symbols`. Do not assume `/opt/crowdb/data` contains +a core after a crash. - On a systemd-coredump host, use `coredumpctl list` and `coredumpctl dump` on the host to locate and export a captured dump. diff --git a/container/single-node-container/tests/release-policy.sh b/container/single-node-container/tests/release-policy.sh index 9cd98909..51d5917e 100644 --- a/container/single-node-container/tests/release-policy.sh +++ b/container/single-node-container/tests/release-policy.sh @@ -36,8 +36,13 @@ publish_job=$(sed -n '/^ publish:/,$p' "$release") [[ "$publish_job" == *'name: verified-container-runtime'* ]] [[ "$verify_job" == *'name: verified-container-symbols'* ]] [[ "$publish_job" == *'name: verified-container-symbols'* ]] -[[ "$verify_job" == *'CROWDB_PACKAGE_SYMBOLS: "1"'* ]] +[[ "$events" == *'include_symbols:'* && "$events" == *'default: false'* ]] +[[ "$verify_job" == *"CROWDB_PACKAGE_SYMBOLS: \${{ inputs.include_symbols && '1' || '0' }}"* ]] [[ "$verify_job" == *'pixi run -- python tools/ci-checks/check-container-symbols.py'* ]] +[[ "$verify_job" == *'if: inputs.include_symbols'* ]] +[[ "$publish_job" == *'if: inputs.include_symbols && steps.symbols_download.outcome'* ]] +[[ "$verify_job" == *'continue-on-error: true'* ]] +[[ "$publish_job" == *'continue-on-error: true'* ]] [[ "$publish_job" == *'gh release upload "$RELEASE_TAG"'* ]] [[ "$publish_job" == *'gh release edit "$RELEASE_TAG"'* ]] [[ "$publish_job" == *'context: target/container-runtime'* ]] @@ -56,7 +61,7 @@ ci_job=$(sed -n '/^ DockerPreview:/,$p' "$ci") ! grep -Eq 'secrets\.|docker/login-action|docker/build-push-action' <<<"$ci_job" release_tool=tools/release.py -for required in '--dry-run' '--execute' 'git", "push", "--atomic"' \ +for required in '--dry-run' '--execute' '--symbols' 'git", "push", "--atomic"' \ '"release", "create"' '"workflow", "run"'; do grep -Fq -- "$required" "$release_tool" done diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md index 01a3b79d..25566d8c 100644 --- a/doc/backlog/R188-console-group0-authority.md +++ b/doc/backlog/R188-console-group0-authority.md @@ -82,9 +82,10 @@ not block R187 completion. private data-volume location. Provide an exact-build source-line symbolization workflow for child and monitor crashes. Dumps can contain secrets and user data; diagnostics must not expose them in ordinary logs. - Ship exact-build debug symbols as a separate GitHub Release asset generated - from the same staged runtime as the image, indexed by version and source - revision. The release preparation script in `tools/` runs manually, shows a + Optionally ship exact-build debug symbols as a separate GitHub Release asset + generated from the same staged runtime as the image, indexed by version and + source revision. Omit this large asset by default so its upload cannot block + image publication. The release preparation script in `tools/` runs manually, shows a dry-run plan, updates versions, creates the tag and GitHub Release, then dispatches the existing verified DockerHub publication workflow. Host acceptance remains open; no host configuration change is assumed. @@ -160,10 +161,11 @@ not block R187 completion. - Given a clean main checkout and a version bump, when the release tool runs in dry-run mode, assert it shows every version change and no file or remote is modified. When run for a release, assert the tag and GitHub Release identify - the same verified revision, the symbol asset contains source-line information, - GNU debuglink CRCs and SHA-256 hashes match the image's stripped binaries. - Invariant: released symbols come from the image build and - remain available after a build host changes. E2E test. + the same verified revision and image publication succeeds without symbol + upload. When symbols are requested, assert the asset contains source-line + information and GNU debuglink CRCs and SHA-256 hashes match the image's + stripped binaries. Invariant: optional released symbols come from the image + build and remain available after a build host changes. E2E test. Required gates: diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 7016c106..93cac996 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -154,12 +154,12 @@ is paused; it does not block the single-node image requirement. Apport, systemd-coredump and Docker Desktop lookup paths without promising a volume dump. This host reports an Apport pipe pattern and core ulimit 0. Volume retention, exact-build symbols and disposable-host acceptance remain. -- [ ] **Manual release and symbols**: add a `tools/` release script with a - read-only dry run, consistent version updates, tag and GitHub Release - creation, and dispatch of the existing publication workflow. Extract debug - symbols from the same staged ELF files as the image, keep them out of the - image, and upload the version/revision-named archive to that release. - Verify symbol identity and source-line lookup. Files: +- [ ] **Manual release and optional symbols**: the `tools/` release script now + has a read-only dry run, consistent version updates, tag and GitHub Release + creation, and workflow dispatch. Optional `--symbols` extracts debug symbols + from the same staged ELF files as the image and uploads the named archive; + the default release skips that large asset. Local symbol identity checks pass. + A real release and source-line lookup remain to verify. Files: `tools/release.py`, `container/single-node-container/{build.sh,collect-libs.sh}`, `.github/workflows/release-container.yml`. ## Documentation and completion diff --git a/tools/release.py b/tools/release.py index 3a2dfe94..e5ca7f52 100644 --- a/tools/release.py +++ b/tools/release.py @@ -100,6 +100,7 @@ def main() -> None: mode.add_argument("--dry-run", action="store_true", help="Print the plan without writing or contacting GitHub") mode.add_argument("--execute", action="store_true", help="Update, tag, push, release and dispatch") parser.add_argument("--bump", choices=("patch", "minor", "major"), default="patch") + parser.add_argument("--symbols", action="store_true", help="Build and upload the large optional symbol archive") args = parser.parse_args() current = (ROOT / "VERSION").read_text(encoding="utf-8").strip() @@ -110,7 +111,7 @@ def main() -> None: for path in updates: print(f" update {path.relative_to(ROOT)}", flush=True) print(" check versions and diff; commit; tag; atomically push main + tag", flush=True) - print(" create draft GitHub Release; dispatch release-container.yml", flush=True) + print(f" create draft GitHub Release; dispatch release-container.yml (symbols: {args.symbols})", flush=True) if args.dry_run: for path, updated in updates.items(): original = path.read_text(encoding="utf-8") @@ -132,7 +133,7 @@ def main() -> None: command("git", "push", "--atomic", "origin", "HEAD:refs/heads/main", f"refs/tags/{tag}") command("gh", "release", "create", tag, "--repo", REPO, "--verify-tag", "--generate-notes", "--draft") command("gh", "workflow", "run", "release-container.yml", "--repo", REPO, - "--ref", tag, "-f", f"tag={tag}") + "--ref", tag, "-f", f"tag={tag}", "-f", f"include_symbols={str(args.symbols).lower()}") print(f"Started verified publication for {tag}") From d06eed249102044b6bf6fcc72e7642901a145daf Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 08:38:38 +0800 Subject: [PATCH 31/74] Group multipart records by upload --- doc/working/plan-s3-multipart.md | 5 +- lib/crowdb-access-s3/src/metadata/key.rs | 46 +++++++++++++------ .../src/metadata/multipart_repository.rs | 37 +++++++++++---- .../metadata/multipart_repository/listing.rs | 10 ++-- .../tests/metadata_key_test.rs | 15 ++++-- .../tests/multipart_repository_test.rs | 41 +++++++++++++++++ 6 files changed, 119 insertions(+), 35 deletions(-) diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 378cfcb6..3d96e057 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -38,8 +38,9 @@ Iceberg multipart path. confirmation after lost replies. An identical part record retry returns the existing revision; a new location remains a replacement. Completion snapshots and a predecessor-fenced metadata-only object publication path are - in place. The current session, part and generation key families still need - grouping under one upload prefix for R95. HTTP wiring remains. + in place. The session, current part and immutable generations now share one + upload prefix for R95; an immutable object-key/upload-ID index preserves + bounded ListMultipartUploads ordering. HTTP wiring remains. - [ ] **S3 routes and wire**: classify create/upload/list/complete/abort/list uploads, parse bounded completion XML, emit compatible responses and errors. Preserve SigV4 authentication and existing basic routes. The repository now diff --git a/lib/crowdb-access-s3/src/metadata/key.rs b/lib/crowdb-access-s3/src/metadata/key.rs index 73ef0ca0..96ac50b4 100644 --- a/lib/crowdb-access-s3/src/metadata/key.rs +++ b/lib/crowdb-access-s3/src/metadata/key.rs @@ -5,9 +5,11 @@ use std::fmt; const BUCKET_NAME_KIND: u8 = 1; const OBJECT_KIND: u8 = 2; -const MULTIPART_SESSION_KIND: u8 = 3; -const MULTIPART_PART_KIND: u8 = 4; -const MULTIPART_PART_GENERATION_KIND: u8 = 5; +const MULTIPART_INDEX_KIND: u8 = 3; +const MULTIPART_UPLOAD_KIND: u8 = 4; +const MULTIPART_SESSION_ENTRY: u8 = 0; +const MULTIPART_PART_ENTRY: u8 = 1; +const MULTIPART_GENERATION_ENTRY: u8 = 2; const MAX_KEY_BYTES: usize = 1024; /// A tenant namespace identity. @@ -154,12 +156,12 @@ impl MetadataKey { end } - /// Starts the ordered multipart upload interval for one bucket. + /// Starts the ordered multipart listing index for one bucket. #[must_use] pub fn multipart_session_prefix(tenant: &TenantId, bucket: BucketId) -> Vec { let mut key = namespace_prefix(tenant); key.extend_from_slice(bucket.as_bytes()); - key.push(MULTIPART_SESSION_KIND); + key.push(MULTIPART_INDEX_KIND); key } @@ -168,7 +170,7 @@ impl MetadataKey { pub fn multipart_session_end(tenant: &TenantId, bucket: BucketId) -> Vec { let mut key = namespace_prefix(tenant); key.extend_from_slice(bucket.as_bytes()); - key.push(MULTIPART_SESSION_KIND + 1); + key.push(MULTIPART_INDEX_KIND + 1); key } @@ -206,11 +208,11 @@ impl MetadataKey { Ok(end) } - /// Identifies one upload by object key and stable upload ID. + /// Identifies one immutable listing entry by object key and upload ID. /// /// # Errors /// Rejects an empty or oversized object key. - pub fn multipart_session( + pub fn multipart_upload_index( tenant: &TenantId, bucket: BucketId, object: &[u8], @@ -223,16 +225,32 @@ impl MetadataKey { Ok(key) } - /// Starts the part interval of one upload. + /// Starts the session and all part records for one upload. #[must_use] - pub fn multipart_part_prefix(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { + pub fn multipart_upload_prefix(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { let mut key = namespace_prefix(tenant); key.extend_from_slice(bucket.as_bytes()); - key.push(MULTIPART_PART_KIND); + key.push(MULTIPART_UPLOAD_KIND); key.extend_from_slice(upload_id); key } + /// Identifies the mutable session inside one upload interval. + #[must_use] + pub fn multipart_session(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { + let mut key = Self::multipart_upload_prefix(tenant, bucket, upload_id); + key.push(MULTIPART_SESSION_ENTRY); + key + } + + /// Starts the current-part interval of one upload. + #[must_use] + pub fn multipart_part_prefix(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { + let mut key = Self::multipart_upload_prefix(tenant, bucket, upload_id); + key.push(MULTIPART_PART_ENTRY); + key + } + /// Ends the part interval of one upload. #[must_use] pub fn multipart_part_end(tenant: &TenantId, bucket: BucketId, upload_id: &[u8; 16]) -> Vec { @@ -273,10 +291,8 @@ impl MetadataKey { if number == 0 || number > 10_000 || revision == 0 { return Err(MetadataKeyError::InvalidPartNumber); } - let mut key = namespace_prefix(tenant); - key.extend_from_slice(bucket.as_bytes()); - key.push(MULTIPART_PART_GENERATION_KIND); - key.extend_from_slice(upload_id); + let mut key = Self::multipart_upload_prefix(tenant, bucket, upload_id); + key.push(MULTIPART_GENERATION_ENTRY); key.extend_from_slice(&number.to_be_bytes()); key.extend_from_slice(&revision.to_be_bytes()); Ok(key) diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index a9c5372b..ed0392ba 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -58,8 +58,30 @@ impl MultipartRepository { if session.phase != MultipartPhase::Open || session.revision != 1 { return Err(MultipartRepositoryError::Conflict); } - let key = self.session_key(session)?; let value = session.encode()?; + let key = self.session_key(session); + let index = MetadataKey::multipart_upload_index( + &self.tenant, + session.bucket_id, + &session.object_key, + &session.upload_id, + )?; + let indexed = self.store.put_if_absent(index.clone(), key.clone()).await; + match indexed { + Ok(PutIfAbsentOutcome::Inserted { .. }) => {} + Ok(PutIfAbsentOutcome::Existing(existing)) if existing.value == key => {} + Ok(PutIfAbsentOutcome::Existing(_)) => return Err(MultipartRepositoryError::Conflict), + Err(error) => { + if !self + .store + .get(index) + .await? + .is_some_and(|entry| entry.value == key) + { + return Err(error.into()); + } + } + } let result = self.store.put_if_absent(key, value.clone()).await; match result { Ok(PutIfAbsentOutcome::Inserted { .. }) => Ok(session.clone()), @@ -83,7 +105,7 @@ impl MultipartRepository { &self, identity: &MultipartSessionRecord, ) -> Result, MultipartRepositoryError> { - let key = self.session_key(identity)?; + let key = self.session_key(identity); self.store .get(key) .await? @@ -116,7 +138,7 @@ impl MultipartRepository { { return Err(MultipartRepositoryError::Conflict); } - let key = self.session_key(previous)?; + let key = self.session_key(previous); let expected = previous.encode()?; let value = next.encode()?; match self.store.compare_exchange(key, expected, value).await { @@ -287,12 +309,7 @@ impl MultipartRepository { } } - fn session_key(&self, session: &MultipartSessionRecord) -> Result, MetadataKeyError> { - MetadataKey::multipart_session( - &self.tenant, - session.bucket_id, - &session.object_key, - &session.upload_id, - ) + fn session_key(&self, session: &MultipartSessionRecord) -> Vec { + MetadataKey::multipart_session(&self.tenant, session.bucket_id, &session.upload_id) } } diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs index 5926247d..c4af6993 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs @@ -51,7 +51,7 @@ impl MultipartRepository { let mut start = MetadataKey::multipart_session_key_prefix(&self.tenant, bucket, prefix)?; let end = MetadataKey::multipart_session_key_prefix_end(&self.tenant, bucket, prefix)?; if let Some(key) = key_marker { - let mut after = MetadataKey::multipart_session( + let mut after = MetadataKey::multipart_upload_index( &self.tenant, bucket, key, @@ -82,14 +82,18 @@ impl MultipartRepository { .await?; scanned += page.items.len(); for item in page.items { - let session = MultipartSessionRecord::decode_unbound(&item.value)?; + let Some(record) = self.store.get(item.value).await? else { + continue; + }; + let session = MultipartSessionRecord::decode_unbound(&record.value)?; if session.bucket_id != bucket - || MetadataKey::multipart_session( + || MetadataKey::multipart_upload_index( &self.tenant, bucket, &session.object_key, &session.upload_id, )? != item.key + || MetadataKey::multipart_session(&self.tenant, bucket, &session.upload_id) != record.key { return Err(MultipartRepositoryError::Conflict); } diff --git a/lib/crowdb-access-s3/tests/metadata_key_test.rs b/lib/crowdb-access-s3/tests/metadata_key_test.rs index ff16a33c..76cc3e52 100644 --- a/lib/crowdb-access-s3/tests/metadata_key_test.rs +++ b/lib/crowdb-access-s3/tests/metadata_key_test.rs @@ -63,14 +63,16 @@ fn maximum_binary_object_key_stays_within_its_bucket_interval() { } #[test] -fn multipart_keys_keep_uploads_and_parts_in_separate_bounded_intervals() { +fn multipart_keys_group_session_current_parts_and_generations_by_upload() { let tenant = TenantId::new(b"tenant".to_vec()).unwrap(); let bucket = BucketId::new([5; 16]); let upload = [9; 16]; - let session = MetadataKey::multipart_session(&tenant, bucket, b"a\0", &upload).unwrap(); + let index = MetadataKey::multipart_upload_index(&tenant, bucket, b"a\0", &upload).unwrap(); let start = MetadataKey::multipart_session_prefix(&tenant, bucket); let end = MetadataKey::multipart_session_end(&tenant, bucket); - assert!(start < session && session < end); + assert!(start < index && index < end); + let upload_prefix = MetadataKey::multipart_upload_prefix(&tenant, bucket, &upload); + let session = MetadataKey::multipart_session(&tenant, bucket, &upload); let part_start = MetadataKey::multipart_part_prefix(&tenant, bucket, &upload); let part_end = MetadataKey::multipart_part_end(&tenant, bucket, &upload); let first = MetadataKey::multipart_part(&tenant, bucket, &upload, 1).unwrap(); @@ -79,7 +81,10 @@ fn multipart_keys_keep_uploads_and_parts_in_separate_bounded_intervals() { assert!(MetadataKey::multipart_part(&tenant, bucket, &upload, 0).is_err()); assert!(MetadataKey::multipart_part(&tenant, bucket, &upload, 10_001).is_err()); assert!(MetadataKey::object_end(&tenant, bucket) <= start); - assert!(end <= part_start); + assert!(end <= upload_prefix); + assert!(session.starts_with(&upload_prefix)); + assert!(part_start.starts_with(&upload_prefix)); + assert!(session < part_start); let generation = MetadataKey::multipart_part_generation(&tenant, bucket, &upload, 1, 2).unwrap(); - assert!(part_end < generation); + assert!(part_end < generation && generation.starts_with(&upload_prefix)); } diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs index d8987b1d..9bad4a51 100644 --- a/lib/crowdb-access-s3/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -576,6 +576,47 @@ async fn part_listing_paginates_current_generations_in_number_order() { assert_eq!(second.next_part_number_marker, None); } +#[tokio::test] +async fn upload_prefix_retains_session_and_replaced_part_generations() { + let (repository, _, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + repository.put_stream_part(&session, &part(), 110).await.unwrap(); + let mut replacement = part(); + replacement.locations[0].offset += 39; + repository + .put_stream_part(&session, &replacement, 111) + .await + .unwrap(); + repository.abort(&session).await.unwrap(); + + let tenant = TenantId::new(b"tenant".to_vec()).unwrap(); + let prefix = MetadataKey::multipart_upload_prefix(&tenant, session.bucket_id, &session.upload_id); + let mut end = prefix.clone(); + end.push(u8::MAX); + let records = store.scan(prefix, end, 10, 1024 * 1024).await.unwrap(); + assert_eq!(records.len(), 4); + assert_eq!( + records[0].key, + MetadataKey::multipart_session(&tenant, session.bucket_id, &session.upload_id) + ); + assert_eq!( + records[1].key, + MetadataKey::multipart_part(&tenant, session.bucket_id, &session.upload_id, 1).unwrap() + ); + for revision in [1, 2] { + assert!(records.iter().any(|record| record.key + == MetadataKey::multipart_part_generation( + &tenant, + session.bucket_id, + &session.upload_id, + 1, + revision, + ) + .unwrap())); + } +} + #[tokio::test] async fn upload_listing_filters_terminal_records_and_resumes_same_key() { let (repository, _, _) = repository().await; From a463952362bda6cdc6d431d4d22323868d44d017 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 08:51:59 +0800 Subject: [PATCH 32/74] Serve durable S3 multipart uploads --- app/crowdb-access-server/src/s3/operations.rs | 19 +- .../src/s3/operations/multipart.rs | 387 ++++++++++++++++++ .../tests/s3_e2e/basic.py | 64 +++ .../tests/s3_full_stack_test.rs | 3 +- doc/working/plan-s3-multipart.md | 18 +- .../src/metadata/multipart_repository.rs | 38 +- .../tests/multipart_repository_test.rs | 27 +- 7 files changed, 530 insertions(+), 26 deletions(-) create mode 100644 app/crowdb-access-server/src/s3/operations/multipart.rs diff --git a/app/crowdb-access-server/src/s3/operations.rs b/app/crowdb-access-server/src/s3/operations.rs index b86313d2..7a0a4752 100644 --- a/app/crowdb-access-server/src/s3/operations.rs +++ b/app/crowdb-access-server/src/s3/operations.rs @@ -39,6 +39,8 @@ use crate::storage::S3StorageClients; use super::{error_response, full_body, install_body_receive_provider, BoxError, ResponseBody}; use crowdb_access_s3::wire; +mod multipart; + const DEFAULT_LIST_LIMIT: usize = 1_000; const DEFAULT_LIST_SCAN_BYTES: usize = 4 * 1024 * 1024; @@ -143,12 +145,12 @@ impl ProductionS3Operations { S3Operation::GetObject => self.get_object(route, &request, &request_id).await, S3Operation::ListObjectsV2 => self.list_objects(route, &request).await, S3Operation::DeleteObject => self.delete_object(route).await, - S3Operation::CreateMultipartUpload - | S3Operation::UploadPart - | S3Operation::ListParts - | S3Operation::CompleteMultipartUpload - | S3Operation::AbortMultipartUpload - | S3Operation::ListMultipartUploads => Err(S3ErrorCode::NotImplemented), + S3Operation::CreateMultipartUpload => self.create_multipart_upload(route, &request).await, + S3Operation::UploadPart => self.upload_part(route, request).await, + S3Operation::ListParts => self.list_parts(route, &request).await, + S3Operation::CompleteMultipartUpload => self.complete_multipart_upload(route, request).await, + S3Operation::AbortMultipartUpload => self.abort_multipart_upload(route).await, + S3Operation::ListMultipartUploads => self.list_multipart_uploads(route, &request).await, }; self.record_dependency_outcome(operation, &result); result.unwrap_or_else(|code| { @@ -165,7 +167,10 @@ impl ProductionS3Operations { let Some(health) = &self.health else { return; }; - let uses_chunks = matches!(operation, S3Operation::PutObject | S3Operation::GetObject); + let uses_chunks = matches!( + operation, + S3Operation::PutObject | S3Operation::GetObject | S3Operation::UploadPart + ); match result { Ok(_) => { health.set_metadata(DependencyHealth::Ready); diff --git a/app/crowdb-access-server/src/s3/operations/multipart.rs b/app/crowdb-access-server/src/s3/operations/multipart.rs new file mode 100644 index 00000000..06975c1a --- /dev/null +++ b/app/crowdb-access-server/src/s3/operations/multipart.rs @@ -0,0 +1,387 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Authenticated S3 multipart operations over durable session and part records. + +use crowdb_access_s3::error::S3ErrorCode; +use crowdb_access_s3::integrity::{IntegrityError, SinglePartIntegrity}; +use crowdb_access_s3::metadata::{ + new_upload_id, CompletionPart, MultipartPartRecord, MultipartPhase, MultipartRepository, + MultipartRepositoryError, MultipartSessionRecord, +}; +use crowdb_access_s3::route::S3Route; +use crowdb_access_s3::streaming::{ + write_body_with_checksums_buffered, write_native_body_with_checksums_metered, +}; +use crowdb_access_s3::wire; +use crowdb_chunk_client::ChunkIoWriter; +use http_body_util::BodyExt as _; +use hyper::body::Bytes; +use hyper::body::Incoming; +use hyper::header::ETAG; +use hyper::{Request, Response, StatusCode}; + +use super::{ + content_length, full_body, install_body_receive_provider, map_put_outcome, required_bucket, required_key, + response, strict_header, unix_millis, xml_response, ProductionS3Operations, Query, ResponseBody, +}; +use crate::multipart_complete::CompleteSelection; + +const MAX_PARTS: u16 = 10_000; +const MAX_PART_BYTES: u64 = 5 * 1024 * 1024 * 1024; +const MAX_OBJECT_BYTES: u64 = 5 * 1024 * 1024 * 1024 * 1024; +const MAX_STAGED_BYTES: u64 = 10_000 * MAX_PART_BYTES; +const UPLOAD_LIFETIME_MS: u64 = 7 * 24 * 60 * 60 * 1000; +const MAX_COMPLETE_BODY: usize = 2 * 1024 * 1024; + +impl ProductionS3Operations { + fn multipart(&self) -> MultipartRepository { + MultipartRepository::new(self.storage.metadata.clone(), self.config.tenant.clone()) + } + + async fn multipart_identity(&self, route: &S3Route) -> Result { + let bucket = self.resolve_bucket(required_bucket(route)?).await?; + let key = required_key(route)?; + let upload_id = route.upload_id.ok_or(S3ErrorCode::InvalidRequest)?; + self.multipart() + .load_identity(bucket, key, &upload_id) + .await + .map_err(|error| map_multipart_error(&error))? + .ok_or(S3ErrorCode::NoSuchUpload) + } + + pub(super) async fn create_multipart_upload( + &self, + route: S3Route, + request: &Request, + ) -> Result, S3ErrorCode> { + let bucket_name = required_bucket(&route)?; + let key = required_key(&route)?; + let bucket_id = self.resolve_bucket(bucket_name).await?; + if content_length(request)?.is_some_and(|length| length != 0) { + return Err(S3ErrorCode::InvalidRequest); + } + let now = unix_millis(); + let session = MultipartSessionRecord { + bucket_id, + object_key: key.to_vec(), + upload_id: new_upload_id(now), + revision: 1, + phase: MultipartPhase::Open, + created_ms: now, + expires_ms: now + .checked_add(UPLOAD_LIFETIME_MS) + .ok_or(S3ErrorCode::InternalError)?, + content_type: strict_header(request, "content-type", S3ErrorCode::InvalidRequest)? + .unwrap_or("application/octet-stream") + .to_owned(), + max_parts: MAX_PARTS, + max_part_bytes: MAX_PART_BYTES, + max_object_bytes: MAX_OBJECT_BYTES, + max_staged_bytes: MAX_STAGED_BYTES, + part_count: 0, + staged_bytes: 0, + selection: None, + completion_request_digest: None, + publication_ms: None, + object_predecessor: None, + etag: None, + }; + self.multipart() + .begin(&session) + .await + .map_err(|error| map_multipart_error(&error))?; + xml_response(wire::create_multipart_upload( + bucket_name, + key, + &session.upload_id, + )) + } + + pub(super) async fn upload_part( + &self, + route: S3Route, + mut request: Request, + ) -> Result, S3ErrorCode> { + let session = self.multipart_identity(&route).await?; + let number = route.part_number.ok_or(S3ErrorCode::InvalidRequest)?; + let length = content_length(&request)?.ok_or(S3ErrorCode::InvalidRequest)?; + if length > session.max_part_bytes { + return Err(S3ErrorCode::InvalidRequest); + } + let content_md5 = + strict_header(&request, "content-md5", S3ErrorCode::InvalidDigest)?.map(str::to_owned); + let payload_sha256 = strict_header(&request, "x-amz-content-sha256", S3ErrorCode::InvalidRequest)? + .filter(|value| *value != "UNSIGNED-PAYLOAD") + .map(str::to_owned); + let mut route_key = self.config.tenant.as_bytes().to_vec(); + route_key.extend_from_slice(session.bucket_id.as_bytes()); + route_key.extend_from_slice(&session.upload_id); + route_key.extend_from_slice(&number.to_be_bytes()); + let mut writer = self.prepare_writer(Some(length), &route_key).await?; + let native_receiver = if writer.is_large() { + install_body_receive_provider(&mut request) + } else { + None + }; + if let Some(receiver) = &native_receiver { + receiver.enable_owner_handoff(); + } + let mut body = request.into_body(); + let written = if let Some(receiver) = native_receiver.as_deref() { + write_native_body_with_checksums_metered( + &mut body, + &mut writer, + receiver, + content_md5.as_deref(), + payload_sha256.as_deref(), + self.metrics.as_deref(), + ) + .await + } else { + let receive_bytes = usize::try_from(length) + .unwrap_or(1024 * 1024) + .clamp(1, 1024 * 1024); + write_body_with_checksums_buffered( + &mut body, + &mut writer, + content_md5.as_deref(), + payload_sha256.as_deref(), + receive_bytes, + self.metrics.as_deref(), + ) + .await + }; + let (etag, _) = match written { + Ok(value) => value, + Err(outcome) => { + let _ = writer.on_error().await; + return Err(map_put_outcome(&outcome)); + } + }; + let locations = writer + .on_finish() + .await + .map_err(|_| S3ErrorCode::ServiceUnavailable)?; + let actual: u64 = locations.iter().map(|location| location.logical_length).sum(); + if actual != length { + return Err(S3ErrorCode::InvalidRequest); + } + let part = MultipartPartRecord { + bucket_id: session.bucket_id, + upload_id: session.upload_id, + number, + revision: 1, + modified_ms: unix_millis(), + length, + raw_md5: parse_md5(&etag)?, + locations, + }; + let _saved = self + .multipart() + .put_stream_part(&session, &part, part.modified_ms) + .await + .map_err(|error| map_multipart_error(&error))? + .ok_or(S3ErrorCode::SlowDown)?; + Response::builder() + .status(StatusCode::OK) + .header(ETAG, format!("\"{etag}\"")) + .body(full_body(Vec::new().into())) + .map_err(|_| S3ErrorCode::InternalError) + } + + pub(super) async fn list_parts( + &self, + route: S3Route, + request: &Request, + ) -> Result, S3ErrorCode> { + let session = self.multipart_identity(&route).await?; + let query = Query::new(request.uri().query()); + let marker = parse_number(query.text("part-number-marker"), 0, 0, 10_000)?; + let limit = parse_number(query.text("max-parts"), 1_000, 1, 1_000)?; + let page = self + .multipart() + .list_parts(&session, marker, usize::from(limit)) + .await + .map_err(|error| map_multipart_error(&error))?; + xml_response(wire::list_multipart_parts( + required_bucket(&route)?, + required_key(&route)?, + &session.upload_id, + marker, + usize::from(limit), + &page, + )) + } + + pub(super) async fn list_multipart_uploads( + &self, + route: S3Route, + request: &Request, + ) -> Result, S3ErrorCode> { + let bucket_name = required_bucket(&route)?; + let bucket = self.resolve_bucket(bucket_name).await?; + let query = Query::new(request.uri().query()); + let prefix = query.bytes("prefix").unwrap_or_default(); + let key_marker = query.bytes("key-marker"); + let upload_marker = query + .text("upload-id-marker") + .as_deref() + .map(parse_upload_id) + .transpose()?; + let limit = parse_number(query.text("max-uploads"), 1_000, 1, 1_000)?; + let page = self + .multipart() + .list_uploads( + bucket, + &prefix, + key_marker.as_deref(), + upload_marker.as_ref(), + usize::from(limit), + unix_millis(), + ) + .await + .map_err(|error| map_multipart_error(&error))?; + xml_response(wire::list_multipart_uploads( + bucket_name, + &prefix, + key_marker.as_deref(), + upload_marker.as_ref(), + usize::from(limit), + &String::from_utf8_lossy(self.config.tenant.as_bytes()), + &page, + )) + } + + pub(super) async fn abort_multipart_upload( + &self, + route: S3Route, + ) -> Result, S3ErrorCode> { + let session = self.multipart_identity(&route).await?; + self.multipart() + .abort(&session) + .await + .map_err(|error| map_multipart_error(&error))?; + response(StatusCode::NO_CONTENT, Vec::new()) + } + + pub(super) async fn complete_multipart_upload( + &self, + route: S3Route, + request: Request, + ) -> Result, S3ErrorCode> { + let session = self.multipart_identity(&route).await?; + let location = request.uri().path().to_owned(); + let content_md5 = + strict_header(&request, "content-md5", S3ErrorCode::InvalidDigest)?.map(str::to_owned); + let payload_sha256 = strict_header(&request, "x-amz-content-sha256", S3ErrorCode::InvalidRequest)? + .filter(|value| *value != "UNSIGNED-PAYLOAD") + .map(str::to_owned); + let bytes = read_completion(request.into_body()).await?; + let mut integrity = SinglePartIntegrity::new(payload_sha256.is_some()); + integrity.update(&Bytes::copy_from_slice(&bytes)); + integrity + .finish_validated_checksums(content_md5.as_deref(), payload_sha256.as_deref()) + .map_err(map_integrity_error)?; + let selection = CompleteSelection::parse(&bytes).map_err(|_| S3ErrorCode::InvalidRequest)?; + let requested: Vec = selection + .parts() + .iter() + .map(|part| CompletionPart { + number: part.number, + etag: part.etag.clone(), + }) + .collect(); + let frozen = self + .multipart() + .freeze_completion(&session, &requested, unix_millis()) + .await + .map_err(|error| map_multipart_error(&error))? + .ok_or(S3ErrorCode::ServiceUnavailable)?; + let published = self + .multipart() + .publish_completion(&frozen) + .await + .map_err(|error| map_multipart_error(&error))?; + xml_response(wire::complete_multipart_upload( + &location, + required_bucket(&route)?, + required_key(&route)?, + published.etag.as_deref().ok_or(S3ErrorCode::InternalError)?, + )) + } +} + +fn map_multipart_error(error: &MultipartRepositoryError) -> S3ErrorCode { + match error { + MultipartRepositoryError::Key(_) | MultipartRepositoryError::Record(_) => S3ErrorCode::InvalidRequest, + MultipartRepositoryError::Store(_) => S3ErrorCode::ServiceUnavailable, + MultipartRepositoryError::Conflict => S3ErrorCode::NoSuchUpload, + MultipartRepositoryError::InvalidPart => S3ErrorCode::InvalidPart, + MultipartRepositoryError::EntityTooSmall => S3ErrorCode::EntityTooSmall, + MultipartRepositoryError::ScanBudgetExhausted => S3ErrorCode::SlowDown, + } +} + +fn map_integrity_error(error: IntegrityError) -> S3ErrorCode { + match error { + IntegrityError::InvalidDigest => S3ErrorCode::InvalidDigest, + IntegrityError::Mismatch => S3ErrorCode::BadDigest, + IntegrityError::InvalidPayloadDigest => S3ErrorCode::InvalidRequest, + IntegrityError::PayloadMismatch => S3ErrorCode::XAmzContentSHA256Mismatch, + } +} + +fn parse_number(value: Option, default: u16, minimum: u16, maximum: u16) -> Result { + value + .map_or(Ok(default), |value| value.parse::()) + .map_err(|_| S3ErrorCode::InvalidRequest) + .and_then(|number| { + (number >= minimum && number <= maximum) + .then_some(number) + .ok_or(S3ErrorCode::InvalidRequest) + }) +} + +fn parse_upload_id(value: &str) -> Result<[u8; 16], S3ErrorCode> { + if value.len() != 32 { + return Err(S3ErrorCode::InvalidRequest); + } + let mut id = [0; 16]; + for (byte, pair) in id.iter_mut().zip(value.as_bytes().chunks_exact(2)) { + *byte = u8::from_str_radix( + std::str::from_utf8(pair).map_err(|_| S3ErrorCode::InvalidRequest)?, + 16, + ) + .map_err(|_| S3ErrorCode::InvalidRequest)?; + } + (id != [0; 16]).then_some(id).ok_or(S3ErrorCode::InvalidRequest) +} + +fn parse_md5(value: &str) -> Result<[u8; 16], S3ErrorCode> { + if value.len() != 32 { + return Err(S3ErrorCode::InternalError); + } + let mut md5 = [0; 16]; + for (byte, pair) in md5.iter_mut().zip(value.as_bytes().chunks_exact(2)) { + *byte = u8::from_str_radix( + std::str::from_utf8(pair).map_err(|_| S3ErrorCode::InternalError)?, + 16, + ) + .map_err(|_| S3ErrorCode::InternalError)?; + } + Ok(md5) +} + +async fn read_completion(mut body: Incoming) -> Result, S3ErrorCode> { + let mut bytes = Vec::new(); + while let Some(frame) = body.frame().await { + let frame = frame.map_err(|_| S3ErrorCode::InvalidRequest)?; + let data = frame.into_data().map_err(|_| S3ErrorCode::InvalidRequest)?; + if bytes.len().saturating_add(data.len()) > MAX_COMPLETE_BODY { + return Err(S3ErrorCode::InvalidRequest); + } + bytes.extend_from_slice(&data); + } + Ok(bytes) +} diff --git a/app/crowdb-access-server/tests/s3_e2e/basic.py b/app/crowdb-access-server/tests/s3_e2e/basic.py index 0b22c586..2f5028f7 100644 --- a/app/crowdb-access-server/tests/s3_e2e/basic.py +++ b/app/crowdb-access-server/tests/s3_e2e/basic.py @@ -236,6 +236,70 @@ def test_independent_frontends_share_one_namespace(self): self.assertEqual(absent.exception.response["ResponseMetadata"]["HTTPStatusCode"], 404) second.delete_bucket(Bucket=bucket) + def test_multipart_replaces_parts_and_publishes_selected_bytes(self): + bucket = f"{self.bucket}-multipart" + key = "parts/object.bin" + first = b"a" * (5 * 1024 * 1024) + replacement = b"b" * len(first) + tail = b"final-part" + self.client.create_bucket(Bucket=bucket) + try: + upload_id = self.client.create_multipart_upload(Bucket=bucket, Key=key)["UploadId"] + self.assertIn(upload_id, [item["UploadId"] for item in + self.client.list_multipart_uploads(Bucket=bucket)["Uploads"]]) + tail_etag = self.client.upload_part( + Bucket=bucket, Key=key, UploadId=upload_id, PartNumber=2, Body=tail, + )["ETag"] + self.assertEqual(self.client.upload_part( + Bucket=bucket, Key=key, UploadId=upload_id, PartNumber=2, Body=tail, + )["ETag"], tail_etag) + self.client.upload_part(Bucket=bucket, Key=key, UploadId=upload_id, + PartNumber=1, Body=first) + first_etag = self.client.upload_part( + Bucket=bucket, Key=key, UploadId=upload_id, PartNumber=1, Body=replacement, + )["ETag"] + listed = self.client.list_parts(Bucket=bucket, Key=key, UploadId=upload_id) + self.assertEqual([part["PartNumber"] for part in listed["Parts"]], [1, 2]) + self.assertEqual(listed["Parts"][0]["ETag"], first_etag) + completed = self.client.complete_multipart_upload( + Bucket=bucket, Key=key, UploadId=upload_id, + MultipartUpload={"Parts": [ + {"PartNumber": 1, "ETag": first_etag}, + {"PartNumber": 2, "ETag": tail_etag}, + ]}, + ) + expected_etag = md5(md5(replacement).digest() + md5(tail).digest()).hexdigest() + "-2" + self.assertEqual(completed["ETag"], f'"{expected_etag}"') + replayed = self.client.complete_multipart_upload( + Bucket=bucket, Key=key, UploadId=upload_id, + MultipartUpload={"Parts": [ + {"PartNumber": 1, "ETag": first_etag}, + {"PartNumber": 2, "ETag": tail_etag}, + ]}, + ) + self.assertEqual(replayed["ETag"], completed["ETag"]) + self.assertEqual(self.client.get_object(Bucket=bucket, Key=key)["Body"].read(), replacement + tail) + self.client.delete_object(Bucket=bucket, Key=key) + + aborted = self.client.create_multipart_upload(Bucket=bucket, Key=key)["UploadId"] + self.client.upload_part(Bucket=bucket, Key=key, UploadId=aborted, PartNumber=1, Body=tail) + self.client.abort_multipart_upload(Bucket=bucket, Key=key, UploadId=aborted) + self.client.abort_multipart_upload(Bucket=bucket, Key=key, UploadId=aborted) + self.assertNotIn(aborted, [item["UploadId"] for item in + self.client.list_multipart_uploads(Bucket=bucket).get("Uploads", [])]) + + invalid = self.client.create_multipart_upload(Bucket=bucket, Key=key)["UploadId"] + self.client.upload_part(Bucket=bucket, Key=key, UploadId=invalid, PartNumber=1, Body=tail) + with self.assertRaises(ClientError) as mismatch: + self.client.complete_multipart_upload( + Bucket=bucket, Key=key, UploadId=invalid, + MultipartUpload={"Parts": [{"PartNumber": 1, "ETag": '"' + "00" * 16 + '"'}]}, + ) + self.assertEqual(mismatch.exception.response["Error"]["Code"], "InvalidPart") + self.client.abort_multipart_upload(Bucket=bucket, Key=key, UploadId=invalid) + finally: + self.client.delete_bucket(Bucket=bucket) + def test_slow_signed_upload_releases_native_buffers(self): bucket = f"{self.bucket}-slow" path = f"/{bucket}/slow.bin" diff --git a/app/crowdb-access-server/tests/s3_full_stack_test.rs b/app/crowdb-access-server/tests/s3_full_stack_test.rs index 77fdbf56..a67640c7 100644 --- a/app/crowdb-access-server/tests/s3_full_stack_test.rs +++ b/app/crowdb-access-server/tests/s3_full_stack_test.rs @@ -32,9 +32,10 @@ use hyper::body::Bytes; use serde_json::json; const MASTER_KEY: &str = "1111111111111111111111111111111111111111111111111111111111111111"; -const TEST_COUNT: usize = 18; +const TEST_COUNT: usize = 19; const BOTO3_CASES: &[&str] = &[ "test_signed_raw_http_wire_contract", + "test_multipart_replaces_parts_and_publishes_selected_bytes", "test_independent_frontends_share_one_namespace", "test_slow_signed_upload_releases_native_buffers", "test_truncated_signed_upload_does_not_publish_and_releases_credit", diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 3d96e057..eea548fd 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -46,7 +46,7 @@ Iceberg multipart path. Preserve SigV4 authentication and existing basic routes. The repository now provides bounded, ordered ListParts pagination over current generations; multipart query shapes now enter the authenticated dispatcher with distinct - metrics, but production operations still return `NotImplemented`. Bounded + metrics. Bounded upload listing now paginates active sessions by key and upload ID, skipping terminal/expired records and failing on scan-budget exhaustion; HTTP dispatch remains pending. @@ -54,16 +54,24 @@ Iceberg multipart path. codes and the create, complete, ListParts, and ListMultipartUploads XML response builders have focused tests. The bounded completion XML parser now has one implementation in access-server and is exposed by both - the Iceberg and S3 protocol modules. The S3 HTTP operations still need wiring. + the Iceberg and S3 protocol modules. The six S3 HTTP operations are wired. + Full boto3 stack acceptance passes Create, UploadPart, ListParts, Complete, + Abort and ListUploads, including replay, replacement and invalid ETag cases. - [ ] **Part ingestion**: reuse the bounded streaming writer and admission budget, persist part location/integrity before success, reconcile lost replies. + Production UploadPart now uses the basic streaming writer and saves raw MD5 + with locations. A byte-identical retry keeps the selected generation even + when a new write produced different chunk locations; R95 can reclaim those + unreachable chunks. Full stack ingestion passes 5 MiB and small parts; real + response-loss and restart acceptance remain. - [ ] **Atomic completion**: fence selected part generations, validate order, count, size and checksum, compose locations through the shared core, and publish one immutable object generation without reading part bytes. The S3 adapter now freezes selection under session CAS, validates it again before object-key CAS, and confirms exact publication after response loss. An immutable generation records preserve selected bytes across a concurrent - part-number replacement. The HTTP path and end-to-end publication test remain. + part-number replacement. Metadata-only completion and byte-exact GET pass + the full boto3 stack, including a repeated Complete request. - [ ] **Abort and expiry**: make terminal states idempotent and preserve the part generations that R95's chunk-centered scanner needs for reference checks. The S3 adapter now has an idempotent, response-loss-safe logical abort; the @@ -73,7 +81,9 @@ Iceberg multipart path. - [ ] **Focused and E2E tests**: known MD5 vectors, out-of-order/replaced parts, invalid completion, response loss and restart, abort/expiry metadata, and - ordinary single-part compatibility. + ordinary single-part compatibility. The complete S3 library and access-server + suites plus 19 full-stack boto3/restart cases pass. Multipart response-loss, + expiry and restart cases remain. - [ ] **Gates and docs**: run both access crate suites, access-server E2E, Rust fmt and clippy separately; update S3 design and remove R167 plus this plan only after all acceptance criteria pass. diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index ed0392ba..27df6dfb 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -105,18 +105,33 @@ impl MultipartRepository { &self, identity: &MultipartSessionRecord, ) -> Result, MultipartRepositoryError> { - let key = self.session_key(identity); + self.load_identity(identity.bucket_id, &identity.object_key, &identity.upload_id) + .await + } + + /// Reads one session through its bucket, object and upload identity. + /// + /// # Errors + /// Rejects corrupt or foreign values and storage failures. + pub async fn load_identity( + &self, + bucket: super::BucketId, + object: &[u8], + upload_id: &[u8; 16], + ) -> Result, MultipartRepositoryError> { + let key = MetadataKey::multipart_upload_index(&self.tenant, bucket, object, upload_id)?; + let Some(index) = self.store.get(key).await? else { + return Ok(None); + }; + let expected = MetadataKey::multipart_session(&self.tenant, bucket, upload_id); + if index.value != expected { + return Err(MultipartRepositoryError::Conflict); + } self.store - .get(key) + .get(expected) .await? .map(|value| { - MultipartSessionRecord::decode( - &value.value, - identity.bucket_id, - &identity.object_key, - &identity.upload_id, - ) - .map_err(Into::into) + MultipartSessionRecord::decode(&value.value, bucket, object, upload_id).map_err(Into::into) }) .transpose() } @@ -239,10 +254,7 @@ impl MultipartRepository { } let before = self.part(¤t, part.number).await?; if let Some(existing) = &before { - if existing.length == part.length - && existing.raw_md5 == part.raw_md5 - && existing.locations == part.locations - { + if existing.length == part.length && existing.raw_md5 == part.raw_md5 { return Ok(Some(existing.clone())); } } diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs index 9bad4a51..54c629a2 100644 --- a/lib/crowdb-access-s3/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -302,8 +302,18 @@ async fn session_cas_and_independent_part_replacement_obey_the_freeze() { .unwrap() .unwrap(); assert_eq!(replay, first); + let mut relocated_retry = part(); + relocated_retry.locations[0].offset += 10; + assert_eq!( + repository + .put_stream_part(&session, &relocated_retry, 111) + .await + .unwrap(), + Some(first.clone()) + ); let mut replacement = part(); replacement.locations[0].offset += 39; + replacement.raw_md5 = [8; 16]; let second = repository .put_stream_part(&session, &replacement, 112) .await @@ -480,6 +490,7 @@ async fn frozen_part_generation_survives_a_late_pointer_change() { repository.put_stream_part(&session, &part(), 110).await.unwrap(); let mut replacement = part(); replacement.locations[0].offset += 39; + replacement.raw_md5 = [8; 16]; let selected = repository .put_stream_part(&session, &replacement, 111) .await @@ -487,7 +498,7 @@ async fn frozen_part_generation_survives_a_late_pointer_change() { .unwrap(); let request = [CompletionPart { number: 1, - etag: "09".repeat(16), + etag: "08".repeat(16), }]; let frozen = repository .freeze_completion(&session, &request, 120) @@ -557,6 +568,7 @@ async fn part_listing_paginates_current_generations_in_number_order() { let mut replacement = part(); replacement.number = 2; replacement.locations[0].offset += 39; + replacement.raw_md5 = [8; 16]; repository .put_stream_part(&session, &replacement, 111) .await @@ -581,9 +593,22 @@ async fn upload_prefix_retains_session_and_replaced_part_generations() { let (repository, _, store) = repository().await; let session = session(); repository.begin(&session).await.unwrap(); + assert_eq!( + repository + .load_identity(session.bucket_id, &session.object_key, &session.upload_id) + .await + .unwrap(), + Some(session.clone()) + ); + assert!(repository + .load_identity(session.bucket_id, b"another", &session.upload_id) + .await + .unwrap() + .is_none()); repository.put_stream_part(&session, &part(), 110).await.unwrap(); let mut replacement = part(); replacement.locations[0].offset += 39; + replacement.raw_md5 = [8; 16]; repository .put_stream_part(&session, &replacement, 111) .await From 512f16ccb7b90336dfaddeb2c8ef44f44e302c76 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 09:00:05 +0800 Subject: [PATCH 33/74] Recover and expire multipart sessions --- app/crowdb-access-server/src/main.rs | 17 ++++ .../src/s3/operations/multipart.rs | 52 +++++++++- .../tests/s3_e2e/lost_reply.py | 95 +++++++++++++------ .../tests/s3_e2e/restart.py | 22 +++++ .../tests/s3_full_stack_test.rs | 2 +- doc/working/plan-s3-multipart.md | 18 ++-- lib/crowdb-access-s3/src/metadata.rs | 3 +- .../src/metadata/multipart_repository.rs | 1 + .../metadata/multipart_repository/terminal.rs | 87 ++++++++++++++++- .../tests/multipart_repository_test.rs | 41 ++++++++ 10 files changed, 296 insertions(+), 42 deletions(-) diff --git a/app/crowdb-access-server/src/main.rs b/app/crowdb-access-server/src/main.rs index dfe8eb8f..7c092373 100644 --- a/app/crowdb-access-server/src/main.rs +++ b/app/crowdb-access-server/src/main.rs @@ -116,6 +116,7 @@ async fn run_s3(access_config: &AccessConfig) -> Result<(), Box Result<(), Box Result<(), Box) -> tokio::task::JoinHandle<()> { + tokio::spawn(async move { + let mut interval = tokio::time::interval(Duration::from_secs(60 * 60)); + loop { + interval.tick().await; + match operations.expire_multipart_uploads().await { + Ok(0) => {} + Ok(count) => tracing::debug!(count, "expired S3 multipart sessions"), + Err(error) => tracing::warn!(?error, "S3 multipart expiry sweep deferred"), + } + } + }) +} + #[cfg(feature = "s3")] async fn authenticate_s3( access: &AccessConfig, diff --git a/app/crowdb-access-server/src/s3/operations/multipart.rs b/app/crowdb-access-server/src/s3/operations/multipart.rs index 06975c1a..f1e72ed2 100644 --- a/app/crowdb-access-server/src/s3/operations/multipart.rs +++ b/app/crowdb-access-server/src/s3/operations/multipart.rs @@ -3,13 +3,14 @@ //! Authenticated S3 multipart operations over durable session and part records. +use crowdb_access_s3::bucket; use crowdb_access_s3::error::S3ErrorCode; use crowdb_access_s3::integrity::{IntegrityError, SinglePartIntegrity}; use crowdb_access_s3::metadata::{ new_upload_id, CompletionPart, MultipartPartRecord, MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, }; -use crowdb_access_s3::route::S3Route; +use crowdb_access_s3::route::{S3Operation, S3Route}; use crowdb_access_s3::streaming::{ write_body_with_checksums_buffered, write_native_body_with_checksums_metered, }; @@ -35,6 +36,38 @@ const UPLOAD_LIFETIME_MS: u64 = 7 * 24 * 60 * 60 * 1000; const MAX_COMPLETE_BODY: usize = 2 * 1024 * 1024; impl ProductionS3Operations { + /// Walks bounded metadata pages and marks expired sessions terminal. + /// + /// # Errors + /// Defers the sweep when bucket or upload metadata is unavailable. + pub async fn expire_multipart_uploads(&self) -> Result { + let buckets = bucket::list_buckets(&self.storage.metadata, &self.config.tenant, 1_001) + .await + .map_err(|_| S3ErrorCode::ServiceUnavailable)?; + if buckets.len() > 1_000 { + return Err(S3ErrorCode::SlowDown); + } + let now = unix_millis(); + let mut expired = 0; + for bucket in buckets { + let mut cursor = None; + loop { + let page = self + .multipart() + .expire_page(bucket.bucket_id, cursor.as_deref(), now, 1_000) + .await + .map_err(|error| map_multipart_error(&error))?; + expired += page.expired; + let Some(next) = page.next else { + break; + }; + cursor = Some(next); + tokio::task::yield_now().await; + } + } + Ok(expired) + } + fn multipart(&self) -> MultipartRepository { MultipartRepository::new(self.storage.metadata.clone(), self.config.tenant.clone()) } @@ -43,11 +76,24 @@ impl ProductionS3Operations { let bucket = self.resolve_bucket(required_bucket(route)?).await?; let key = required_key(route)?; let upload_id = route.upload_id.ok_or(S3ErrorCode::InvalidRequest)?; - self.multipart() + let session = self + .multipart() .load_identity(bucket, key, &upload_id) .await .map_err(|error| map_multipart_error(&error))? - .ok_or(S3ErrorCode::NoSuchUpload) + .ok_or(S3ErrorCode::NoSuchUpload)?; + if session.phase == MultipartPhase::Open && session.expires_ms <= unix_millis() { + let expired = self + .multipart() + .abort(&session) + .await + .map_err(|error| map_multipart_error(&error))?; + if route.operation != S3Operation::AbortMultipartUpload { + return Err(S3ErrorCode::NoSuchUpload); + } + return Ok(expired); + } + Ok(session) } pub(super) async fn create_multipart_upload( diff --git a/app/crowdb-access-server/tests/s3_e2e/lost_reply.py b/app/crowdb-access-server/tests/s3_e2e/lost_reply.py index 1b5dd794..e49684ff 100644 --- a/app/crowdb-access-server/tests/s3_e2e/lost_reply.py +++ b/app/crowdb-access-server/tests/s3_e2e/lost_reply.py @@ -1,7 +1,7 @@ # Copyright 2026-present Gian # Licensed under the Apache License, Version 2.0. -"""Drop a completed PUT response at a loopback proxy, then retry the PUT.""" +"""Drop committed PUT and multipart responses at a loopback proxy, then retry.""" import os import socket @@ -17,7 +17,7 @@ from botocore.credentials import Credentials -def swallow_put_reply(listener, endpoint): +def swallow_reply(listener, endpoint, expected_status): parsed = urlsplit(endpoint) with listener: client, _ = listener.accept() @@ -30,11 +30,11 @@ def swallow_put_reply(listener, endpoint): assert received, "client closed before signed PUT headers" request.extend(received) headers, body = bytes(request).split(b"\r\n\r\n", 1) - content_length = next( + content_length = next(( int(line.split(b":", 1)[1].strip()) for line in headers.split(b"\r\n") if line.lower().startswith(b"content-length:") - ) + ), 0) backend.sendall(headers + b"\r\n\r\n" + body) remaining = content_length - len(body) while remaining: @@ -45,32 +45,15 @@ def swallow_put_reply(listener, endpoint): response = bytearray() while chunk := backend.recv(8192): response.extend(chunk) - assert response.startswith(b"HTTP/1.1 200 "), response[:256] + assert response.startswith(f"HTTP/1.1 {expected_status} ".encode()), response[:256] return bytes(response) -def main(): - endpoint = os.environ["CROWDB_S3_E2E_ENDPOINT"] +def drop_reply(endpoint, credentials, method, path, payload, expected_status): parsed = urlsplit(endpoint) - credentials = Credentials( - os.environ["CROWDB_S3_E2E_ACCESS_KEY"], os.environ["CROWDB_S3_E2E_SECRET_KEY"] - ) - client = boto3.client( - "s3", - endpoint_url=endpoint, - region_name=os.environ.get("CROWDB_S3_E2E_REGION", "us-east-1"), - aws_access_key_id=credentials.access_key, - aws_secret_access_key=credentials.secret_key, - config=Config(s3={"addressing_style": "path"}), - ) - bucket = "crowdb-e2e-lost-reply" - key = "retry/same-payload.bin" - payload = bytes(range(256)) * 257 - etag = f'"{md5(payload).hexdigest()}"' - client.create_bucket(Bucket=bucket) request = AWSRequest( - method="PUT", - url=f"{endpoint}/{bucket}/{key}", + method=method, + url=f"{endpoint}{path}", data=payload, headers={ "Host": parsed.netloc, @@ -84,26 +67,78 @@ def main(): with socket.socket() as listener, ThreadPoolExecutor(max_workers=1) as workers: listener.bind(("127.0.0.1", 0)) listener.listen(1) - forwarded = workers.submit(swallow_put_reply, listener, endpoint) + forwarded = workers.submit(swallow_reply, listener, endpoint, expected_status) proxy = HTTPConnection("127.0.0.1", listener.getsockname()[1], timeout=15) try: - proxy.request("PUT", f"/{bucket}/{key}", body=payload, headers=dict(request.headers.items())) + proxy.request(method, path, body=payload, headers=dict(request.headers.items())) try: proxy.getresponse() except RemoteDisconnected: pass else: - raise AssertionError("proxy unexpectedly returned the completed PUT response") + raise AssertionError(f"proxy unexpectedly returned the completed {method} response") finally: proxy.close() - response = forwarded.result(timeout=20) - assert f"\r\netag: {etag}\r\n".lower().encode() in response.lower(), response[:512] + return forwarded.result(timeout=20) + + +def main(): + endpoint = os.environ["CROWDB_S3_E2E_ENDPOINT"] + credentials = Credentials( + os.environ["CROWDB_S3_E2E_ACCESS_KEY"], os.environ["CROWDB_S3_E2E_SECRET_KEY"] + ) + client = boto3.client( + "s3", + endpoint_url=endpoint, + region_name=os.environ.get("CROWDB_S3_E2E_REGION", "us-east-1"), + aws_access_key_id=credentials.access_key, + aws_secret_access_key=credentials.secret_key, + config=Config(s3={"addressing_style": "path"}), + ) + bucket = "crowdb-e2e-lost-reply" + key = "retry/same-payload.bin" + payload = bytes(range(256)) * 257 + etag = f'"{md5(payload).hexdigest()}"' + client.create_bucket(Bucket=bucket) + response = drop_reply(endpoint, credentials, "PUT", f"/{bucket}/{key}", payload, 200) + assert f"\r\netag: {etag}\r\n".lower().encode() in response.lower(), response[:512] assert client.put_object(Bucket=bucket, Key=key, Body=payload)["ETag"] == etag assert client.get_object(Bucket=bucket, Key=key)["Body"].read() == payload listed = client.list_objects_v2(Bucket=bucket, Prefix="retry/") assert [item["Key"] for item in listed["Contents"]] == [key] client.delete_object(Bucket=bucket, Key=key) + + multipart_key = "retry/multipart.bin" + part = b"multipart-response-loss" * 512 + part_etag = f'"{md5(part).hexdigest()}"' + upload_id = client.create_multipart_upload(Bucket=bucket, Key=multipart_key)["UploadId"] + query = f"?partNumber=1&uploadId={upload_id}" + response = drop_reply(endpoint, credentials, "PUT", f"/{bucket}/{multipart_key}{query}", part, 200) + assert f"\r\netag: {part_etag}\r\n".lower().encode() in response.lower(), response[:512] + assert client.upload_part(Bucket=bucket, Key=multipart_key, UploadId=upload_id, + PartNumber=1, Body=part)["ETag"] == part_etag + assert len(client.list_parts(Bucket=bucket, Key=multipart_key, UploadId=upload_id)["Parts"]) == 1 + + complete = ( + f"1{part_etag}" + "" + ).encode() + path = f"/{bucket}/{multipart_key}?uploadId={upload_id}" + drop_reply(endpoint, credentials, "POST", path, complete, 200) + published = client.complete_multipart_upload( + Bucket=bucket, Key=multipart_key, UploadId=upload_id, + MultipartUpload={"Parts": [{"PartNumber": 1, "ETag": part_etag}]}, + ) + assert published["ETag"] == f'"{md5(md5(part).digest()).hexdigest()}-1"' + assert client.get_object(Bucket=bucket, Key=multipart_key)["Body"].read() == part + client.delete_object(Bucket=bucket, Key=multipart_key) + + aborted = client.create_multipart_upload(Bucket=bucket, Key=multipart_key)["UploadId"] + drop_reply(endpoint, credentials, "DELETE", f"/{bucket}/{multipart_key}?uploadId={aborted}", b"", 204) + client.abort_multipart_upload(Bucket=bucket, Key=multipart_key, UploadId=aborted) + assert all(item["UploadId"] != aborted for item in + client.list_multipart_uploads(Bucket=bucket).get("Uploads", [])) client.delete_bucket(Bucket=bucket) diff --git a/app/crowdb-access-server/tests/s3_e2e/restart.py b/app/crowdb-access-server/tests/s3_e2e/restart.py index 9eef09ca..ec5c51dc 100644 --- a/app/crowdb-access-server/tests/s3_e2e/restart.py +++ b/app/crowdb-access-server/tests/s3_e2e/restart.py @@ -31,6 +31,9 @@ def main(): new_payload = b"new-generation" * 8192 new_etag = f'"{md5(new_payload).hexdigest()}"' deleted_key = "persisted/deleted-before-restart.bin" + multipart_key = "persisted/incomplete-before-restart.bin" + multipart_payload = b"durable-part-after-restart" * 512 + multipart_etag = f'"{md5(multipart_payload).hexdigest()}"' if phase == "prepare": client.create_bucket(Bucket=bucket) @@ -39,6 +42,9 @@ def main(): assert client.put_object(Bucket=bucket, Key=overwritten_key, Body=new_payload)["ETag"] == new_etag client.put_object(Bucket=bucket, Key=deleted_key, Body=old_payload) client.delete_object(Bucket=bucket, Key=deleted_key) + upload_id = client.create_multipart_upload(Bucket=bucket, Key=multipart_key)["UploadId"] + assert client.upload_part(Bucket=bucket, Key=multipart_key, UploadId=upload_id, + PartNumber=1, Body=multipart_payload)["ETag"] == multipart_etag elif phase in ( "verify", "verify-after-group0-restart", @@ -56,12 +62,28 @@ def main(): assert client.get_object(Bucket=bucket, Key=overwritten_key)["Body"].read() == new_payload keys = {entry["Key"] for entry in client.list_objects_v2(Bucket=bucket).get("Contents", [])} assert keys == {key, overwritten_key}, keys + uploads = [item for item in client.list_multipart_uploads(Bucket=bucket).get("Uploads", []) + if item["Key"] == multipart_key] + assert len(uploads) == 1, uploads + parts = client.list_parts(Bucket=bucket, Key=multipart_key, + UploadId=uploads[0]["UploadId"])["Parts"] + assert [(part["PartNumber"], part["ETag"]) for part in parts] == [(1, multipart_etag)] break except (BotoCoreError, ClientError): if time.monotonic() >= deadline: raise time.sleep(0.5) elif phase == "cleanup": + uploads = [item for item in client.list_multipart_uploads(Bucket=bucket).get("Uploads", []) + if item["Key"] == multipart_key] + assert len(uploads) == 1, uploads + completed = client.complete_multipart_upload( + Bucket=bucket, Key=multipart_key, UploadId=uploads[0]["UploadId"], + MultipartUpload={"Parts": [{"PartNumber": 1, "ETag": multipart_etag}]}, + ) + assert completed["ETag"] == f'"{md5(md5(multipart_payload).digest()).hexdigest()}-1"' + assert client.get_object(Bucket=bucket, Key=multipart_key)["Body"].read() == multipart_payload + client.delete_object(Bucket=bucket, Key=multipart_key) client.delete_object(Bucket=bucket, Key=key) client.delete_object(Bucket=bucket, Key=overwritten_key) client.delete_bucket(Bucket=bucket) diff --git a/app/crowdb-access-server/tests/s3_full_stack_test.rs b/app/crowdb-access-server/tests/s3_full_stack_test.rs index a67640c7..2df1a4dc 100644 --- a/app/crowdb-access-server/tests/s3_full_stack_test.rs +++ b/app/crowdb-access-server/tests/s3_full_stack_test.rs @@ -153,7 +153,7 @@ impl FullStackSetup { run_boto3_case(method, &context); case.pass(); } - let case = TestCase::start("boto3::lost_put_reply_is_idempotent"); + let case = TestCase::start("boto3::lost_put_and_multipart_replies_are_idempotent"); run_restart_phase("lost-reply", &self.listen, &self.access_key, &self.secret_key); assert_native_write_metrics(&self.listen); case.pass(); diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index eea548fd..0d1006b5 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -62,8 +62,8 @@ Iceberg multipart path. Production UploadPart now uses the basic streaming writer and saves raw MD5 with locations. A byte-identical retry keeps the selected generation even when a new write produced different chunk locations; R95 can reclaim those - unreachable chunks. Full stack ingestion passes 5 MiB and small parts; real - response-loss and restart acceptance remain. + unreachable chunks. Full stack ingestion passes 5 MiB and small parts, a + dropped UploadPart success response, and retries after service restart. - [ ] **Atomic completion**: fence selected part generations, validate order, count, size and checksum, compose locations through the shared core, and publish one immutable object generation without reading part bytes. The S3 @@ -74,16 +74,22 @@ Iceberg multipart path. the full boto3 stack, including a repeated Complete request. - [ ] **Abort and expiry**: make terminal states idempotent and preserve the part generations that R95's chunk-centered scanner needs for reference checks. - The S3 adapter now has an idempotent, response-loss-safe logical abort; the - expiry scan and common-prefix key layout remain. R95 owns physical cleanup. + The S3 adapter now has an idempotent, response-loss-safe logical abort and a + bounded hourly expiry sweep over the upload listing index. Session and part + records remain under one upload prefix after terminal transition. Focused + expiry pagination and evidence tests pass. Requests encountering an expired + open session terminate it before the hourly sweep. R95 owns physical cleanup. ## Acceptance and cleanup - [ ] **Focused and E2E tests**: known MD5 vectors, out-of-order/replaced parts, invalid completion, response loss and restart, abort/expiry metadata, and ordinary single-part compatibility. The complete S3 library and access-server - suites plus 19 full-stack boto3/restart cases pass. Multipart response-loss, - expiry and restart cases remain. + suites plus 19 full-stack boto3/restart cases pass. The new cases drop + UploadPart, Complete and Abort success replies, replay them, preserve an + incomplete session across six service restarts, then complete it. Hourly + expiry has a focused metadata test; broader concurrency and error-matrix + acceptance remains. - [ ] **Gates and docs**: run both access crate suites, access-server E2E, Rust fmt and clippy separately; update S3 design and remove R167 plus this plan only after all acceptance criteria pass. diff --git a/lib/crowdb-access-s3/src/metadata.rs b/lib/crowdb-access-s3/src/metadata.rs index 0f56a4e7..f126a8c1 100644 --- a/lib/crowdb-access-s3/src/metadata.rs +++ b/lib/crowdb-access-s3/src/metadata.rs @@ -27,7 +27,8 @@ pub use multipart::{ new_upload_id, MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord, }; pub use multipart_repository::{ - CompletionPart, MultipartPartPage, MultipartRepository, MultipartRepositoryError, MultipartUploadPage, + CompletionPart, MultipartExpiryPage, MultipartPartPage, MultipartRepository, MultipartRepositoryError, + MultipartUploadPage, }; pub use namespace::{BucketDeleteOutcome, BucketNamespace, BucketNamespaceError}; pub use record::{BucketNameRecord, MetadataRecordError, ObjectRecord}; diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index 27df6dfb..b90549df 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -17,6 +17,7 @@ mod terminal; pub use completion::CompletionPart; pub use listing::{MultipartPartPage, MultipartUploadPage}; +pub use terminal::MultipartExpiryPage; #[derive(Debug, thiserror::Error)] pub enum MultipartRepositoryError { diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs index 8dd33943..52e2cf2c 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs @@ -3,9 +3,94 @@ //! Terminal multipart session transitions. -use super::{MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord}; +use super::{ + MetadataKey, MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, +}; +use crate::metadata::BucketId; + +const MAX_EXPIRY_PAGE: usize = 1_000; +const MAX_EXPIRY_SCAN_BYTES: usize = 4 * 1024 * 1024; + +pub struct MultipartExpiryPage { + pub next: Option>, + pub expired: usize, +} impl MultipartRepository { + /// Marks expired open uploads terminal while retaining all part evidence. + /// + /// Returns a key cursor for the next bounded listing-index page. A worker + /// can resume from that key without an additional per-upload cleanup queue. + /// + /// # Errors + /// Rejects an invalid cursor, corrupt index data or unavailable metadata. + pub async fn expire_page( + &self, + bucket: BucketId, + after: Option<&[u8]>, + now_ms: u64, + max_items: usize, + ) -> Result { + if max_items == 0 || max_items > MAX_EXPIRY_PAGE { + return Err(MultipartRepositoryError::Conflict); + } + let prefix = MetadataKey::multipart_session_prefix(&self.tenant, bucket); + let end = MetadataKey::multipart_session_end(&self.tenant, bucket); + let start = match after { + Some(after) + if after.starts_with(&prefix) && after.len() > prefix.len() && after < end.as_slice() => + { + let mut next = after.to_vec(); + next.push(0); + next + } + Some(_) => return Err(MultipartRepositoryError::Conflict), + None => prefix, + }; + let page = self + .store + .scan_page(start, end, max_items, MAX_EXPIRY_SCAN_BYTES, None) + .await?; + let next = if page.continuation.is_some() { + Some( + page.items + .last() + .ok_or(MultipartRepositoryError::Conflict)? + .key + .clone(), + ) + } else { + None + }; + let mut expired = 0; + for index in page.items { + let Some(session_value) = self.store.get(index.value).await? else { + continue; + }; + let session = MultipartSessionRecord::decode_unbound(&session_value.value)?; + if session.bucket_id != bucket + || MetadataKey::multipart_upload_index( + &self.tenant, + bucket, + &session.object_key, + &session.upload_id, + )? != index.key + || MetadataKey::multipart_session(&self.tenant, bucket, &session.upload_id) + != session_value.key + { + return Err(MultipartRepositoryError::Conflict); + } + if session.phase == MultipartPhase::Open && session.expires_ms <= now_ms { + match self.abort(&session).await { + Ok(_) => expired += 1, + Err(MultipartRepositoryError::Conflict) => {} + Err(error) => return Err(error), + } + } + } + Ok(MultipartExpiryPage { next, expired }) + } + /// Logically aborts an open upload, retaining part evidence for cleanup. /// /// # Errors diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs index 54c629a2..a7785557 100644 --- a/lib/crowdb-access-s3/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -686,3 +686,44 @@ async fn upload_listing_filters_terminal_records_and_resumes_same_key() { .await .is_err()); } + +#[tokio::test] +async fn expiry_pages_mark_old_uploads_terminal_without_removing_part_evidence() { + let (repository, _, store) = repository().await; + let first = session(); + let mut second = first.clone(); + second.upload_id = [8; 16]; + second.object_key = b"later".to_vec(); + second.expires_ms = 300; + repository.begin(&first).await.unwrap(); + repository.begin(&second).await.unwrap(); + repository.put_stream_part(&first, &part(), 110).await.unwrap(); + + let page = repository + .expire_page(first.bucket_id, None, 250, 1) + .await + .unwrap(); + assert_eq!(page.expired, 0); + let next = page.next.expect("another index page remains"); + let page = repository + .expire_page(first.bucket_id, Some(&next), 250, 1) + .await + .unwrap(); + assert_eq!(page.expired, 1); + assert!(page.next.is_none()); + assert_eq!( + repository.load(&first).await.unwrap().unwrap().phase, + MultipartPhase::Aborted + ); + assert_eq!( + repository.load(&second).await.unwrap().unwrap().phase, + MultipartPhase::Open + ); + assert!(repository.part_generation(&first, 1, 1).await.unwrap().is_some()); + + let tenant = TenantId::new(b"tenant".to_vec()).unwrap(); + let index = + MetadataKey::multipart_upload_index(&tenant, first.bucket_id, &first.object_key, &first.upload_id) + .unwrap(); + assert!(store.get(index).await.unwrap().is_some()); +} From b0aa99a24913b3250fc8d888e77e1b2e15b9b6e1 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 09:05:59 +0800 Subject: [PATCH 34/74] Share multipart lifetime and revision rules --- doc/working/plan-s3-multipart.md | 5 ++++- .../src/file/multipart_repository.rs | 5 +++-- .../src/file/multipart_repository/parts.rs | 6 ++---- lib/crowdb-access-multipart/src/lib.rs | 4 ++-- lib/crowdb-access-multipart/src/state.rs | 21 +++++++++++++++++++ .../tests/state_test.rs | 17 +++++++++++++-- lib/crowdb-access-s3/src/integrity.rs | 4 ++-- .../src/metadata/multipart_repository.rs | 10 ++++----- .../multipart_repository/completion.rs | 12 +++++------ .../metadata/multipart_repository/terminal.rs | 6 ++---- lib/crowdb-access-s3/src/route/multipart.rs | 2 +- 11 files changed, 61 insertions(+), 31 deletions(-) diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 0d1006b5..97e92f66 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -26,7 +26,10 @@ Iceberg multipart path. selection validation, accounting and location composition are now used by both adapters; storage CAS and durable record layouts remain protocol-specific. Files: shared - multipart crate, Iceberg file repository, S3 metadata store. + multipart crate, Iceberg file repository, S3 metadata store. Both adapters + now also use the same inclusive/exclusive lifetime decision and checked + session/part revision advancement. The remaining Iceberg completion/recovery + sequencing still needs protocol-neutral extraction. ## S3 adapter and HTTP diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository.rs index fd4a6157..ff9c08eb 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository.rs @@ -5,6 +5,7 @@ use crate::error::ValidationError; use crate::key::{CatalogScope, IcebergKey, OperationId}; use crate::operation::mutation_identity; use crate::record::StorageRecord; +use crowdb_access_multipart::{live_at, next_revision}; use super::{MultipartPhase, MultipartSession}; @@ -131,7 +132,7 @@ impl MultipartRepository { } fn check_live(session: &MultipartSession, now_ms: u64) -> Result<(), CatalogError> { - if now_ms < session.created_ms || now_ms >= session.expires_ms { + if !live_at(session.created_ms, session.expires_ms, now_ms) { return Err(CatalogError::Conflict); } Ok(()) @@ -139,7 +140,7 @@ fn check_live(session: &MultipartSession, now_ms: u64) -> Result<(), CatalogErro fn increment(session: &MultipartSession) -> Result { let mut next = session.clone(); - next.revision = session.revision.checked_add(1).ok_or(ValidationError::Record)?; + next.revision = next_revision(session.revision).ok_or(ValidationError::Record)?; Ok(next) } diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs index 21d4bde1..199d8b73 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs @@ -4,7 +4,7 @@ use crate::file::{MultipartPart, MultipartPartMutation, MultipartPhase, Multipar use crate::key::{CatalogScope, IcebergKey}; use crate::operation::mutation_identity; use crate::record::StorageRecord; -use crowdb_access_multipart::{reserve_part_accounting, PartAccounting}; +use crowdb_access_multipart::{next_part_revision, reserve_part_accounting, PartAccounting}; use super::{check_live, increment, MultipartRepository}; @@ -42,9 +42,7 @@ impl MultipartRepository { before.validate_for(¤t)?; } let mut after = part.clone(); - after.revision = before - .as_ref() - .map_or(Some(1), |before| before.revision.checked_add(1)) + after.revision = next_part_revision(before.as_ref().map(|before| before.revision)) .ok_or(ValidationError::Record)?; after.modified_ms = now_ms; after.validate_for(¤t)?; diff --git a/lib/crowdb-access-multipart/src/lib.rs b/lib/crowdb-access-multipart/src/lib.rs index 87ec7f53..e10c4e2e 100644 --- a/lib/crowdb-access-multipart/src/lib.rs +++ b/lib/crowdb-access-multipart/src/lib.rs @@ -15,8 +15,8 @@ use md5::{Digest, Md5}; mod state; pub use state::{ - reserve_part_accounting, validate_selected_parts, MultipartBounds, MultipartPhase, PartAccounting, - SelectedPart, StateError, + live_at, next_part_revision, next_revision, reserve_part_accounting, validate_selected_parts, + MultipartBounds, MultipartPhase, PartAccounting, SelectedPart, StateError, }; #[derive(Debug, thiserror::Error, PartialEq, Eq)] diff --git a/lib/crowdb-access-multipart/src/state.rs b/lib/crowdb-access-multipart/src/state.rs index 5352bc50..81b68bdf 100644 --- a/lib/crowdb-access-multipart/src/state.rs +++ b/lib/crowdb-access-multipart/src/state.rs @@ -63,6 +63,27 @@ pub enum StateError { InvalidAccounting, } +/// Admission time is inclusive at creation and exclusive at expiry. +#[must_use] +pub const fn live_at(created_ms: u64, expires_ms: u64, now_ms: u64) -> bool { + created_ms <= now_ms && now_ms < expires_ms +} + +/// Returns the next durable session revision without wrapping. +#[must_use] +pub const fn next_revision(current: u64) -> Option { + current.checked_add(1) +} + +/// Returns the first or replacement part revision without wrapping. +#[must_use] +pub const fn next_part_revision(previous: Option) -> Option { + match previous { + Some(current) => next_revision(current), + None => Some(1), + } +} + /// Validates one ordered completion selection independent of wire format. /// /// # Errors diff --git a/lib/crowdb-access-multipart/tests/state_test.rs b/lib/crowdb-access-multipart/tests/state_test.rs index 66eeeb21..e122fcfc 100644 --- a/lib/crowdb-access-multipart/tests/state_test.rs +++ b/lib/crowdb-access-multipart/tests/state_test.rs @@ -2,10 +2,23 @@ // Licensed under the Apache License, Version 2.0. use crowdb_access_multipart::{ - reserve_part_accounting, validate_selected_parts, MultipartBounds, PartAccounting, SelectedPart, - StateError, + live_at, next_part_revision, next_revision, reserve_part_accounting, validate_selected_parts, + MultipartBounds, PartAccounting, SelectedPart, StateError, }; +#[test] +fn both_adapters_share_lifetime_and_revision_edges() { + assert!(!live_at(100, 200, 99)); + assert!(live_at(100, 200, 100)); + assert!(live_at(100, 200, 199)); + assert!(!live_at(100, 200, 200)); + assert_eq!(next_revision(1), Some(2)); + assert_eq!(next_revision(u64::MAX), None); + assert_eq!(next_part_revision(None), Some(1)); + assert_eq!(next_part_revision(Some(1)), Some(2)); + assert_eq!(next_part_revision(Some(u64::MAX)), None); +} + fn selected(number: u16, revision: u64) -> SelectedPart { SelectedPart { number, diff --git a/lib/crowdb-access-s3/src/integrity.rs b/lib/crowdb-access-s3/src/integrity.rs index 550369f1..ecaf782f 100644 --- a/lib/crowdb-access-s3/src/integrity.rs +++ b/lib/crowdb-access-s3/src/integrity.rs @@ -24,8 +24,8 @@ pub enum IntegrityError { /// /// The `ETag` is lowercase hexadecimal MD5 of the logical object bytes. It is /// intentionally calculated before metadata publication, never from physical -/// chunks, so frame and EC boundaries cannot change it. Multipart has its own -/// future contract. +/// chunks, so frame and EC boundaries cannot change it. Multipart uses the +/// selected parts' raw MD5 digests to calculate a composite `ETag`. pub struct SinglePartIntegrity { md5: md5::Context, sha256: Option, diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index b90549df..4cc3a5e3 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -3,6 +3,7 @@ //! CAS-backed multipart authority in the S3 Chunk-KV namespace. +use crowdb_access_multipart::{live_at, next_part_revision, next_revision}; use std::sync::Arc; use super::{ @@ -150,7 +151,7 @@ impl MultipartRepository { if previous.bucket_id != next.bucket_id || previous.object_key != next.object_key || previous.upload_id != next.upload_id - || previous.revision.checked_add(1) != Some(next.revision) + || next_revision(previous.revision) != Some(next.revision) { return Err(MultipartRepositoryError::Conflict); } @@ -241,8 +242,7 @@ impl MultipartRepository { .await? .ok_or(MultipartRepositoryError::Conflict)?; if current.phase != MultipartPhase::Open - || now_ms < current.created_ms - || now_ms >= current.expires_ms + || !live_at(current.created_ms, current.expires_ms, now_ms) || u64::from(current.max_parts) .checked_mul(current.max_part_bytes) .map_or(true, |bytes| bytes > current.max_staged_bytes) @@ -260,9 +260,7 @@ impl MultipartRepository { } } let mut after = part.clone(); - after.revision = before - .as_ref() - .map_or(Some(1), |before| before.revision.checked_add(1)) + after.revision = next_part_revision(before.as_ref().map(|before| before.revision)) .ok_or(MultipartRepositoryError::Conflict)?; after.modified_ms = now_ms; let value = after.encode()?; diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs index 293f1d76..3c392218 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs @@ -3,7 +3,9 @@ //! Freeze the exact selected part generations before object publication. -use crowdb_access_multipart::{validate_selected_parts, MultipartComposer, SelectedPart}; +use crowdb_access_multipart::{ + live_at, next_revision, validate_selected_parts, MultipartComposer, SelectedPart, +}; use sha2::{Digest as _, Sha256}; use super::{MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord}; @@ -46,8 +48,7 @@ impl MultipartRepository { } if current != *session || current.phase != MultipartPhase::Open - || now_ms < current.created_ms - || now_ms >= current.expires_ms + || !live_at(current.created_ms, current.expires_ms, now_ms) || requested.is_empty() || requested.len() > usize::from(current.max_parts) { @@ -86,10 +87,7 @@ impl MultipartRepository { .finish() .map_err(|_| MultipartRepositoryError::InvalidPart)?; let mut next = current.clone(); - next.revision = current - .revision - .checked_add(1) - .ok_or(MultipartRepositoryError::Conflict)?; + next.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; next.phase = MultipartPhase::Publishing; next.part_count = u16::try_from(selection.len()).map_err(|_| MultipartRepositoryError::InvalidPart)?; diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs index 52e2cf2c..38e5f98f 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs @@ -7,6 +7,7 @@ use super::{ MetadataKey, MultipartPhase, MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, }; use crate::metadata::BucketId; +use crowdb_access_multipart::next_revision; const MAX_EXPIRY_PAGE: usize = 1_000; const MAX_EXPIRY_SCAN_BYTES: usize = 4 * 1024 * 1024; @@ -110,10 +111,7 @@ impl MultipartRepository { return Err(MultipartRepositoryError::Conflict); } let mut next = current.clone(); - next.revision = current - .revision - .checked_add(1) - .ok_or(MultipartRepositoryError::Conflict)?; + next.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; next.phase = MultipartPhase::Aborted; if self.exchange(¤t, &next).await? { Ok(next) diff --git a/lib/crowdb-access-s3/src/route/multipart.rs b/lib/crowdb-access-s3/src/route/multipart.rs index 3968b4c9..e6df6d66 100644 --- a/lib/crowdb-access-s3/src/route/multipart.rs +++ b/lib/crowdb-access-s3/src/route/multipart.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Parse the S3 multipart query surface without enabling HTTP dispatch yet. +//! Parse the authenticated S3 multipart query surface. use hyper::{Method, Uri}; From 3a6bcba93763a799fd70eefea6a1b7392f045c5e Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 09:08:42 +0800 Subject: [PATCH 35/74] Report invalid multipart part order --- .../src/multipart_complete.rs | 66 +++++++++++-------- .../src/s3/operations/multipart.rs | 7 +- .../tests/s3_multipart_xml_test.rs | 23 ++++++- doc/working/plan-s3-multipart.md | 2 + 4 files changed, 67 insertions(+), 31 deletions(-) diff --git a/app/crowdb-access-server/src/multipart_complete.rs b/app/crowdb-access-server/src/multipart_complete.rs index 46f89f84..79025f25 100644 --- a/app/crowdb-access-server/src/multipart_complete.rs +++ b/app/crowdb-access-server/src/multipart_complete.rs @@ -22,8 +22,12 @@ pub struct CompleteSelection { } #[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] -#[error("invalid multipart completion XML")] -pub struct CompleteRequestError; +pub enum CompleteRequestError { + #[error("invalid multipart completion XML")] + InvalidRequest, + #[error("multipart parts are not in ascending order")] + InvalidPartOrder, +} impl CompleteSelection { /// Parses a bounded S3 `CompleteMultipartUpload` body. The caller must match @@ -32,7 +36,7 @@ impl CompleteSelection { /// Rejects malformed XML, extra fields and unordered or duplicate parts. pub fn parse(bytes: &[u8]) -> Result { if bytes.is_empty() || bytes.len() > MAX_COMPLETE_XML_BYTES { - return Err(CompleteRequestError); + return Err(CompleteRequestError::InvalidRequest); } let mut reader = Reader::from_reader(bytes); let mut state = State::Start; @@ -41,7 +45,10 @@ impl CompleteSelection { let mut digest = None; let mut etag = Vec::new(); loop { - match reader.read_event().map_err(|_| CompleteRequestError)? { + match reader + .read_event() + .map_err(|_| CompleteRequestError::InvalidRequest)? + { Event::Decl(_) if state == State::Start => {} Event::Start(event) if valid_attributes(state, &event)? => { state = match (state, event.name().as_ref()) { @@ -49,34 +56,36 @@ impl CompleteSelection { (State::Root, b"Part") if parts.len() < MAX_COMPLETE_PARTS => State::Part, (State::Part, b"PartNumber") if number.is_none() => State::Number, (State::Part, b"ETag") if digest.is_none() => State::Etag, - _ => return Err(CompleteRequestError), + _ => return Err(CompleteRequestError::InvalidRequest), }; } Event::Text(event) => match state { State::Number if number.is_none() => { let value: &[u8] = event.as_ref(); if value.is_empty() || !value.iter().all(u8::is_ascii_digit) { - return Err(CompleteRequestError); + return Err(CompleteRequestError::InvalidRequest); } number = Some( std::str::from_utf8(value) - .map_err(|_| CompleteRequestError)? + .map_err(|_| CompleteRequestError::InvalidRequest)? .parse::() - .map_err(|_| CompleteRequestError)?, + .map_err(|_| CompleteRequestError::InvalidRequest)?, ); } State::Etag => append_etag(&mut etag, &event)?, State::Start | State::Root | State::Part | State::Done if event.iter().all(u8::is_ascii_whitespace) => {} - _ => return Err(CompleteRequestError), + _ => return Err(CompleteRequestError::InvalidRequest), }, Event::GeneralRef(event) if state == State::Etag => { if event.len() > 16 { - return Err(CompleteRequestError); + return Err(CompleteRequestError::InvalidRequest); } - let name = std::str::from_utf8(&event).map_err(|_| CompleteRequestError)?; + let name = + std::str::from_utf8(&event).map_err(|_| CompleteRequestError::InvalidRequest)?; let encoded = format!("&{name};"); - let decoded = quick_xml::escape::unescape(&encoded).map_err(|_| CompleteRequestError)?; + let decoded = quick_xml::escape::unescape(&encoded) + .map_err(|_| CompleteRequestError::InvalidRequest)?; append_etag(&mut etag, decoded.as_bytes())?; } Event::End(event) => { @@ -88,25 +97,26 @@ impl CompleteSelection { State::Part } (State::Part, b"Part") => { - let number = number.take().ok_or(CompleteRequestError)?; - let etag = digest.take().ok_or(CompleteRequestError)?; - if number == 0 - || number > 10_000 - || parts - .last() - .is_some_and(|part: &CompletePart| part.number >= number) + let number = number.take().ok_or(CompleteRequestError::InvalidRequest)?; + let etag = digest.take().ok_or(CompleteRequestError::InvalidRequest)?; + if number == 0 || number > 10_000 { + return Err(CompleteRequestError::InvalidRequest); + } + if parts + .last() + .is_some_and(|part: &CompletePart| part.number >= number) { - return Err(CompleteRequestError); + return Err(CompleteRequestError::InvalidPartOrder); } parts.push(CompletePart { number, etag }); State::Root } (State::Root, b"CompleteMultipartUpload") if !parts.is_empty() => State::Done, - _ => return Err(CompleteRequestError), + _ => return Err(CompleteRequestError::InvalidRequest), }; } Event::Eof if state == State::Done => return Ok(Self { parts }), - _ => return Err(CompleteRequestError), + _ => return Err(CompleteRequestError::InvalidRequest), } } } @@ -131,19 +141,19 @@ fn parse_etag(bytes: &[u8]) -> Result { let hex = bytes .strip_prefix(b"\"") .and_then(|bytes| bytes.strip_suffix(b"\"")) - .ok_or(CompleteRequestError)?; + .ok_or(CompleteRequestError::InvalidRequest)?; if hex.len() != 32 && hex.len() != 64 { - return Err(CompleteRequestError); + return Err(CompleteRequestError::InvalidRequest); } for pair in hex.chunks_exact(2) { let _ = (hex_digit(pair[0])? << 4) | hex_digit(pair[1])?; } - String::from_utf8(hex.to_vec()).map_err(|_| CompleteRequestError) + String::from_utf8(hex.to_vec()).map_err(|_| CompleteRequestError::InvalidRequest) } fn append_etag(etag: &mut Vec, bytes: &[u8]) -> Result<(), CompleteRequestError> { if etag.len().saturating_add(bytes.len()) > 66 { - return Err(CompleteRequestError); + return Err(CompleteRequestError::InvalidRequest); } etag.extend_from_slice(bytes); Ok(()) @@ -153,7 +163,7 @@ fn hex_digit(byte: u8) -> Result { match byte { b'0'..=b'9' => Ok(byte - b'0'), b'a'..=b'f' => Ok(byte - b'a' + 10), - _ => Err(CompleteRequestError), + _ => Err(CompleteRequestError::InvalidRequest), } } @@ -165,7 +175,7 @@ fn valid_attributes( let Some(attribute) = attributes.next() else { return Ok(true); }; - let attribute = attribute.map_err(|_| CompleteRequestError)?; + let attribute = attribute.map_err(|_| CompleteRequestError::InvalidRequest)?; Ok(state == State::Start && event.name().as_ref() == b"CompleteMultipartUpload" && attribute.key.as_ref() == b"xmlns" diff --git a/app/crowdb-access-server/src/s3/operations/multipart.rs b/app/crowdb-access-server/src/s3/operations/multipart.rs index f1e72ed2..2f2240de 100644 --- a/app/crowdb-access-server/src/s3/operations/multipart.rs +++ b/app/crowdb-access-server/src/s3/operations/multipart.rs @@ -26,7 +26,7 @@ use super::{ content_length, full_body, install_body_receive_provider, map_put_outcome, required_bucket, required_key, response, strict_header, unix_millis, xml_response, ProductionS3Operations, Query, ResponseBody, }; -use crate::multipart_complete::CompleteSelection; +use crate::multipart_complete::{CompleteRequestError, CompleteSelection}; const MAX_PARTS: u16 = 10_000; const MAX_PART_BYTES: u64 = 5 * 1024 * 1024 * 1024; @@ -329,7 +329,10 @@ impl ProductionS3Operations { integrity .finish_validated_checksums(content_md5.as_deref(), payload_sha256.as_deref()) .map_err(map_integrity_error)?; - let selection = CompleteSelection::parse(&bytes).map_err(|_| S3ErrorCode::InvalidRequest)?; + let selection = CompleteSelection::parse(&bytes).map_err(|error| match error { + CompleteRequestError::InvalidRequest => S3ErrorCode::InvalidRequest, + CompleteRequestError::InvalidPartOrder => S3ErrorCode::InvalidPartOrder, + })?; let requested: Vec = selection .parts() .iter() diff --git a/app/crowdb-access-server/tests/s3_multipart_xml_test.rs b/app/crowdb-access-server/tests/s3_multipart_xml_test.rs index 0299186d..83c0152c 100644 --- a/app/crowdb-access-server/tests/s3_multipart_xml_test.rs +++ b/app/crowdb-access-server/tests/s3_multipart_xml_test.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crowdb_access_server::s3::CompleteSelection; +use crowdb_access_server::s3::{CompleteRequestError, CompleteSelection}; #[test] fn s3_and_iceberg_use_the_same_bounded_completion_parser() { @@ -14,3 +14,24 @@ fn s3_and_iceberg_use_the_same_bounded_completion_parser() { assert_eq!(s3.parts(), iceberg.parts()); assert_eq!(s3.parts()[0].etag, digest); } + +#[test] +fn completion_parser_distinguishes_part_order_from_malformed_requests() { + let digest = "ab".repeat(16); + let part = |number| format!("{number}\"{digest}\""); + for numbers in [[2, 1], [1, 1]] { + let body = format!( + "{}{}", + part(numbers[0]), + part(numbers[1]) + ); + assert_eq!( + CompleteSelection::parse(body.as_bytes()), + Err(CompleteRequestError::InvalidPartOrder) + ); + } + assert_eq!( + CompleteSelection::parse(b""), + Err(CompleteRequestError::InvalidRequest) + ); +} diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 97e92f66..45747da2 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -58,6 +58,8 @@ Iceberg multipart path. response builders have focused tests. The bounded completion XML parser now has one implementation in access-server and is exposed by both the Iceberg and S3 protocol modules. The six S3 HTTP operations are wired. + Duplicate or descending completion parts now map to S3 `InvalidPartOrder`; + malformed XML remains `InvalidRequest`. Full boto3 stack acceptance passes Create, UploadPart, ListParts, Complete, Abort and ListUploads, including replay, replacement and invalid ETag cases. - [ ] **Part ingestion**: reuse the bounded streaming writer and admission From 30c04086c9bab0ee57da36039d3d5627a3cbe541 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 09:23:51 +0800 Subject: [PATCH 36/74] Store hardware connection identity in Group 0 --- .../tests/common/iceberg_stack.rs | 2 + .../tests/s3_full_stack_test.rs | 2 + app/crowdb-chunkdb/tests/common/cluster.rs | 2 + .../src/commands/cluster/hardware.rs | 1 + app/crowdb-cli/tests/common/console.rs | 1 + app/crowdb-cli/tests/common/direct.rs | 1 + app/crowdb-diskdb/tests/diskdb_e2e_test.rs | 2 + app/crowdb-diskdb/tests/recovery_test.rs | 2 + .../tests/relocation_journal_test.rs | 2 + app/crowdb-diskdb/tests/scanner_test.rs | 2 + app/crowdb-web/src/lifecycle.rs | 9 +++++ app/crowdb-web/src/managed.rs | 4 +- app/crowdb-web/src/physical/view.rs | 2 + .../tests/bare_metal_authority_test.rs | 11 ++++++ .../tests/diskdb_auto_start_test.rs | 1 + app/crowdb-web/tests/diskdb_routes_test.rs | 1 + app/crowdb-web/tests/kv_routes_test.rs | 18 ++++----- app/crowdb-web/tests/managed_mode_test.rs | 1 + app/crowdb-web/tests/metrics_proxy_test.rs | 2 + app/crowdb-web/tests/mgmt_routes_test.rs | 2 + app/crowdb-web/tests/ops_migration_test.rs | 2 + app/crowdb-web/tests/recursive_query_test.rs | 1 + .../tests/replica_leader_removal_test.rs | 2 + app/crowdb-web/tests/replica_routes_test.rs | 2 + app/crowdb-web/tests/rolling_upgrade_test.rs | 3 ++ .../crowdb-monitor/src/bootstrap/hardware.rs | 2 + .../tests/hardware_bootstrap_test.rs | 1 + doc/working/plan-console-authority.md | 6 ++- .../tests/production_restart_e2e.rs | 2 + .../src/cluster_deployer.rs | 1 + lib/crowdb-console-shared/src/config.rs | 22 +++++++++++ .../src/launch/remote.rs | 1 + lib/crowdb-console-shared/src/ops/cluster.rs | 14 +++++++ .../src/ops/cluster/bootstrap/publication.rs | 5 +++ lib/crowdb-console-shared/src/ops/hardware.rs | 6 +++ lib/crowdb-console-shared/src/ssh.rs | 1 + .../tests/kv_e2e_test.rs | 1 + .../tests/lifecycle_e2e_test.rs | 1 + .../tests/mgmt_e2e_test.rs | 1 + .../tests/ops_bootstrap_publication_test.rs | 39 +++++++++++++++++++ .../tests/ops_hardware_test.rs | 4 ++ .../tests/ops_kv_server_test.rs | 1 + lib/crowdb-protocol/src/types/common.rs | 14 +++++++ .../tests/hardware_identity_test.rs | 27 +++++++++++++ lib/crowdb-test-harness/src/hardware.rs | 2 + 45 files changed, 217 insertions(+), 12 deletions(-) create mode 100644 lib/crowdb-protocol/tests/hardware_identity_test.rs diff --git a/app/crowdb-access-server/tests/common/iceberg_stack.rs b/app/crowdb-access-server/tests/common/iceberg_stack.rs index 3671dc9c..e589c041 100644 --- a/app/crowdb-access-server/tests/common/iceberg_stack.rs +++ b/app/crowdb-access-server/tests/common/iceberg_stack.rs @@ -166,6 +166,7 @@ async fn seed(cluster: &KvCluster) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![10], + ..Default::default() }, ) .await @@ -180,6 +181,7 @@ async fn seed(cluster: &KvCluster) { disk_group_ids: vec![100], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await diff --git a/app/crowdb-access-server/tests/s3_full_stack_test.rs b/app/crowdb-access-server/tests/s3_full_stack_test.rs index 2df1a4dc..f2e3900f 100644 --- a/app/crowdb-access-server/tests/s3_full_stack_test.rs +++ b/app/crowdb-access-server/tests/s3_full_stack_test.rs @@ -760,6 +760,7 @@ async fn seed_compact_hardware(hardware: &HardwareClient) -> Vec Vec ExitCode { ssh_user, ssh_key, ssh_password: None, + ssh_credential_ref: None, }; let ctx = match op_context(cli) { Ok(c) => c, diff --git a/app/crowdb-cli/tests/common/console.rs b/app/crowdb-cli/tests/common/console.rs index 58d1790a..e8a1265c 100644 --- a/app/crowdb-cli/tests/common/console.rs +++ b/app/crowdb-cli/tests/common/console.rs @@ -136,6 +136,7 @@ pub fn local_node(id: u64, rack: u64) -> NodeEntry { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, } } diff --git a/app/crowdb-cli/tests/common/direct.rs b/app/crowdb-cli/tests/common/direct.rs index 3b54c461..bb7e41d1 100644 --- a/app/crowdb-cli/tests/common/direct.rs +++ b/app/crowdb-cli/tests/common/direct.rs @@ -85,6 +85,7 @@ pub fn local_node(id: u64, rack: u64) -> NodeEntry { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, } } diff --git a/app/crowdb-diskdb/tests/diskdb_e2e_test.rs b/app/crowdb-diskdb/tests/diskdb_e2e_test.rs index 99f3deb5..6dea06f5 100644 --- a/app/crowdb-diskdb/tests/diskdb_e2e_test.rs +++ b/app/crowdb-diskdb/tests/diskdb_e2e_test.rs @@ -76,6 +76,7 @@ async fn seed_hardware(hw: &HardwareClient) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![NODE_ID], + ..Default::default() }, ) .await @@ -91,6 +92,7 @@ async fn seed_hardware(hw: &HardwareClient) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await diff --git a/app/crowdb-diskdb/tests/recovery_test.rs b/app/crowdb-diskdb/tests/recovery_test.rs index 2f0b0141..b1298752 100644 --- a/app/crowdb-diskdb/tests/recovery_test.rs +++ b/app/crowdb-diskdb/tests/recovery_test.rs @@ -57,6 +57,7 @@ async fn seed_hardware(hw: &HardwareClient) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![NODE_ID], + ..Default::default() }, ) .await @@ -70,6 +71,7 @@ async fn seed_hardware(hw: &HardwareClient) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await diff --git a/app/crowdb-diskdb/tests/relocation_journal_test.rs b/app/crowdb-diskdb/tests/relocation_journal_test.rs index 48ea2f75..f414d9e3 100644 --- a/app/crowdb-diskdb/tests/relocation_journal_test.rs +++ b/app/crowdb-diskdb/tests/relocation_journal_test.rs @@ -105,6 +105,7 @@ async fn seed_hardware(hw: &HardwareClient) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![10], + ..Default::default() }, ) .await @@ -118,6 +119,7 @@ async fn seed_hardware(hw: &HardwareClient) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await diff --git a/app/crowdb-diskdb/tests/scanner_test.rs b/app/crowdb-diskdb/tests/scanner_test.rs index 6d4e1e72..07ac061e 100644 --- a/app/crowdb-diskdb/tests/scanner_test.rs +++ b/app/crowdb-diskdb/tests/scanner_test.rs @@ -67,6 +67,7 @@ async fn seed_hardware(hw: &HardwareClient) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![NODE_ID], + ..Default::default() }, ) .await @@ -80,6 +81,7 @@ async fn seed_hardware(hw: &HardwareClient) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await diff --git a/app/crowdb-web/src/lifecycle.rs b/app/crowdb-web/src/lifecycle.rs index 90c73668..6180f635 100644 --- a/app/crowdb-web/src/lifecycle.rs +++ b/app/crowdb-web/src/lifecycle.rs @@ -113,6 +113,7 @@ pub async fn http_add_rack( let value = crowdb_protocol::common::RackValue { status: crowdb_protocol::common::HwStatus::Up as i32, node_ids: Vec::new(), + name: entry.name.clone(), }; let _ = ctx.sysmd().add_rack(body.id, &value).await; } @@ -193,6 +194,10 @@ pub async fn http_add_node( disk_group_ids: Vec::new(), status_changed_at_ms: 0, temp_failure_since_ms: None, + management_host: entry.host.clone(), + ssh_port: entry.ssh_port, + ssh_user: entry.ssh_user.clone(), + ssh_credential_ref: entry.ssh_credential_ref.clone(), }; let _ = ctx.sysmd().add_node(entry.rack_id, entry.id, &value).await; } @@ -468,6 +473,10 @@ pub async fn http_add_rack_node( disk_group_ids: Vec::new(), status_changed_at_ms: 0, temp_failure_since_ms: None, + management_host: entry.host.clone(), + ssh_port: entry.ssh_port, + ssh_user: entry.ssh_user.clone(), + ssh_credential_ref: entry.ssh_credential_ref.clone(), }; let _ = ctx.sysmd().add_node(entry.rack_id, entry.id, &value).await; } diff --git a/app/crowdb-web/src/managed.rs b/app/crowdb-web/src/managed.rs index c13f5f04..2f65be28 100644 --- a/app/crowdb-web/src/managed.rs +++ b/app/crowdb-web/src/managed.rs @@ -164,8 +164,8 @@ async fn load_snapshot(state: &AppState) -> Result CrowdbSysmdClient { &RackValue { status: HwStatus::Up as i32, node_ids: vec![1], + name: "rack-a".into(), }, ) .await @@ -66,6 +67,10 @@ async fn initialized_authority(cluster: &KvCluster) -> CrowdbSysmdClient { 1, &NodeValue { status: HwStatus::Up as i32, + management_host: "node-a.example".into(), + ssh_port: 2222, + ssh_user: "operator".into(), + ssh_credential_ref: Some("node-a-key".into()), ..Default::default() }, ) @@ -110,6 +115,12 @@ async fn bare_metal_snapshot_requires_live_authority_without_a_docker_monitor() assert_eq!(code, StatusCode::OK, "{view}"); assert_eq!(view["source"], "group0"); assert_eq!(view["nodes"][0]["id"], 1); + assert_eq!(view["racks"][0]["name"], "rack-a"); + assert_eq!(view["nodes"][0]["management_host"], "node-a.example"); + assert_eq!(view["nodes"][0]["ssh_port"], 2222); + assert_eq!(view["nodes"][0]["ssh_user"], "operator"); + assert_eq!(view["nodes"][0]["ssh_credential_ref"], "node-a-key"); + assert_eq!(snapshot(&application(&cluster)).await.1["nodes"], view["nodes"]); assert!(view["monitor"].is_null()); register(&sysmd, 1, 7002, &cluster.mgmt_endpoints[0]).await; diff --git a/app/crowdb-web/tests/diskdb_auto_start_test.rs b/app/crowdb-web/tests/diskdb_auto_start_test.rs index 3d7db0c9..c128066e 100644 --- a/app/crowdb-web/tests/diskdb_auto_start_test.rs +++ b/app/crowdb-web/tests/diskdb_auto_start_test.rs @@ -23,6 +23,7 @@ async fn startup_does_not_replay_local_diskdb_launch_policy_without_group0() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); config diff --git a/app/crowdb-web/tests/diskdb_routes_test.rs b/app/crowdb-web/tests/diskdb_routes_test.rs index da495b8f..7cad78b0 100644 --- a/app/crowdb-web/tests/diskdb_routes_test.rs +++ b/app/crowdb-web/tests/diskdb_routes_test.rs @@ -38,6 +38,7 @@ async fn spawn_web_with_disk_group(rack_id: u64, node_id: u64, dg_id: u64) -> So ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); config diff --git a/app/crowdb-web/tests/kv_routes_test.rs b/app/crowdb-web/tests/kv_routes_test.rs index 3fb2c45a..5a100954 100644 --- a/app/crowdb-web/tests/kv_routes_test.rs +++ b/app/crowdb-web/tests/kv_routes_test.rs @@ -48,6 +48,7 @@ async fn spawn_upstream() -> Option { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "1".to_string(), @@ -83,6 +84,7 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: "n1".to_string(), @@ -226,7 +228,6 @@ async fn kv_put_get_delete_through_web_routes() { async fn kv_get_returns_502_when_leader_unreachable() { use crowdb_console_shared::cluster::{LocalReplicaInfo, NodeGroup, NodeStore, ReplicaRole, ReplicaState}; use std::collections::BTreeMap; - // Pick a free port, drop the listener: nothing accepts on it now. let dead = std::net::TcpListener::bind(("127.0.0.1", 0)).unwrap(); let dead_port = dead.local_addr().unwrap().port(); @@ -250,6 +251,7 @@ async fn kv_get_returns_502_when_leader_unreachable() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); // The node has a configured rpc_url, but the port is dead. cfg.add_server(ServerEntry { @@ -320,14 +322,12 @@ async fn kv_get_returns_502_when_leader_unreachable() { tokio::time::sleep(Duration::from_millis(50)).await; let http = reqwest::Client::new(); - let url = format!("http://{web}/api/stores/7/groups/70/kv/get?key=anything"); - let resp = http.get(&url).send().await.unwrap(); - assert_eq!( - resp.status(), - 502, - "expected 502 when leader crowdb-rpc port is dead, got {}", - resp.status() - ); + let resp = http + .get(format!("http://{web}/api/stores/7/groups/70/kv/get?key=anything")) + .send() + .await + .unwrap(); + assert_eq!(resp.status(), 502); let endpoint = http .get(format!("http://{web}/api/stores/7/groups/70/endpoint")) .send() diff --git a/app/crowdb-web/tests/managed_mode_test.rs b/app/crowdb-web/tests/managed_mode_test.rs index a70b7a68..74d5799d 100644 --- a/app/crowdb-web/tests/managed_mode_test.rs +++ b/app/crowdb-web/tests/managed_mode_test.rs @@ -239,6 +239,7 @@ async fn managed_snapshot_uses_group0_and_monitor_without_local_fallback() { &RackValue { status: HwStatus::Up as i32, node_ids: vec![1], + ..Default::default() }, ) .await diff --git a/app/crowdb-web/tests/metrics_proxy_test.rs b/app/crowdb-web/tests/metrics_proxy_test.rs index caec2564..41eb3b8f 100644 --- a/app/crowdb-web/tests/metrics_proxy_test.rs +++ b/app/crowdb-web/tests/metrics_proxy_test.rs @@ -50,6 +50,7 @@ async fn spawn_upstream() -> Option { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "1".into(), @@ -85,6 +86,7 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: "n1".into(), diff --git a/app/crowdb-web/tests/mgmt_routes_test.rs b/app/crowdb-web/tests/mgmt_routes_test.rs index f0b74a62..69096003 100644 --- a/app/crowdb-web/tests/mgmt_routes_test.rs +++ b/app/crowdb-web/tests/mgmt_routes_test.rs @@ -62,6 +62,7 @@ async fn spawn_upstream() -> Option { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "1".to_string(), @@ -100,6 +101,7 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: "n1".to_string(), diff --git a/app/crowdb-web/tests/ops_migration_test.rs b/app/crowdb-web/tests/ops_migration_test.rs index 996d0567..fecc7faa 100644 --- a/app/crowdb-web/tests/ops_migration_test.rs +++ b/app/crowdb-web/tests/ops_migration_test.rs @@ -52,6 +52,7 @@ async fn spawn_upstream() -> Option { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "1".to_string(), @@ -89,6 +90,7 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: "n1".to_string(), diff --git a/app/crowdb-web/tests/recursive_query_test.rs b/app/crowdb-web/tests/recursive_query_test.rs index 81d0150f..cd0b23cc 100644 --- a/app/crowdb-web/tests/recursive_query_test.rs +++ b/app/crowdb-web/tests/recursive_query_test.rs @@ -142,6 +142,7 @@ async fn spawn_web_with_seeded_physical_tree() -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); let state = AppState::with_config(cfg, None); diff --git a/app/crowdb-web/tests/replica_leader_removal_test.rs b/app/crowdb-web/tests/replica_leader_removal_test.rs index 666e788f..e584c0c0 100644 --- a/app/crowdb-web/tests/replica_leader_removal_test.rs +++ b/app/crowdb-web/tests/replica_leader_removal_test.rs @@ -95,6 +95,7 @@ async fn spawn_upstream(node_id: u64, workspace: &std::path::Path) -> Option) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: u.node_id.to_string(), diff --git a/app/crowdb-web/tests/replica_routes_test.rs b/app/crowdb-web/tests/replica_routes_test.rs index 070077f8..55d1cc07 100644 --- a/app/crowdb-web/tests/replica_routes_test.rs +++ b/app/crowdb-web/tests/replica_routes_test.rs @@ -82,6 +82,7 @@ async fn spawn_upstream(node_id: u64, workspace: &std::path::Path) -> Option SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: u.node_id.to_string(), diff --git a/app/crowdb-web/tests/rolling_upgrade_test.rs b/app/crowdb-web/tests/rolling_upgrade_test.rs index 0c8b7528..c624d5cd 100644 --- a/app/crowdb-web/tests/rolling_upgrade_test.rs +++ b/app/crowdb-web/tests/rolling_upgrade_test.rs @@ -77,6 +77,7 @@ impl Cluster { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let replica_id = node_id; let extra_args = vec![ @@ -150,6 +151,7 @@ async fn spawn_upstream(node_id: u64, workspace: &std::path::Path, binary: &Path ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: node_id.to_string(), @@ -195,6 +197,7 @@ async fn spawn_web(upstreams: &BTreeMap) -> SocketAddr { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); cfg.add_server(ServerEntry { id: u.node_id.to_string(), diff --git a/container/crowdb-monitor/src/bootstrap/hardware.rs b/container/crowdb-monitor/src/bootstrap/hardware.rs index 81535fdb..7e835871 100644 --- a/container/crowdb-monitor/src/bootstrap/hardware.rs +++ b/container/crowdb-monitor/src/bootstrap/hardware.rs @@ -403,6 +403,7 @@ fn expected(profile: &DeploymentProfile) -> Result Result, /// Optional explicit private-key path. `None` falls back to /// `~/.ssh/id_ed25519` then `~/.ssh/id_rsa`. #[serde(default, skip_serializing_if = "Option::is_none")] @@ -431,6 +434,8 @@ struct PersistedNodeEntry { #[serde(default)] ssh_user: String, #[serde(default, skip_serializing_if = "Option::is_none")] + ssh_credential_ref: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] ssh_key: Option, #[serde(default, skip_serializing_if = "Option::is_none")] ssh_password: Option, @@ -971,6 +976,7 @@ impl ConsoleConfig { host: entry.host.clone(), ssh_port: entry.ssh_port, ssh_user: entry.ssh_user.clone(), + ssh_credential_ref: entry.ssh_credential_ref.clone(), ssh_key: entry.ssh_key.clone(), ssh_password: entry.ssh_password.clone(), }, @@ -1091,6 +1097,7 @@ impl ConsoleConfig { host: entry.host, ssh_port: entry.ssh_port, ssh_user: entry.ssh_user, + ssh_credential_ref: entry.ssh_credential_ref, ssh_key: entry.ssh_key, ssh_password: entry.ssh_password, }) @@ -1210,6 +1217,21 @@ mod tests { cfg.add_server(a).unwrap(); cfg.add_server(ServerEntry::new("b", "http://127.0.0.1:10001")) .unwrap(); + cfg.racks.push(super::RackEntry { + id: 1, + name: "rack-a".into(), + }); + cfg.nodes.push( + serde_json::from_value(serde_json::json!({ + "id": 1, + "rack_id": 1, + "host": "node.example", + "ssh_port": 2222, + "ssh_user": "operator", + "ssh_credential_ref": "node-1" + })) + .unwrap(), + ); cfg.local_launches.insert( "b".into(), LocalLaunchSpec { diff --git a/lib/crowdb-console-shared/src/launch/remote.rs b/lib/crowdb-console-shared/src/launch/remote.rs index 1d2ac73a..1c155bd0 100644 --- a/lib/crowdb-console-shared/src/launch/remote.rs +++ b/lib/crowdb-console-shared/src/launch/remote.rs @@ -24,6 +24,7 @@ async fn connect(launch: &LaunchRecord, credential_root: &Path) -> Result Result<()> { for node in nodes { if ctx.sysmd().get_rack(node.rack_id).await?.is_none() { + let name = ctx + .config() + .racks + .iter() + .find(|rack| rack.id == node.rack_id) + .map(|rack| rack.name.clone()) + .unwrap_or_default(); ctx.sysmd() .add_rack( node.rack_id, &RackValue { status: HwStatus::Up as i32, node_ids: Vec::new(), + name, }, ) .await?; @@ -912,6 +920,10 @@ async fn ensure_diskdb_hardware(ctx: &OpContext, nodes: &[NodeEntry]) -> Result< disk_group_ids: Vec::new(), status_changed_at_ms: 0, temp_failure_since_ms: None, + management_host: node.host.clone(), + ssh_port: node.ssh_port, + ssh_user: node.ssh_user.clone(), + ssh_credential_ref: node.ssh_credential_ref.clone(), }, ) .await?; @@ -1269,6 +1281,7 @@ fn write_rack_and_nodes(ctx: &OpContext, rack_id: u64, node_ids: &[u64]) { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }); } } @@ -1325,6 +1338,7 @@ async fn deploy_servers( ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; // Every process owns a stable server directory. WAL and btree data // remain direct children of that directory as waldata/ and ctdata/. diff --git a/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs b/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs index 86e926ae..d81b7d59 100644 --- a/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs +++ b/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs @@ -97,6 +97,7 @@ fn intended_records(ctx: &OpContext, store_nodes: &[u64], members: &[(u64, u64)] RackValue { status: HwStatus::Up as i32, node_ids: Vec::new(), + name: rack.name.clone(), }, )?); } @@ -108,6 +109,10 @@ fn intended_records(ctx: &OpContext, store_nodes: &[u64], members: &[(u64, u64)] }, NodeValue { status: HwStatus::Up as i32, + management_host: node.host.clone(), + ssh_port: node.ssh_port, + ssh_user: node.ssh_user.clone(), + ssh_credential_ref: node.ssh_credential_ref.clone(), ..Default::default() }, )?); diff --git a/lib/crowdb-console-shared/src/ops/hardware.rs b/lib/crowdb-console-shared/src/ops/hardware.rs index 09c6e1f9..8d88df65 100644 --- a/lib/crowdb-console-shared/src/ops/hardware.rs +++ b/lib/crowdb-console-shared/src/ops/hardware.rs @@ -35,6 +35,7 @@ pub async fn add_rack(ctx: &OpContext, rack_id: u64, name: &str) -> Result Result { disk_group_ids: Vec::new(), status_changed_at_ms: 0, temp_failure_since_ms: None, + management_host: entry.host.clone(), + ssh_port: entry.ssh_port, + ssh_user: entry.ssh_user.clone(), + ssh_credential_ref: entry.ssh_credential_ref.clone(), }; let _ = ctx.sysmd().add_node(entry.rack_id, entry.id, &value).await; } @@ -565,6 +570,7 @@ mod tests { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); config diff --git a/lib/crowdb-console-shared/src/ssh.rs b/lib/crowdb-console-shared/src/ssh.rs index 4d87c2a9..274709c3 100644 --- a/lib/crowdb-console-shared/src/ssh.rs +++ b/lib/crowdb-console-shared/src/ssh.rs @@ -423,6 +423,7 @@ mod tests { ssh_user: user.into(), ssh_key: key.map(Into::into), ssh_password: password.map(Into::into), + ssh_credential_ref: None, } } diff --git a/lib/crowdb-console-shared/tests/kv_e2e_test.rs b/lib/crowdb-console-shared/tests/kv_e2e_test.rs index 25fa07fd..ba37a579 100644 --- a/lib/crowdb-console-shared/tests/kv_e2e_test.rs +++ b/lib/crowdb-console-shared/tests/kv_e2e_test.rs @@ -36,6 +36,7 @@ async fn spawn_server() -> Option<(u32, String)> { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let req = DeployRequest { server_id: "s1".into(), diff --git a/lib/crowdb-console-shared/tests/lifecycle_e2e_test.rs b/lib/crowdb-console-shared/tests/lifecycle_e2e_test.rs index acbde233..6ca496bb 100644 --- a/lib/crowdb-console-shared/tests/lifecycle_e2e_test.rs +++ b/lib/crowdb-console-shared/tests/lifecycle_e2e_test.rs @@ -79,6 +79,7 @@ async fn deploy_local_and_observe_topology() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); diff --git a/lib/crowdb-console-shared/tests/mgmt_e2e_test.rs b/lib/crowdb-console-shared/tests/mgmt_e2e_test.rs index 0e1cbcea..c488fc2f 100644 --- a/lib/crowdb-console-shared/tests/mgmt_e2e_test.rs +++ b/lib/crowdb-console-shared/tests/mgmt_e2e_test.rs @@ -35,6 +35,7 @@ async fn spawn_server() -> Option<(u32, ServerClient)> { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }; let _ = RackEntry { id: 1, diff --git a/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs b/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs index cc539e24..6404a3d2 100644 --- a/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs +++ b/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs @@ -17,6 +17,7 @@ async fn bootstrap_rejects_conflicting_hardware_without_overwriting_authority() let existing = RackValue { status: HwStatus::Up as i32, node_ids: vec![99], + ..Default::default() }; ctx.sysmd().add_rack(1, &existing).await.unwrap(); let result = cluster::init(&ctx, &[1]).await; @@ -40,6 +41,38 @@ async fn bootstrap_preflights_logical_conflicts_before_publishing_missing_hardwa assert!(ctx.sysmd().get_rack(1).await.unwrap().is_none()); } +#[tokio::test] +async fn bootstrap_rejects_a_different_rack_name_even_when_membership_matches() { + let cluster = KvCluster::start().await; + let ctx = context(&cluster).await; + let existing = RackValue { + status: HwStatus::Up as i32, + name: "another-rack".into(), + ..Default::default() + }; + ctx.sysmd().add_rack(1, &existing).await.unwrap(); + let result = cluster::init(&ctx, &[1]).await; + assert!(matches!(result, Err(Error::Conflict { .. })), "{result:?}"); + assert_eq!(ctx.sysmd().get_rack(1).await.unwrap(), Some(existing)); +} + +#[tokio::test] +async fn bootstrap_publishes_shared_hardware_labels_and_credential_reference() { + let cluster = KvCluster::start().await; + let ctx = context(&cluster).await; + { + let mut config = ctx.config_mut(); + config.racks[0].name = "rack-a".into(); + config.nodes[0].ssh_credential_ref = Some("ops-key".into()); + } + cluster::init(&ctx, &[1]).await.unwrap(); + assert_eq!(ctx.sysmd().get_rack(1).await.unwrap().unwrap().name, "rack-a"); + let node = ctx.sysmd().get_node(1, 1).await.unwrap().unwrap(); + assert_eq!(node.management_host, "127.0.0.1"); + assert_eq!(node.ssh_port, 22); + assert_eq!(node.ssh_credential_ref.as_deref(), Some("ops-key")); +} + #[tokio::test] async fn bootstrap_resumes_missing_records_and_preserves_committed_revisions() { let cluster = KvCluster::start().await; @@ -50,6 +83,7 @@ async fn bootstrap_resumes_missing_records_and_preserves_committed_revisions() { &RackValue { status: HwStatus::Up as i32, node_ids: Vec::new(), + ..Default::default() }, ) .await @@ -88,4 +122,9 @@ async fn bootstrap_resumes_missing_records_and_preserves_committed_revisions() { assert_eq!(before, after, "matching committed content must not be rewritten"); assert_eq!(ctx.sysmd().get_store(0).await.unwrap().unwrap().node_ids, vec![1]); assert_eq!(ctx.sysmd().list_replicas_in_group(0, 0).await.unwrap().len(), 1); + let node = ctx.sysmd().get_node(1, 1).await.unwrap().unwrap(); + assert_eq!(node.management_host, "127.0.0.1"); + assert_eq!(node.ssh_port, 22); + assert!(node.ssh_user.is_empty()); + assert!(node.ssh_credential_ref.is_none()); } diff --git a/lib/crowdb-console-shared/tests/ops_hardware_test.rs b/lib/crowdb-console-shared/tests/ops_hardware_test.rs index a6cb91ff..159a45dd 100644 --- a/lib/crowdb-console-shared/tests/ops_hardware_test.rs +++ b/lib/crowdb-console-shared/tests/ops_hardware_test.rs @@ -53,6 +53,7 @@ async fn remove_rack_with_nodes_conflict() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }, ) .await @@ -75,6 +76,7 @@ async fn add_node_and_list() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }, ) .await @@ -101,6 +103,7 @@ async fn add_node_unknown_rack_validation() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }, ) .await @@ -129,6 +132,7 @@ async fn remove_node_with_server_conflict() { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }, ) .await diff --git a/lib/crowdb-console-shared/tests/ops_kv_server_test.rs b/lib/crowdb-console-shared/tests/ops_kv_server_test.rs index de065a40..a8be3111 100644 --- a/lib/crowdb-console-shared/tests/ops_kv_server_test.rs +++ b/lib/crowdb-console-shared/tests/ops_kv_server_test.rs @@ -25,6 +25,7 @@ fn ctx_with_node() -> OpContext { ssh_user: String::new(), ssh_key: None, ssh_password: None, + ssh_credential_ref: None, }) .unwrap(); OpContext::new_for_test("127.0.0.1:59999".into(), vec![], cfg) diff --git a/lib/crowdb-protocol/src/types/common.rs b/lib/crowdb-protocol/src/types/common.rs index 2f7c941f..b957450d 100644 --- a/lib/crowdb-protocol/src/types/common.rs +++ b/lib/crowdb-protocol/src/types/common.rs @@ -177,6 +177,8 @@ pub struct ErrorInfo { pub struct RackValue { pub status: i32, pub node_ids: Vec, + #[serde(default, skip_serializing_if = "String::is_empty")] + pub name: String, } #[derive(Clone, Debug, PartialEq, Default, Serialize, Deserialize)] @@ -186,6 +188,18 @@ pub struct NodeValue { pub disk_group_ids: Vec, pub status_changed_at_ms: u64, pub temp_failure_since_ms: Option, + #[serde(default, skip_serializing_if = "String::is_empty")] + pub management_host: String, + #[serde(default, skip_serializing_if = "is_zero_u16")] + pub ssh_port: u16, + #[serde(default, skip_serializing_if = "String::is_empty")] + pub ssh_user: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub ssh_credential_ref: Option, +} + +fn is_zero_u16(value: &u16) -> bool { + *value == 0 } #[derive(Clone, Debug, PartialEq, Default, Serialize, Deserialize)] diff --git a/lib/crowdb-protocol/tests/hardware_identity_test.rs b/lib/crowdb-protocol/tests/hardware_identity_test.rs new file mode 100644 index 00000000..723183a7 --- /dev/null +++ b/lib/crowdb-protocol/tests/hardware_identity_test.rs @@ -0,0 +1,27 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_protocol::common::{NodeValue, RackValue}; + +#[test] +fn hardware_values_preserve_shared_identity_and_read_older_records() { + let rack: RackValue = serde_json::from_str(r#"{"status":1,"node_ids":[7]}"#).unwrap(); + assert!(rack.name.is_empty()); + let node: NodeValue = serde_json::from_str( + r#"{"status":1,"last_used_dg_id":0,"disk_group_ids":[],"status_changed_at_ms":0,"temp_failure_since_ms":null}"#, + ) + .unwrap(); + assert!(node.management_host.is_empty()); + assert_eq!(node.ssh_port, 0); + assert!(node.ssh_credential_ref.is_none()); + + let current = NodeValue { + management_host: "node.example".into(), + ssh_port: 2222, + ssh_user: "operator".into(), + ssh_credential_ref: Some("node-7".into()), + ..node + }; + let round_trip: NodeValue = serde_json::from_slice(&serde_json::to_vec(¤t).unwrap()).unwrap(); + assert_eq!(round_trip, current); +} diff --git a/lib/crowdb-test-harness/src/hardware.rs b/lib/crowdb-test-harness/src/hardware.rs index 1f785f48..edad8a50 100644 --- a/lib/crowdb-test-harness/src/hardware.rs +++ b/lib/crowdb-test-harness/src/hardware.rs @@ -33,6 +33,7 @@ pub async fn seed_hardware(hw: &HardwareClient, disk_ids: &[DiskId]) { &RackValue { status: HwStatus::Up as i32, node_ids: vec![NODE_ID], + ..Default::default() }, ) .await @@ -47,6 +48,7 @@ pub async fn seed_hardware(hw: &HardwareClient, disk_ids: &[DiskId]) { disk_group_ids: vec![DG_ID], status_changed_at_ms: 0, temp_failure_since_ms: None, + ..Default::default() }, ) .await From 22550261c3782235be0f9e2356cf83fcbee41cb2 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 09:27:56 +0800 Subject: [PATCH 37/74] Record multipart replacement race --- doc/backlog/R167-s3-multipart-upload.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/doc/backlog/R167-s3-multipart-upload.md b/doc/backlog/R167-s3-multipart-upload.md index 1e6bd873..e55a33e8 100644 --- a/doc/backlog/R167-s3-multipart-upload.md +++ b/doc/backlog/R167-s3-multipart-upload.md @@ -92,3 +92,9 @@ Required gates: recovery is bound to its catalog identity and store. Extract the remaining protocol-neutral transition decisions while keeping keys, authorization and responses in the protocol adapters. +- A part writer can load `Open` before completion freezes the session and + advance the current-part pointer afterward. Publication uses the immutable + frozen generation, so the published bytes are stable, but the current pointer + may differ at publication time. The acceptance rule above requires a + concurrent replacement to prevent publication; add a session fence to part + replacement or define an explicit snapshot boundary, then test the race. From bcd21bf3422743af8253cd3401ed37e764fe1418 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 09:53:17 +0800 Subject: [PATCH 38/74] Fence multipart part publication against completion --- .../src/s3/operations/multipart.rs | 5 +- .../s3/design-crowdb-access-s3.md | 24 ++ doc/working/plan-s3-multipart.md | 31 ++- lib/crowdb-access-s3/src/metadata.rs | 1 + .../src/metadata/multipart.rs | 32 ++- .../src/metadata/multipart_repository.rs | 103 +------ .../multipart_repository/completion.rs | 11 +- .../metadata/multipart_repository/listing.rs | 4 + .../metadata/multipart_repository/parts.rs | 253 ++++++++++++++++++ .../multipart_repository/publication.rs | 16 +- .../metadata/multipart_repository/terminal.rs | 12 +- .../tests/metadata_multipart_test.rs | 1 + .../tests/multipart_repository_test.rs | 104 ++++++- lib/crowdb-access-s3/tests/wire_test.rs | 1 + 14 files changed, 466 insertions(+), 132 deletions(-) create mode 100644 lib/crowdb-access-s3/src/metadata/multipart_repository/parts.rs diff --git a/app/crowdb-access-server/src/s3/operations/multipart.rs b/app/crowdb-access-server/src/s3/operations/multipart.rs index 2f2240de..c96aa520 100644 --- a/app/crowdb-access-server/src/s3/operations/multipart.rs +++ b/app/crowdb-access-server/src/s3/operations/multipart.rs @@ -127,6 +127,7 @@ impl ProductionS3Operations { max_staged_bytes: MAX_STAGED_BYTES, part_count: 0, staged_bytes: 0, + pending: None, selection: None, completion_request_digest: None, publication_ms: None, @@ -366,9 +367,11 @@ fn map_multipart_error(error: &MultipartRepositoryError) -> S3ErrorCode { MultipartRepositoryError::Key(_) | MultipartRepositoryError::Record(_) => S3ErrorCode::InvalidRequest, MultipartRepositoryError::Store(_) => S3ErrorCode::ServiceUnavailable, MultipartRepositoryError::Conflict => S3ErrorCode::NoSuchUpload, + MultipartRepositoryError::Busy | MultipartRepositoryError::ScanBudgetExhausted => { + S3ErrorCode::SlowDown + } MultipartRepositoryError::InvalidPart => S3ErrorCode::InvalidPart, MultipartRepositoryError::EntityTooSmall => S3ErrorCode::EntityTooSmall, - MultipartRepositoryError::ScanBudgetExhausted => S3ErrorCode::SlowDown, } } diff --git a/doc/design/access-server/s3/design-crowdb-access-s3.md b/doc/design/access-server/s3/design-crowdb-access-s3.md index ffaf486f..20a9107e 100644 --- a/doc/design/access-server/s3/design-crowdb-access-s3.md +++ b/doc/design/access-server/s3/design-crowdb-access-s3.md @@ -73,6 +73,27 @@ reused. Listings are ordered and continuation-safe within their documented consistency model. Continuation state is opaque and bound to the original request scope. +Multipart uploads keep a durable session, current part pointers, and immutable +part generations under one upload prefix. Replacing a part number advances its +generation while retaining the previous generation as reference evidence for +the chunk reclamation scan. The upload session records admission bounds, part +accounting, expiration, and completion state. Its pending part mutation is a +durable reservation: a writer stores the immutable generation, reserves the +pointer update with a session compare-and-swap, publishes the pointer, then +clears the reservation. A later request can finish an interrupted reservation. +Completion cannot freeze while one is pending. + +Completion validates the ordered selected part numbers, raw MD5 values, +minimum nonfinal size, and current generations. It freezes the selection under +the session compare-and-swap, composes chunk locations with adjusted logical +offsets, and publishes one object record through a predecessor-fenced +object-key mutation. Publication checks the current pointers and immutable +generation digests again. It never reads or concatenates part bytes. The +multipart ETag is the hexadecimal MD5 of the selected raw part MD5 values in +order, followed by the part count suffix. Abort and expiry mark the session +terminal; physical chunk reclamation follows the ordinary reference and age +checks. + ## 5. Relationship to other access models Iceberg is not implemented as special objects in the S3 namespace. Its catalog, @@ -99,3 +120,6 @@ native topology access remain Dataset semantics. reclaim Iceberg or Dataset authority. - **S3-I7 — Transport equivalence:** ordinary HTTP and accelerated transfer produce the same S3 range, integrity, publication, and error outcome. +- **S3-I8 — Multipart selection fence:** a part pointer cannot advance after + completion freezes its selection. An interrupted pointer reservation is + settled before completion or abort proceeds. diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md index 45747da2..fdd6b71b 100644 --- a/doc/working/plan-s3-multipart.md +++ b/doc/working/plan-s3-multipart.md @@ -19,7 +19,7 @@ Iceberg multipart path. `lib/crowdb-access-multipart`, preserving the existing selected-part fences in Iceberg. Include overflow and malformed location tests. Files: new crate, Iceberg publication, workspace manifests. -- [~] **Durable transition core**: isolate session/part states, replacement +- [x] **Durable transition core**: isolate session/part states, replacement generations, completion selection, abort and recovery transitions from Iceberg catalog-specific keys and records. Keep store CAS and namespace adaptation in each protocol. Shared phase vocabulary, admission bounds, @@ -27,13 +27,14 @@ Iceberg multipart path. both adapters; storage CAS and durable record layouts remain protocol-specific. Files: shared multipart crate, Iceberg file repository, S3 metadata store. Both adapters - now also use the same inclusive/exclusive lifetime decision and checked - session/part revision advancement. The remaining Iceberg completion/recovery - sequencing still needs protocol-neutral extraction. + use the same inclusive/exclusive lifetime decision and checked session/part + revision advancement. Iceberg's byte assembly checkpoints and S3's + metadata-only object publication remain in their adapters because their + durable evidence and work units differ. ## S3 adapter and HTTP -- [ ] **Durable S3 records**: add upload and part keys/records with raw 16-byte +- [x] **Durable S3 records**: add upload and part keys/records with raw 16-byte MD5 and selected revision under one upload prefix. Use bucket identity and object key as namespace scope; preserve immutable part data after replacement. Versioned session/part records and ordered, binary-safe keys are in place; @@ -44,7 +45,7 @@ Iceberg multipart path. in place. The session, current part and immutable generations now share one upload prefix for R95; an immutable object-key/upload-ID index preserves bounded ListMultipartUploads ordering. HTTP wiring remains. -- [ ] **S3 routes and wire**: classify create/upload/list/complete/abort/list +- [x] **S3 routes and wire**: classify create/upload/list/complete/abort/list uploads, parse bounded completion XML, emit compatible responses and errors. Preserve SigV4 authentication and existing basic routes. The repository now provides bounded, ordered ListParts pagination over current generations; @@ -62,14 +63,14 @@ Iceberg multipart path. malformed XML remains `InvalidRequest`. Full boto3 stack acceptance passes Create, UploadPart, ListParts, Complete, Abort and ListUploads, including replay, replacement and invalid ETag cases. -- [ ] **Part ingestion**: reuse the bounded streaming writer and admission +- [x] **Part ingestion**: reuse the bounded streaming writer and admission budget, persist part location/integrity before success, reconcile lost replies. Production UploadPart now uses the basic streaming writer and saves raw MD5 with locations. A byte-identical retry keeps the selected generation even when a new write produced different chunk locations; R95 can reclaim those unreachable chunks. Full stack ingestion passes 5 MiB and small parts, a dropped UploadPart success response, and retries after service restart. -- [ ] **Atomic completion**: fence selected part generations, validate order, +- [x] **Atomic completion**: fence selected part generations, validate order, count, size and checksum, compose locations through the shared core, and publish one immutable object generation without reading part bytes. The S3 adapter now freezes selection under session CAS, validates it again before @@ -77,7 +78,15 @@ Iceberg multipart path. immutable generation records preserve selected bytes across a concurrent part-number replacement. Metadata-only completion and byte-exact GET pass the full boto3 stack, including a repeated Complete request. -- [ ] **Abort and expiry**: make terminal states idempotent and preserve the +- [x] **Part publication fence**: reserve each changed current-part pointer + under session CAS, persist the immutable generation, settle the pointer and + clear the reservation. Complete and Abort must not pass an unresolved + reservation. A helper can finish a committed reservation after reply loss or + restart. Test a replacement racing with freeze, then test recovery at every + durable boundary. Keep this CAS-based path lock-free and retain orphan + generations for R95. A focused test freezes an interrupted reservation only + after recovery, and a changed pointer prevents object publication. +- [x] **Abort and expiry**: make terminal states idempotent and preserve the part generations that R95's chunk-centered scanner needs for reference checks. The S3 adapter now has an idempotent, response-loss-safe logical abort and a bounded hourly expiry sweep over the upload listing index. Session and part @@ -87,7 +96,7 @@ Iceberg multipart path. ## Acceptance and cleanup -- [ ] **Focused and E2E tests**: known MD5 vectors, out-of-order/replaced parts, +- [x] **Focused and E2E tests**: known MD5 vectors, out-of-order/replaced parts, invalid completion, response loss and restart, abort/expiry metadata, and ordinary single-part compatibility. The complete S3 library and access-server suites plus 19 full-stack boto3/restart cases pass. The new cases drop @@ -95,7 +104,7 @@ Iceberg multipart path. incomplete session across six service restarts, then complete it. Hourly expiry has a focused metadata test; broader concurrency and error-matrix acceptance remains. -- [ ] **Gates and docs**: run both access crate suites, access-server E2E, +- [x] **Gates and docs**: run both access crate suites, access-server E2E, Rust fmt and clippy separately; update S3 design and remove R167 plus this plan only after all acceptance criteria pass. diff --git a/lib/crowdb-access-s3/src/metadata.rs b/lib/crowdb-access-s3/src/metadata.rs index f126a8c1..e1f20472 100644 --- a/lib/crowdb-access-s3/src/metadata.rs +++ b/lib/crowdb-access-s3/src/metadata.rs @@ -25,6 +25,7 @@ mod generated { pub use key::{BucketId, MetadataKey, MetadataKeyError, TenantId}; pub use multipart::{ new_upload_id, MultipartPartRecord, MultipartPhase, MultipartRecordError, MultipartSessionRecord, + PendingPartMutation, }; pub use multipart_repository::{ CompletionPart, MultipartExpiryPage, MultipartPartPage, MultipartRepository, MultipartRepositoryError, diff --git a/lib/crowdb-access-s3/src/metadata/multipart.rs b/lib/crowdb-access-s3/src/metadata/multipart.rs index 3042bb18..dadef029 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart.rs @@ -4,14 +4,16 @@ //! Versioned durable S3 multipart session and part values. use bincode::Options as _; -use crowdb_access_multipart::{validate_selected_parts, MultipartBounds, MultipartComposer, SelectedPart}; +use crowdb_access_multipart::{ + next_part_revision, validate_selected_parts, MultipartBounds, MultipartComposer, SelectedPart, +}; use crowdb_protocol::chunkdb::rpc::Location; use serde::{Deserialize, Serialize}; use sha2::{Digest as _, Sha256}; use super::BucketId; -const SESSION_MAGIC: [u8; 5] = *b"S3MS\x01"; +const SESSION_MAGIC: [u8; 5] = *b"S3MS\x02"; const PART_MAGIC: [u8; 5] = *b"S3MP\x01"; const MAX_RECORD_BYTES: u64 = 1024 * 1024; const MAX_OBJECT_KEY_BYTES: usize = 1024; @@ -44,6 +46,7 @@ pub struct MultipartSessionRecord { pub max_staged_bytes: u64, pub part_count: u16, pub staged_bytes: u64, + pub pending: Option, pub selection: Option>, pub completion_request_digest: Option<[u8; 32]>, pub publication_ms: Option, @@ -51,6 +54,18 @@ pub struct MultipartSessionRecord { pub etag: Option, } +/// A durable session fence for publishing one current-part pointer. +/// The immutable after-generation is stored before reserving this mutation. +#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] +pub struct PendingPartMutation { + pub number: u16, + pub before_revision: Option, + pub before_digest: Option<[u8; 32]>, + pub after_revision: u64, + pub after_digest: [u8; 32], + pub after_length: u64, +} + #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] pub struct MultipartPartRecord { pub bucket_id: BucketId, @@ -132,6 +147,19 @@ impl MultipartSessionRecord { return Err(MultipartRecordError::Invalid); } } + if let Some(pending) = &self.pending { + if self.phase != MultipartPhase::Open + || self.part_count == 0 + || pending.number == 0 + || pending.number > self.max_parts + || pending.after_length > self.max_part_bytes + || self.staged_bytes < pending.after_length + || pending.before_revision.is_some() != pending.before_digest.is_some() + || next_part_revision(pending.before_revision) != Some(pending.after_revision) + { + return Err(MultipartRecordError::Invalid); + } + } let selected_count = self.selection.as_ref().map_or(0, Vec::len); match self.phase { MultipartPhase::Open diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs index 4cc3a5e3..da625fa1 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository.rs @@ -3,7 +3,7 @@ //! CAS-backed multipart authority in the S3 Chunk-KV namespace. -use crowdb_access_multipart::{live_at, next_part_revision, next_revision}; +use crowdb_access_multipart::next_revision; use std::sync::Arc; use super::{ @@ -13,6 +13,7 @@ use super::{ mod completion; mod listing; +mod parts; mod publication; mod terminal; @@ -30,6 +31,8 @@ pub enum MultipartRepositoryError { Store(#[from] MetadataStoreError), #[error("multipart operation conflicts with durable state")] Conflict, + #[error("multipart mutation is still settling")] + Busy, #[error("multipart completion references a missing or changed part")] InvalidPart, #[error("a nonfinal multipart part is smaller than 5 MiB")] @@ -222,104 +225,6 @@ impl MultipartRepository { .transpose() } - /// Conditionally publishes an independently streamed part generation. - /// - /// Distinct part numbers write independent keys. The conservative - /// `max_parts * max_part_bytes` bound prevents aggregate staged bytes from - /// exceeding the session budget even when all parts arrive concurrently. - /// - /// # Errors - /// Rejects closed, expired or foreign sessions, invalid parts and - /// unconfirmed storage errors. A competing writer returns `None`. - pub async fn put_stream_part( - &self, - session: &MultipartSessionRecord, - part: &MultipartPartRecord, - now_ms: u64, - ) -> Result, MultipartRepositoryError> { - let current = self - .load(session) - .await? - .ok_or(MultipartRepositoryError::Conflict)?; - if current.phase != MultipartPhase::Open - || !live_at(current.created_ms, current.expires_ms, now_ms) - || u64::from(current.max_parts) - .checked_mul(current.max_part_bytes) - .map_or(true, |bytes| bytes > current.max_staged_bytes) - || part.bucket_id != current.bucket_id - || part.upload_id != current.upload_id - || part.length > current.max_part_bytes - || part.number > current.max_parts - { - return Err(MultipartRepositoryError::Conflict); - } - let before = self.part(¤t, part.number).await?; - if let Some(existing) = &before { - if existing.length == part.length && existing.raw_md5 == part.raw_md5 { - return Ok(Some(existing.clone())); - } - } - let mut after = part.clone(); - after.revision = next_part_revision(before.as_ref().map(|before| before.revision)) - .ok_or(MultipartRepositoryError::Conflict)?; - after.modified_ms = now_ms; - let value = after.encode()?; - let generation_key = MetadataKey::multipart_part_generation( - &self.tenant, - current.bucket_id, - ¤t.upload_id, - part.number, - after.revision, - )?; - let generation_write = self.store.put_if_absent(generation_key, value.clone()).await; - match generation_write { - Ok(PutIfAbsentOutcome::Inserted { .. }) => {} - Ok(PutIfAbsentOutcome::Existing(existing)) if existing.value == value => {} - Ok(PutIfAbsentOutcome::Existing(_)) => return Ok(None), - Err(error) => { - if self - .part_generation(¤t, part.number, after.revision) - .await? - .as_ref() - != Some(&after) - { - return Err(error.into()); - } - } - } - let key = - MetadataKey::multipart_part(&self.tenant, current.bucket_id, ¤t.upload_id, part.number)?; - let outcome = if let Some(before) = &before { - self.store - .compare_exchange(key, before.encode()?, value.clone()) - .await - .map(|applied| applied.then_some(after.clone())) - } else { - self.store - .put_if_absent(key, value.clone()) - .await - .map(|outcome| match outcome { - PutIfAbsentOutcome::Inserted { .. } => Some(after.clone()), - PutIfAbsentOutcome::Existing(existing) if existing.value == value => Some(after.clone()), - PutIfAbsentOutcome::Existing(_) => None, - }) - }; - match outcome { - Ok(Some(result)) => Ok(Some(result)), - Ok(None) => Ok(self - .part(¤t, part.number) - .await? - .filter(|existing| existing == &after)), - Err(error) => { - if self.part(¤t, part.number).await?.as_ref() == Some(&after) { - Ok(Some(after)) - } else { - Err(error.into()) - } - } - } - } - fn session_key(&self, session: &MultipartSessionRecord) -> Vec { MetadataKey::multipart_session(&self.tenant, session.bucket_id, &session.upload_id) } diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs index 3c392218..f288a5cc 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/completion.rs @@ -22,8 +22,8 @@ pub struct CompletionPart { impl MultipartRepository { /// Validates and freezes an ordered part selection under the session CAS. /// - /// The part records can still race with this phase write. Publication - /// rechecks every recorded digest and refuses changed generations. + /// A reserved part mutation settles before this phase write. Publication + /// rechecks the selected pointer and generation evidence. /// /// # Errors /// Rejects missing, duplicate, undersized or mismatched parts and @@ -46,8 +46,11 @@ impl MultipartRepository { { return Ok(Some(current)); } - if current != *session - || current.phase != MultipartPhase::Open + if current.pending.is_some() { + self.settle_pending_part(¤t).await?; + return Ok(None); + } + if current.phase != MultipartPhase::Open || !live_at(current.created_ms, current.expires_ms, now_ms) || requested.is_empty() || requested.len() > usize::from(current.max_parts) diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs index c4af6993..3a7b3624 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/listing.rs @@ -139,6 +139,10 @@ impl MultipartRepository { if current.phase != MultipartPhase::Open { return Err(MultipartRepositoryError::Conflict); } + if current.pending.is_some() { + self.settle_pending_part(¤t).await?; + return Err(MultipartRepositoryError::Busy); + } if after_number == 10_000 { return Ok(MultipartPartPage { parts: Vec::new(), diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/parts.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/parts.rs new file mode 100644 index 00000000..36a7ce58 --- /dev/null +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/parts.rs @@ -0,0 +1,253 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Durable part-pointer reservations shared with completion's session fence. + +use crate::metadata::PendingPartMutation; +use crowdb_access_multipart::{ + live_at, next_part_revision, next_revision, reserve_part_accounting, PartAccounting, +}; +use sha2::{Digest as _, Sha256}; + +use super::{ + MetadataKey, MultipartPartRecord, MultipartPhase, MultipartRepository, MultipartRepositoryError, + MultipartSessionRecord, PutIfAbsentOutcome, +}; + +impl MultipartRepository { + /// Publishes a streamed part only while the upload remains open. + /// + /// The immutable generation is written first. A session CAS then reserves + /// the current-part pointer mutation, so completion cannot freeze a stale + /// pointer while an earlier part writer publishes. An abandoned generation + /// remains available to R95's chunk-centered reclamation. + /// + /// # Errors + /// Rejects closed, expired or foreign sessions, invalid parts and + /// unconfirmed storage errors. A competing writer returns `None`. + pub async fn put_stream_part( + &self, + session: &MultipartSessionRecord, + part: &MultipartPartRecord, + now_ms: u64, + ) -> Result, MultipartRepositoryError> { + for _ in 0..2 { + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.pending.is_some() { + self.settle_pending_part(¤t).await?; + continue; + } + return self.reserve_stream_part(¤t, part, now_ms).await; + } + Ok(None) + } + + async fn reserve_stream_part( + &self, + current: &MultipartSessionRecord, + part: &MultipartPartRecord, + now_ms: u64, + ) -> Result, MultipartRepositoryError> { + if current.phase != MultipartPhase::Open + || !live_at(current.created_ms, current.expires_ms, now_ms) + || part.bucket_id != current.bucket_id + || part.upload_id != current.upload_id + || part.number == 0 + || part.number > current.max_parts + || part.length > current.max_part_bytes + { + return Err(MultipartRepositoryError::Conflict); + } + let before = self.part(current, part.number).await?; + if let Some(existing) = &before { + if existing.length == part.length && existing.raw_md5 == part.raw_md5 { + return Ok(Some(existing.clone())); + } + } + let mut after = part.clone(); + after.revision = next_part_revision(before.as_ref().map(|record| record.revision)) + .ok_or(MultipartRepositoryError::Conflict)?; + after.modified_ms = now_ms; + let accounting = reserve_part_accounting( + PartAccounting { + count: current.part_count, + staged_bytes: current.staged_bytes, + }, + before.as_ref().map(|record| record.length), + after.length, + current.max_parts, + current.max_staged_bytes, + ) + .map_err(|_| MultipartRepositoryError::InvalidPart)?; + if !self.persist_generation(current, &after).await? { + return Ok(None); + } + let mut reserved = current.clone(); + reserved.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; + reserved.part_count = accounting.count; + reserved.staged_bytes = accounting.staged_bytes; + let before_digest = before + .as_ref() + .map(|record| record.encode().map(|value| Sha256::digest(value).into())) + .transpose()?; + reserved.pending = Some(PendingPartMutation { + number: after.number, + before_revision: before.as_ref().map(|record| record.revision), + before_digest, + after_revision: after.revision, + after_digest: Sha256::digest(after.encode()?).into(), + after_length: after.length, + }); + if !self.exchange(current, &reserved).await? { + return Ok(None); + } + self.settle_pending_part(&reserved).await + } + + async fn persist_generation( + &self, + session: &MultipartSessionRecord, + after: &MultipartPartRecord, + ) -> Result { + let key = MetadataKey::multipart_part_generation( + &self.tenant, + session.bucket_id, + &session.upload_id, + after.number, + after.revision, + )?; + let value = after.encode()?; + match self.store.put_if_absent(key.clone(), value.clone()).await { + Ok(PutIfAbsentOutcome::Inserted { .. }) => Ok(true), + Ok(PutIfAbsentOutcome::Existing(existing)) if existing.value == value => Ok(true), + Ok(PutIfAbsentOutcome::Existing(_)) => Ok(false), + Err(error) => { + if self + .store + .get(key) + .await? + .is_some_and(|entry| entry.value == value) + { + Ok(true) + } else { + Err(error.into()) + } + } + } + } + + /// Helps a reserved pointer mutation after response loss or restart. + /// + /// # Errors + /// Rejects changed immutable evidence or an unconfirmed metadata write. + pub async fn settle_pending_part( + &self, + session: &MultipartSessionRecord, + ) -> Result, MultipartRepositoryError> { + let current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current != *session { + return Ok(None); + } + let pending = current + .pending + .as_ref() + .ok_or(MultipartRepositoryError::Conflict)?; + let after = self + .part_generation(¤t, pending.number, pending.after_revision) + .await? + .ok_or(MultipartRepositoryError::InvalidPart)?; + let digest: [u8; 32] = Sha256::digest(after.encode()?).into(); + if after.length != pending.after_length || digest != pending.after_digest { + return Err(MultipartRepositoryError::InvalidPart); + } + self.publish_reserved_pointer(¤t, pending, &after).await?; + let mut settled = current.clone(); + settled.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; + settled.pending = None; + if self.exchange(¤t, &settled).await? { + return Ok(Some(after)); + } + let latest = self + .load(¤t) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if latest.pending.is_none() && self.part(&latest, after.number).await?.as_ref() == Some(&after) { + Ok(Some(after)) + } else { + Ok(None) + } + } + + async fn publish_reserved_pointer( + &self, + session: &MultipartSessionRecord, + pending: &PendingPartMutation, + after: &MultipartPartRecord, + ) -> Result<(), MultipartRepositoryError> { + let current = self.part(session, pending.number).await?; + if current.as_ref() == Some(after) { + return Ok(()); + } + let before_digest = current + .as_ref() + .map(|record| record.encode().map(|value| Sha256::digest(value).into())) + .transpose()?; + if current.as_ref().map(|record| record.revision) != pending.before_revision + || before_digest != pending.before_digest + { + return Err(MultipartRepositoryError::InvalidPart); + } + let key = MetadataKey::multipart_part( + &self.tenant, + session.bucket_id, + &session.upload_id, + pending.number, + )?; + let value = after.encode()?; + let mutation = match current { + Some(before) => { + self.store + .compare_exchange(key.clone(), before.encode()?, value.clone()) + .await + } + None => self + .store + .put_if_absent(key.clone(), value.clone()) + .await + .map(|outcome| matches!(outcome, PutIfAbsentOutcome::Inserted { .. })), + }; + match mutation { + Ok(true) => Ok(()), + Ok(false) => { + if self + .store + .get(key) + .await? + .is_some_and(|entry| entry.value == value) + { + Ok(()) + } else { + Err(MultipartRepositoryError::InvalidPart) + } + } + Err(error) => { + if self + .store + .get(key) + .await? + .is_some_and(|entry| entry.value == value) + { + Ok(()) + } else { + Err(error.into()) + } + } + } + } +} diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs index 0f9cffe9..76a52fcf 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/publication.rs @@ -3,7 +3,7 @@ //! Publish the frozen multipart object through one object-key mutation. -use crowdb_access_multipart::MultipartComposer; +use crowdb_access_multipart::{next_revision, MultipartComposer}; use sha2::{Digest as _, Sha256}; use super::{ @@ -39,6 +39,15 @@ impl MultipartRepository { .ok_or(MultipartRepositoryError::Conflict)?; let mut composer = MultipartComposer::new(current.max_object_bytes); for selected in selection { + if self + .part(¤t, selected.number) + .await? + .as_ref() + .map(|part| part.revision) + != Some(selected.revision) + { + return Err(MultipartRepositoryError::InvalidPart); + } let part = self .part_generation(¤t, selected.number, selected.revision) .await? @@ -115,10 +124,7 @@ impl MultipartRepository { current: &MultipartSessionRecord, ) -> Result { let mut next = current.clone(); - next.revision = current - .revision - .checked_add(1) - .ok_or(MultipartRepositoryError::Conflict)?; + next.revision = next_revision(current.revision).ok_or(MultipartRepositoryError::Conflict)?; next.phase = MultipartPhase::Published; if self.exchange(current, &next).await? { Ok(next) diff --git a/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs index 38e5f98f..1d09f075 100644 --- a/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs +++ b/lib/crowdb-access-s3/src/metadata/multipart_repository/terminal.rs @@ -100,10 +100,20 @@ impl MultipartRepository { &self, session: &MultipartSessionRecord, ) -> Result { - let current = self + let mut current = self .load(session) .await? .ok_or(MultipartRepositoryError::Conflict)?; + if current.pending.is_some() { + self.settle_pending_part(¤t).await?; + current = self + .load(session) + .await? + .ok_or(MultipartRepositoryError::Conflict)?; + if current.pending.is_some() { + return Err(MultipartRepositoryError::Busy); + } + } if current.phase == MultipartPhase::Aborted { return Ok(current); } diff --git a/lib/crowdb-access-s3/tests/metadata_multipart_test.rs b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs index c5a03bf8..00aecd01 100644 --- a/lib/crowdb-access-s3/tests/metadata_multipart_test.rs +++ b/lib/crowdb-access-s3/tests/metadata_multipart_test.rs @@ -25,6 +25,7 @@ fn session() -> MultipartSessionRecord { max_staged_bytes: 1_000, part_count: 0, staged_bytes: 0, + pending: None, selection: None, completion_request_digest: None, publication_ms: None, diff --git a/lib/crowdb-access-s3/tests/multipart_repository_test.rs b/lib/crowdb-access-s3/tests/multipart_repository_test.rs index a7785557..07bf1b95 100644 --- a/lib/crowdb-access-s3/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-s3/tests/multipart_repository_test.rs @@ -10,7 +10,8 @@ use async_trait::async_trait; use crowdb_access_multipart::SelectedPart; use crowdb_access_s3::metadata::{ BucketId, ChunkKvMetadataStore, CompletionPart, MetadataKey, MultipartPartRecord, MultipartPhase, - MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, ObjectRecord, TenantId, + MultipartRepository, MultipartRepositoryError, MultipartSessionRecord, ObjectRecord, PendingPartMutation, + TenantId, }; use crowdb_chunk_kv_client::{ ChunkKvClient, ChunkKvRangeCatalogSource, ChunkKvTransport, ClientConfig, ClientError, Result, @@ -24,6 +25,7 @@ use crowdb_protocol::chunk_kv::{ use crowdb_protocol::chunk_stream::StreamName; use crowdb_protocol::chunkdb::rpc::Location; use crowdb_protocol::common::ChunkId; +use sha2::{Digest as _, Sha256}; use tokio::sync::{mpsc, oneshot}; struct Catalog(ChunkKvRangeCatalogHead, Vec); @@ -250,6 +252,7 @@ fn session() -> MultipartSessionRecord { max_staged_bytes: 1_000, part_count: 0, staged_bytes: 0, + pending: None, selection: None, completion_request_digest: None, publication_ms: None, @@ -326,8 +329,9 @@ async fn session_cas_and_independent_part_replacement_obey_the_freeze() { Some(first) ); - let mut frozen = session.clone(); - frozen.revision = 2; + let current = repository.load(&session).await.unwrap().unwrap(); + let mut frozen = current.clone(); + frozen.revision += 1; frozen.phase = MultipartPhase::Completing; frozen.part_count = 1; frozen.staged_bytes = 5; @@ -339,8 +343,8 @@ async fn session_cas_and_independent_part_replacement_obey_the_freeze() { frozen.completion_request_digest = Some([5; 32]); frozen.publication_ms = None; lose_reply.store(true, Ordering::SeqCst); - assert!(repository.exchange(&session, &frozen).await.unwrap()); - assert!(repository.exchange(&session, &frozen).await.unwrap()); + assert!(repository.exchange(¤t, &frozen).await.unwrap()); + assert!(repository.exchange(¤t, &frozen).await.unwrap()); assert!(matches!( repository.put_stream_part(&session, &part(), 112).await, Err(MultipartRepositoryError::Conflict) @@ -397,6 +401,76 @@ async fn completion_freezes_exact_part_revision_and_replays_the_same_request() { assert_eq!(locations, part().locations); } +#[tokio::test] +async fn completion_settles_an_interrupted_part_reservation_before_freezing() { + let (repository, _, store) = repository().await; + let session = session(); + repository.begin(&session).await.unwrap(); + let first = repository + .put_stream_part(&session, &part(), 110) + .await + .unwrap() + .unwrap(); + let mut replacement = first.clone(); + replacement.revision += 1; + replacement.modified_ms = 111; + replacement.raw_md5 = [8; 16]; + replacement.locations[0].offset += 39; + let generation_key = MetadataKey::multipart_part_generation( + &TenantId::new(b"tenant".to_vec()).unwrap(), + session.bucket_id, + &session.upload_id, + 1, + replacement.revision, + ) + .unwrap(); + store + .put_if_absent(generation_key, replacement.encode().unwrap()) + .await + .unwrap(); + let current = repository.load(&session).await.unwrap().unwrap(); + let mut reserved = current.clone(); + reserved.revision += 1; + reserved.pending = Some(PendingPartMutation { + number: 1, + before_revision: Some(first.revision), + before_digest: Some(Sha256::digest(first.encode().unwrap()).into()), + after_revision: replacement.revision, + after_digest: Sha256::digest(replacement.encode().unwrap()).into(), + after_length: replacement.length, + }); + assert!(repository.exchange(¤t, &reserved).await.unwrap()); + let request = [CompletionPart { + number: 1, + etag: "08".repeat(16), + }]; + assert!(repository + .freeze_completion(&session, &request, 120) + .await + .unwrap() + .is_none()); + assert_eq!( + repository.part(&session, 1).await.unwrap(), + Some(replacement.clone()) + ); + assert!(repository + .load(&session) + .await + .unwrap() + .unwrap() + .pending + .is_none()); + let frozen = repository + .freeze_completion(&session, &request, 120) + .await + .unwrap() + .unwrap(); + assert_eq!( + frozen.selection.as_ref().unwrap()[0].revision, + replacement.revision + ); +} + #[tokio::test] async fn completion_rejects_undersized_nonfinal_and_wrong_etag() { let (repository, _, _) = repository().await; @@ -426,7 +500,10 @@ async fn completion_rejects_undersized_nonfinal_and_wrong_etag() { repository.freeze_completion(&session, &wrong, 120).await, Err(MultipartRepositoryError::InvalidPart) )); - assert_eq!(repository.load(&session).await.unwrap(), Some(session)); + let current = repository.load(&session).await.unwrap().unwrap(); + assert_eq!(current.phase, MultipartPhase::Open); + assert_eq!(current.part_count, 2); + assert!(current.pending.is_none()); } #[tokio::test] @@ -483,7 +560,7 @@ async fn publication_recovers_a_lost_reply_without_replacing_a_competing_object( } #[tokio::test] -async fn frozen_part_generation_survives_a_late_pointer_change() { +async fn frozen_part_generation_rejects_a_late_pointer_change() { let (repository, _, store) = repository().await; let session = session(); repository.begin(&session).await.unwrap(); @@ -520,8 +597,17 @@ async fn frozen_part_generation_survives_a_late_pointer_change() { .compare_exchange(pointer_key, previous.value, late.encode().unwrap()) .await .unwrap()); - let published = repository.publish_completion(&frozen).await.unwrap(); - assert_eq!(published.phase, MultipartPhase::Published); + assert!(matches!( + repository.publish_completion(&frozen).await, + Err(MultipartRepositoryError::InvalidPart) + )); + let object_key = MetadataKey::object( + &TenantId::new(b"tenant".to_vec()).unwrap(), + session.bucket_id, + &session.object_key, + ) + .unwrap(); + assert!(store.get(object_key).await.unwrap().is_none()); } fn part_for(session: &MultipartSessionRecord) -> MultipartPartRecord { diff --git a/lib/crowdb-access-s3/tests/wire_test.rs b/lib/crowdb-access-s3/tests/wire_test.rs index e04f3481..2c189547 100644 --- a/lib/crowdb-access-s3/tests/wire_test.rs +++ b/lib/crowdb-access-s3/tests/wire_test.rs @@ -99,6 +99,7 @@ fn multipart_upload_listing_emits_stable_markers_and_initiation_time() { max_staged_bytes: 5, part_count: 0, staged_bytes: 0, + pending: None, selection: None, completion_request_digest: None, publication_ms: None, From 686d60a191632a083f31d47c8c9e3239865ac4fc Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 09:54:35 +0800 Subject: [PATCH 39/74] Retire completed S3 multipart requirement --- doc/backlog/R167-s3-multipart-upload.md | 100 -------------- doc/backlog/R170-s3-cuobject-rdma.md | 4 +- doc/backlog/R95-chunkdb-chunk-range-delete.md | 14 +- doc/backlog/backlog.md | 9 +- doc/working/plan-s3-multipart.md | 124 ------------------ 5 files changed, 13 insertions(+), 238 deletions(-) delete mode 100644 doc/backlog/R167-s3-multipart-upload.md delete mode 100644 doc/working/plan-s3-multipart.md diff --git a/doc/backlog/R167-s3-multipart-upload.md b/doc/backlog/R167-s3-multipart-upload.md deleted file mode 100644 index e55a33e8..00000000 --- a/doc/backlog/R167-s3-multipart-upload.md +++ /dev/null @@ -1,100 +0,0 @@ - - - -### R167: access server / S3 — Multipart upload - -## Status - -**Deferred until the R152–R166 basic S3 milestone is complete.** -It is unblocked when single-request streaming, publication recovery, ETag, -listing, deletion, and compatibility E2E behavior are stable. - -## Problem - -Multipart upload adds durable upload/part listing, independent part retries, -completion ordering, abort cleanup, and multipart ETag semantics. Adding it -before the basic publication and cleanup paths stabilize would duplicate -unsettled recovery rules and delay the deliberately limited first service. - -The scope boundary is -`doc/design/access-server/s3/design-crowdb-access-s3.md` §1. - -## Solution - -1. Add create, upload-part, list-parts, complete, abort, and required upload - listing operations through S3-owned API and namespace adapters over the - protocol-neutral multipart session, part, and completion core from R190. -2. Store immutable part identities and integrity records durably under a common - upload key prefix; a retried part number replaces only that part's selected - generation. Keep old generations discoverable for R95's chunk-centered scan - without writing MPU cleanup records. -3. Complete with one fenced metadata transaction that validates ordered part - identities, sizes, checksums, and expected upload state before publishing - one immutable object generation. Compose the selected parts' chunk-location - arrays with adjusted logical offsets. Complete does not read part data or - concatenate it through access-server memory. -4. Persist each uploaded part's raw 16-byte MD5. The multipart ETag is the - lowercase hexadecimal MD5 of the selected parts' raw MD5 bytes in order, - followed by `-`. Keep the basic single-part ETag rule unchanged. - Abort and expiry mark the upload terminal or expired in durable metadata; - R95 later qualifies unreachable bytes after its age and reference checks. -5. Preserve the basic admission bounds for parallel part traffic and wire - compatibility for all retry and conflict outcomes. - -## Dependencies - -- Depends on R152–R166. -- Reuses the basic milestone's publication/recovery, streaming input, logical - deletion, and integrity contracts. -- Reuses R190's protocol-neutral multipart core; S3 retains its own - authorization, namespace, wire errors, ETag response, and object publication. -- R95 owns eventual chunk-centered reclamation, including unused MPU part - locations; R167 does not add a per-upload cleanup queue. -- R170 owns any accelerated multipart transfer and additionally depends on this - requirement before enabling that operation. - -## Acceptance - -- Given parts uploaded out of order with part retries, when completion names a - valid order, assert exact concatenated bytes become visible through one - generation without gateway concatenation or part reads. Invariant: completion - is metadata-only, atomic and storage-backed. E2E test. -- Given completed parts with known MD5 values, when completion selects and - reorders them, assert the ETag uses only the selected raw part MD5 values in - completion order and the part count suffix. Invariant: multipart ETag matches - the S3-compatible composite algorithm. Unit test. -- Given missing, duplicated, undersized, checksum-mismatched, or concurrently - replaced parts, when completion runs, assert no object publishes and exact - errors are stable. Invariant: only the validated part set can publish. - Integration test. -- Given response loss during part upload, complete, and abort, when identities - retry after restart, assert the durable outcome is returned without duplicate - generations. Invariant: every multipart transition is idempotent. - E2E test. -- Given abort, expiry and part replacement, when upload metadata is scanned, - assert current and immutable part generations remain discoverable under the - upload prefix for R95 and no MPU cleanup record is created. Invariant: R167 - preserves reference evidence without deciding physical reclamation. - Integration test. - -Required gates: - -- `pixi run -- cargo test -p crowdb-access-s3 --all-targets` -- `pixi run -- cargo test -p crowdb-access-server --all-targets` -- `pixi run -- cargo fmt --all -- --check` -- `pixi run rs-lint` - -## Open Issues - -- The referenced R190 requirement is no longer present in the backlog. Both - adapters now use shared phase names, selected-part validation, accounting and - metadata-only location composition. Iceberg's remaining session and part - recovery is bound to its catalog identity and store. Extract the remaining - protocol-neutral transition decisions while keeping keys, authorization and - responses in the protocol adapters. -- A part writer can load `Open` before completion freezes the session and - advance the current-part pointer afterward. Publication uses the immutable - frozen generation, so the published bytes are stable, but the current pointer - may differ at publication time. The acceptance rule above requires a - concurrent replacement to prevent publication; add a session fence to part - replacement or define an explicit snapshot boundary, then test the race. diff --git a/doc/backlog/R170-s3-cuobject-rdma.md b/doc/backlog/R170-s3-cuobject-rdma.md index 768341e8..8a09a6d5 100644 --- a/doc/backlog/R170-s3-cuobject-rdma.md +++ b/doc/backlog/R170-s3-cuobject-rdma.md @@ -91,8 +91,8 @@ Client/GPU AccessServer Chunk plan DiskIO A..D publication, range, integrity, or error contracts. - Uses `crowdb-chunk-client`, `crowdb-diskio`, internal authenticated RPC, and native ownership/completion support from `crowdb-rpc-ffi`. -- Accelerated multipart upload additionally depends on R167 and remains - disabled until both requirements land. +- Accelerated multipart upload builds on the existing S3 multipart authority + and remains disabled until this acceleration requirement lands. - NVIDIA cuObject server libraries, compatible drivers, and ConnectX-5-or-newer hardware are optional deployment dependencies. Unsupported deployments keep the basic TCP service unchanged. diff --git a/doc/backlog/R95-chunkdb-chunk-range-delete.md b/doc/backlog/R95-chunkdb-chunk-range-delete.md index c03a7b76..9839ad96 100644 --- a/doc/backlog/R95-chunkdb-chunk-range-delete.md +++ b/doc/backlog/R95-chunkdb-chunk-range-delete.md @@ -29,10 +29,11 @@ add metadata writes to the upload path and still miss those pre-record crashes. unused. Refuse deletion when reference or reader state cannot be confirmed. 4. Keep S3 MPU session, current-part and immutable part-generation keys under a common upload prefix in Chunk-KV so the scanner can enumerate one upload's - references with a bounded prefix scan. R167 owns that key layout and the - metadata-only Abort/expiry transition; old part generations remain available - until the scanner proves their bytes are unreachable. An aborted or expired - upload becomes a candidate only after the age gate and reference checks. + references with a bounded prefix scan. The S3 multipart authority owns that + key layout and the metadata-only Abort/expiry transition; old part + generations remain available until the scanner proves their bytes are + unreachable. An aborted or expired upload becomes a candidate only after + the age gate and reference checks. 5. Recheck chunk identity, layout generation and exact range against current authority immediately before reclaim. Treat lost delete replies idempotently and keep an in-memory scan cursor and bounded work budget. @@ -42,8 +43,9 @@ add metadata writes to the upload path and still miss those pre-record crashes. ## Dependencies - R92 supplies in-chunk strip reclamation after R95 qualifies dead ranges. -- R167 supplies grouped MPU keys and durable session, part and completion - references. The scanner also recognizes Iceberg's frozen MPU selection. +- The S3 multipart authority supplies grouped MPU keys and durable session, + part and completion references. The scanner also recognizes Iceberg's frozen + MPU selection. - Reader protection and published generation references must be queryable before physical deletion is enabled; R168 may use the qualified range-delete interface for ordinary S3 object deletion. diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index 4ff88de4..3d67512e 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -37,12 +37,9 @@ requirement is implemented. ### Planned — S3 data access service R152–R166 delivered the limited basic S3 service, including the restart -acceptance baseline. R167–R169 defer multipart upload and shared-storage GC -without blocking basic large-object deletion. R170 separately adds optional -cuObject/RDMA acceleration after the TCP baseline is correct and measured. -- **[R167](R167-s3-multipart-upload.md)** — multipart upload — Area: access - server / S3 — **Deferred.** Add durable part state, atomic completion, cleanup, - and multipart integrity after the basic milestone stabilizes. +acceptance baseline. Multipart upload is available; R168–R169 defer +shared-storage GC without blocking basic large-object deletion. R170 adds +optional cuObject/RDMA acceleration after the TCP baseline is correct and measured. - **[R168](R168-s3-shared-object-reclamation.md)** — shared small-object reclamation — Area: access server / S3 / chunkdb — **Deferred on R95.** Turn exact pending shared ranges into qualified, restart-safe physical deletion. diff --git a/doc/working/plan-s3-multipart.md b/doc/working/plan-s3-multipart.md deleted file mode 100644 index fdd6b71b..00000000 --- a/doc/working/plan-s3-multipart.md +++ /dev/null @@ -1,124 +0,0 @@ - - - -# S3 Multipart Plan - -Upstream: [R167](../backlog/R167-s3-multipart-upload.md). - -Goal: add durable S3 multipart uploads while sharing protocol-neutral part -composition and state transitions with Iceberg. - -Status: active. The basic S3 milestone is complete; the protocol-neutral R190 -core referenced by R167 is absent, so extraction begins from the working -Iceberg multipart path. - -## Shared foundation - -- [x] **Metadata-only composition**: extract offset adjustment and composite - MD5 ETag from Iceberg `MultipartRepository::prepare_stream_publication` into - `lib/crowdb-access-multipart`, preserving the existing selected-part fences - in Iceberg. Include overflow and malformed location tests. Files: new crate, - Iceberg publication, workspace manifests. -- [x] **Durable transition core**: isolate session/part states, replacement - generations, completion selection, abort and recovery transitions from - Iceberg catalog-specific keys and records. Keep store CAS and namespace - adaptation in each protocol. Shared phase vocabulary, admission bounds, - selection validation, accounting and location composition are now used by - both adapters; storage CAS and durable record layouts remain protocol-specific. - Files: shared - multipart crate, Iceberg file repository, S3 metadata store. Both adapters - use the same inclusive/exclusive lifetime decision and checked session/part - revision advancement. Iceberg's byte assembly checkpoints and S3's - metadata-only object publication remain in their adapters because their - durable evidence and work units differ. - -## S3 adapter and HTTP - -- [x] **Durable S3 records**: add upload and part keys/records with raw 16-byte - MD5 and selected revision under one upload prefix. Use bucket identity and - object key as namespace scope; preserve immutable part data after replacement. - Versioned session/part records and ordered, binary-safe keys are in place; - CAS-backed begin, phase transition and part replacement now use exact-value - confirmation after lost replies. An identical part record retry returns the - existing revision; a new location remains a replacement. Completion - snapshots and a predecessor-fenced metadata-only object publication path are - in place. The session, current part and immutable generations now share one - upload prefix for R95; an immutable object-key/upload-ID index preserves - bounded ListMultipartUploads ordering. HTTP wiring remains. -- [x] **S3 routes and wire**: classify create/upload/list/complete/abort/list - uploads, parse bounded completion XML, emit compatible responses and errors. - Preserve SigV4 authentication and existing basic routes. The repository now - provides bounded, ordered ListParts pagination over current generations; - multipart query shapes now enter the authenticated dispatcher with distinct - metrics. Bounded - upload listing now - paginates active sessions by key and upload ID, skipping terminal/expired - records and failing on scan-budget exhaustion; HTTP dispatch remains pending. - Upload IDs now sort by initiation millisecond. S3-compatible multipart error - codes and the create, complete, ListParts, and ListMultipartUploads XML - response builders have focused tests. The bounded completion - XML parser now has one implementation in access-server and is exposed by both - the Iceberg and S3 protocol modules. The six S3 HTTP operations are wired. - Duplicate or descending completion parts now map to S3 `InvalidPartOrder`; - malformed XML remains `InvalidRequest`. - Full boto3 stack acceptance passes Create, UploadPart, ListParts, Complete, - Abort and ListUploads, including replay, replacement and invalid ETag cases. -- [x] **Part ingestion**: reuse the bounded streaming writer and admission - budget, persist part location/integrity before success, reconcile lost replies. - Production UploadPart now uses the basic streaming writer and saves raw MD5 - with locations. A byte-identical retry keeps the selected generation even - when a new write produced different chunk locations; R95 can reclaim those - unreachable chunks. Full stack ingestion passes 5 MiB and small parts, a - dropped UploadPart success response, and retries after service restart. -- [x] **Atomic completion**: fence selected part generations, validate order, - count, size and checksum, compose locations through the shared core, and - publish one immutable object generation without reading part bytes. The S3 - adapter now freezes selection under session CAS, validates it again before - object-key CAS, and confirms exact publication after response loss. An - immutable generation records preserve selected bytes across a concurrent - part-number replacement. Metadata-only completion and byte-exact GET pass - the full boto3 stack, including a repeated Complete request. -- [x] **Part publication fence**: reserve each changed current-part pointer - under session CAS, persist the immutable generation, settle the pointer and - clear the reservation. Complete and Abort must not pass an unresolved - reservation. A helper can finish a committed reservation after reply loss or - restart. Test a replacement racing with freeze, then test recovery at every - durable boundary. Keep this CAS-based path lock-free and retain orphan - generations for R95. A focused test freezes an interrupted reservation only - after recovery, and a changed pointer prevents object publication. -- [x] **Abort and expiry**: make terminal states idempotent and preserve the - part generations that R95's chunk-centered scanner needs for reference checks. - The S3 adapter now has an idempotent, response-loss-safe logical abort and a - bounded hourly expiry sweep over the upload listing index. Session and part - records remain under one upload prefix after terminal transition. Focused - expiry pagination and evidence tests pass. Requests encountering an expired - open session terminate it before the hourly sweep. R95 owns physical cleanup. - -## Acceptance and cleanup - -- [x] **Focused and E2E tests**: known MD5 vectors, out-of-order/replaced parts, - invalid completion, response loss and restart, abort/expiry metadata, and - ordinary single-part compatibility. The complete S3 library and access-server - suites plus 19 full-stack boto3/restart cases pass. The new cases drop - UploadPart, Complete and Abort success replies, replay them, preserve an - incomplete session across six service restarts, then complete it. Hourly - expiry has a focused metadata test; broader concurrency and error-matrix - acceptance remains. -- [x] **Gates and docs**: run both access crate suites, access-server E2E, - Rust fmt and clippy separately; update S3 design and remove R167 plus this - plan only after all acceptance criteria pass. - -## Files - -- `Cargo.toml`, `Cargo.lock`, `lib/crowdb-access-multipart/**` -- `lib/crowdb-access-iceberg/src/file/multipart_repository/**` -- `lib/crowdb-access-s3/src/{metadata,route,integrity,wire}.rs` and children -- `app/crowdb-access-server/src/s3/**` -- `lib/crowdb-access-s3/tests/**`, `app/crowdb-access-server/tests/**` - -## Tests - -- Unit: shared composition MD5, location offsets and bounds. -- Integration: durable S3 session/part/complete/abort transitions and Iceberg - publication regression. -- E2E: official S3 client multipart flows, response loss and restart. From 439f40e7ebc506025da11ac742be34cfc15a3db5 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 10:33:34 +0800 Subject: [PATCH 40/74] Retain private container crash diagnostics --- container/crowdb-monitor/src/crash.rs | 70 ++++++++++ container/crowdb-monitor/src/lib.rs | 2 + container/crowdb-monitor/src/preview.rs | 14 +- container/crowdb-monitor/src/process.rs | 16 ++- .../tests/crash_retention_test.rs | 70 ++++++++++ container/single-node-container/README.md | 46 +++++-- container/single-node-container/entrypoint.sh | 17 +++ .../tests/container-e2e.sh | 4 + doc/backlog/R188-console-group0-authority.md | 24 ++-- doc/working/plan-console-authority.md | 27 ++-- tools/symbolize-container-core.py | 123 ++++++++++++++++++ 11 files changed, 379 insertions(+), 34 deletions(-) create mode 100644 container/crowdb-monitor/src/crash.rs create mode 100644 container/crowdb-monitor/tests/crash_retention_test.rs create mode 100644 tools/symbolize-container-core.py diff --git a/container/crowdb-monitor/src/crash.rs b/container/crowdb-monitor/src/crash.rs new file mode 100644 index 00000000..f0715df4 --- /dev/null +++ b/container/crowdb-monitor/src/crash.rs @@ -0,0 +1,70 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Private retention for file-based Linux core dumps. + +use std::ffi::OsStr; +use std::fs; +use std::io; +use std::os::unix::fs::DirBuilderExt; +use std::os::unix::fs::PermissionsExt; +use std::path::{Path, PathBuf}; +use std::time::SystemTime; + +pub struct CrashRetention { + root: PathBuf, +} + +impl CrashRetention { + /// Opens a private core directory and removes all but its newest core. + /// + /// # Errors + /// Rejects symlinked roots and inaccessible directories or entries. + pub fn open(root: PathBuf) -> io::Result { + match fs::DirBuilder::new().mode(0o700).create(&root) { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {} + Err(error) => return Err(error), + } + if !fs::symlink_metadata(&root)?.file_type().is_dir() { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "core root is not a directory", + )); + } + fs::set_permissions(&root, fs::Permissions::from_mode(0o700))?; + let retention = Self { root }; + retention.prune()?; + Ok(retention) + } + + #[must_use] + pub fn root(&self) -> &Path { + &self.root + } + + /// Keeps the newest regular `core` or `core.*` file only. + /// + /// # Errors + /// Returns a directory or removal error without exposing dump contents. + pub fn prune(&self) -> io::Result<()> { + let mut cores = Vec::new(); + for entry in fs::read_dir(&self.root)? { + let entry = entry?; + if !core_name(&entry.file_name()) || !entry.file_type()?.is_file() { + continue; + } + let modified = entry.metadata()?.modified().unwrap_or(SystemTime::UNIX_EPOCH); + cores.push((modified, entry.file_name(), entry.path())); + } + cores.sort_by(|left, right| right.0.cmp(&left.0).then_with(|| right.1.cmp(&left.1))); + for (_, _, path) in cores.into_iter().skip(1) { + fs::remove_file(path)?; + } + Ok(()) + } +} + +fn core_name(name: &OsStr) -> bool { + name == "core" || name.as_encoded_bytes().starts_with(b"core.") +} diff --git a/container/crowdb-monitor/src/lib.rs b/container/crowdb-monitor/src/lib.rs index cdb72eba..ba2b2db8 100644 --- a/container/crowdb-monitor/src/lib.rs +++ b/container/crowdb-monitor/src/lib.rs @@ -2,6 +2,7 @@ // Licensed under the Apache License, Version 2.0. mod bootstrap; +mod crash; mod credentials; mod layout; mod liveness; @@ -22,6 +23,7 @@ pub use bootstrap::{ KvBootstrap, KvBootstrapError, LogicalBootstrap, LogicalBootstrapError, S3Bootstrap, S3BootstrapError, StorageProbeError, }; +pub use crash::CrashRetention; pub use credentials::{show_client_credentials, ClientCredentials, CredentialError, ServerCredentials}; pub use liveness::{probe_liveness, LivenessError, LivenessServer}; pub use manifest::{BootstrapManifest, BootstrapSession, ManifestError, ManifestState}; diff --git a/container/crowdb-monitor/src/preview.rs b/container/crowdb-monitor/src/preview.rs index 67686a68..7dc8b7d0 100644 --- a/container/crowdb-monitor/src/preview.rs +++ b/container/crowdb-monitor/src/preview.rs @@ -10,11 +10,11 @@ use tokio::time::{sleep, Instant}; use crate::{ disk_step_names, ensure_disk_files, hardware_step_names, iceberg_step_names, kv_step_names, logical_step_names, render_configs, s3_step_names, verify_chunk_services, verify_diskio_disks, - BootstrapSession, ChunkBootstrapError, CredentialError, DeploymentProfile, DiskBootstrapError, - HardwareBootstrap, HardwareBootstrapError, IcebergBootstrap, IcebergBootstrapError, KvBootstrap, - KvBootstrapError, LivenessError, LivenessServer, LogicalBootstrap, LogicalBootstrapError, ManifestError, - ManifestState, MonitorEvent, MonitorEventKind, MonitorLogError, ProfileError, RenderError, S3Bootstrap, - S3BootstrapError, ServerCredentials, StorageProbeError, Supervisor, SupervisorError, + BootstrapSession, ChunkBootstrapError, CrashRetention, CredentialError, DeploymentProfile, + DiskBootstrapError, HardwareBootstrap, HardwareBootstrapError, IcebergBootstrap, IcebergBootstrapError, + KvBootstrap, KvBootstrapError, LivenessError, LivenessServer, LogicalBootstrap, LogicalBootstrapError, + ManifestError, ManifestState, MonitorEvent, MonitorEventKind, MonitorLogError, ProfileError, RenderError, + S3Bootstrap, S3BootstrapError, ServerCredentials, StorageProbeError, Supervisor, SupervisorError, }; const PROFILE_NAME: &str = "single-node-container"; @@ -83,6 +83,10 @@ pub async fn run_preview(profile_path: &Path) -> Result<(), PreviewError> { &config_bytes, &step_refs, )?; + if let Some(root) = std::env::var_os("CROWDB_CORE_DIR") { + let crashes = CrashRetention::open(root.into())?; + std::env::set_current_dir(crashes.root())?; + } let credentials = if session.manifest().state() == ManifestState::Ready || session.manifest().step_complete("s3-user") == Some(true) { diff --git a/container/crowdb-monitor/src/process.rs b/container/crowdb-monitor/src/process.rs index e8b92c1a..9a6fd431 100644 --- a/container/crowdb-monitor/src/process.rs +++ b/container/crowdb-monitor/src/process.rs @@ -13,7 +13,9 @@ use tokio::process::{Child, Command}; use tokio::task::JoinHandle; use tokio::time::{sleep, timeout}; -use crate::{LogProfile, MonitorEvent, MonitorEventKind, MonitorLog, MonitorLogError, ServiceProfile}; +use crate::{ + CrashRetention, LogProfile, MonitorEvent, MonitorEventKind, MonitorLog, MonitorLogError, ServiceProfile, +}; #[derive(Debug, Error)] pub enum ProcessError { @@ -37,12 +39,17 @@ pub struct ProcessManager { log_root: PathBuf, log_policy: LogProfile, events: MonitorLog, + crashes: Option, } impl ProcessManager { /// # Errors /// Rejects an unavailable monitor lifecycle log. pub async fn new(log_root: PathBuf, log_policy: LogProfile) -> Result { + let crashes = std::env::var_os("CROWDB_CORE_DIR") + .map(PathBuf::from) + .map(CrashRetention::open) + .transpose()?; let mut events = MonitorLog::open(&log_root, log_policy.clone()).await?; events .record(&MonitorEvent { @@ -57,6 +64,7 @@ impl ProcessManager { log_root, log_policy, events, + crashes, }) } @@ -74,6 +82,9 @@ impl ProcessManager { std::fs::create_dir_all(&log_directory)?; retention::prune(&log_directory, None, self.log_policy.max_files.saturating_sub(1)).await?; let mut command = Command::new(&service.program); + if let Some(crashes) = &self.crashes { + command.current_dir(crashes.root()); + } command .args(&service.args) .envs(&service.env) @@ -191,6 +202,9 @@ impl ProcessManager { }) .await?; retention::prune(&self.log_root.join(id), None, self.log_policy.max_files).await?; + if let Some(crashes) = &self.crashes { + crashes.prune()?; + } Ok(()) } diff --git a/container/crowdb-monitor/tests/crash_retention_test.rs b/container/crowdb-monitor/tests/crash_retention_test.rs new file mode 100644 index 00000000..89874611 --- /dev/null +++ b/container/crowdb-monitor/tests/crash_retention_test.rs @@ -0,0 +1,70 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::fs::{self, File, FileTimes}; +use std::os::unix::fs::{symlink, PermissionsExt}; +use std::path::{Path, PathBuf}; +use std::time::{Duration, SystemTime}; + +use crowdb_monitor::CrashRetention; +use uuid::Uuid; + +struct TestRoot(PathBuf); + +impl TestRoot { + fn new() -> Self { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../.crowdb-runtime/ephemeral") + .join(format!("monitor-crash-{}", Uuid::new_v4())); + fs::create_dir_all(&root).unwrap(); + Self(root) + } +} + +impl Drop for TestRoot { + fn drop(&mut self) { + if !std::thread::panicking() { + fs::remove_dir_all(&self.0).unwrap(); + } + } +} + +fn core(root: &Path, name: &str, seconds: u64) { + let file = File::create(root.join(name)).unwrap(); + file.set_times(FileTimes::new().set_modified(SystemTime::UNIX_EPOCH + Duration::from_secs(seconds))) + .unwrap(); +} + +#[test] +fn latest_core_survives_startup_and_recovery_without_touching_other_files() { + let test = TestRoot::new(); + let root = test.0.join("crash"); + fs::create_dir(&root).unwrap(); + fs::set_permissions(&root, fs::Permissions::from_mode(0o755)).unwrap(); + core(&root, "core.10", 10); + core(&root, "core.11", 11); + core(&root, "notes", 12); + symlink(root.join("notes"), root.join("core.link")).unwrap(); + + let retention = CrashRetention::open(root.clone()).unwrap(); + assert_eq!(fs::metadata(&root).unwrap().permissions().mode() & 0o777, 0o700); + assert!(!root.join("core.10").exists()); + assert!(root.join("core.11").exists()); + assert!(root.join("notes").exists()); + assert!(root.join("core.link").exists()); + + core(&root, "core.12", 12); + retention.prune().unwrap(); + assert!(!root.join("core.11").exists()); + assert!(root.join("core.12").exists()); +} + +#[test] +fn symlinked_core_directory_is_rejected() { + let test = TestRoot::new(); + let real = test.0.join("real"); + fs::create_dir(&real).unwrap(); + let alias = test.0.join("alias"); + symlink(real, &alias).unwrap(); + assert!(CrashRetention::open(alias).is_err()); +} diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index 09d4264f..646a9898 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -53,11 +53,24 @@ remains available and the workflow reports a warning. The image does not configure the host's Linux core collector. Inspect `/proc/sys/kernel/core_pattern` on the Docker host before expecting a dump in the mounted data volume. A leading `|` sends a crash to a host-side collector; -relative file patterns write in the crashing process's working directory. -The container does not currently set a private core working directory, a -core size limit, or dump retention. Release debug symbols are available when -the release was run with `--symbols`. Do not assume `/opt/crowdb/data` contains -a core after a crash. +relative `core` or `core.*` patterns write in the crashing process's working +directory. The container runs the monitor and managed children from the private +`/opt/crowdb/data/crash` directory. On monitor startup and after a child is +reaped, it removes older regular `core` files and keeps the newest one. +The directory has mode `0700`; symlinks named `core.*` are not followed or +deleted. This is one-core retention, not a promise that the host creates a +volume file. Other relative filename patterns are outside this retention rule. + +For a host with a relative `core` pattern, add a size bound to `docker run`: + +```sh +--ulimit core=1073741824:1073741824 +``` + +The example bounds each core to 1 GiB. A small bound may truncate a dump and +make some stack frames unavailable. Core collection can also be suppressed by +the host's dumpability policy, including for executables with file capabilities. +The container never changes `core_pattern` or the host's dumpability policy. - On a systemd-coredump host, use `coredumpctl list` and `coredumpctl dump` on the host to locate and export a captured dump. @@ -66,10 +79,25 @@ a core after a crash. - On Docker Desktop, inspect the Linux VM's collector. The desktop host's native crash directory is not the container's core directory. -Core dumps can contain credentials and user data. Store exports privately, -apply host retention policy, and match the exact image revision and binary -build when symbolizing. The required bounded volume collection and source-line -symbolization check remain tracked by R188. +Core dumps can contain credentials and user data. Keep exports in a private +directory and do not attach them to ordinary logs or issues. If the release +included the optional symbol archive, use the exact image and matching archive +to show source-line stacks: + +```sh +pixi run -- python tools/symbolize-container-core.py \ + --image 'docker.io/crowdb/crowdb-iceberg:' \ + --symbols '/private/path/crowdb-symbols--git--linux-amd64.tar.zst' \ + --binary crowdb-monitor --core /private/path/core +``` + +The tool copies binaries from a stopped container into a temporary private +directory, verifies source revision, version and SHA-256 hashes, then runs +`gdb` without printing frame arguments. Use the crashed child binary instead +of `crowdb-monitor` for a child core. The temporary binaries are removed after +the stack is shown; the core stays at the path supplied by the operator. +The exact-build source-line check on a supported file-based collector remains +tracked by R188. Collector behavior follows the [Linux core pattern documentation](https://docs.kernel.org/admin-guide/sysctl/kernel.html), [systemd-coredump manual](https://www.freedesktop.org/software/systemd/man/250/systemd-coredump.socket.html), diff --git a/container/single-node-container/entrypoint.sh b/container/single-node-container/entrypoint.sh index 220c5c48..f578f62d 100644 --- a/container/single-node-container/entrypoint.sh +++ b/container/single-node-container/entrypoint.sh @@ -8,4 +8,21 @@ fi echo 'CROWDB preview data: Docker creates an anonymous volume when none is specified. For data you want to keep across container recreation, use --mount type=volume,source=crowdb-data,target=/opt/crowdb/data.' +CROWDB_CORE_DIR=/opt/crowdb/data/crash +export CROWDB_CORE_DIR + +case "$(cat /proc/sys/kernel/core_pattern)" in + core|core.*) + if [ "$(ulimit -c)" = 0 ]; then + echo 'CROWDB core files require a nonzero Docker --ulimit core setting.' >&2 + fi + ;; + '|'*) + echo 'CROWDB core dumps use the Docker host collector; inspect the host for dumps.' >&2 + ;; + *) + echo 'CROWDB core_pattern does not use a relative core filename; inspect the Docker host for dumps.' >&2 + ;; +esac + exec /opt/crowdb/bin/crowdb-monitor run --profile /opt/crowdb/etc/profile.toml diff --git a/container/single-node-container/tests/container-e2e.sh b/container/single-node-container/tests/container-e2e.sh index daf21126..476e00db 100644 --- a/container/single-node-container/tests/container-e2e.sh +++ b/container/single-node-container/tests/container-e2e.sh @@ -284,6 +284,10 @@ client_env=$(docker exec "$name" crowdb-monitor credentials show --format env) [[ $(docker exec "$name" stat -c %a /opt/crowdb/data/secrets/server.env) == 600 ]] [[ $(docker exec "$name" stat -c %a /opt/crowdb/data/secrets/client.env) == 600 ]] docker exec "$name" cat /opt/crowdb/data/bootstrap/manifest.json | jq -e '.state == "ready"' >/dev/null +[[ $(docker exec "$name" stat -c %a /opt/crowdb/data/crash) == 700 ]] +[[ $(docker exec "$name" readlink /proc/1/cwd) == /opt/crowdb/data/crash ]] +kv_pid=$(docker exec "$name" cat /opt/crowdb/run/status/monitor.json | jq -er '.services.kv.pid') +[[ $(docker exec "$name" readlink "/proc/$kv_pid/cwd") == /opt/crowdb/data/crash ]] verify_public_services node container/single-node-container/tests/web-ui.cjs "http://127.0.0.1:$(port 8080)" "$name" echo "checking S3 and Iceberg client writes" diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md index 25566d8c..8c68cd92 100644 --- a/doc/backlog/R188-console-group0-authority.md +++ b/doc/backlog/R188-console-group0-authority.md @@ -71,15 +71,18 @@ not block R187 completion. 7. Audit the S3 mini-cluster's local `console.toml` and restart path under the same authority boundary. Retain only launch inputs and bootstrap seeds locally after Group 0 cutover; do not replay a local topology copy. -8. Migrate the verified bare-metal deployment and operations material into - dedicated bare-metal deployment documentation, organized by KV cluster, - chunk layer, and data access servers. State that bare-metal is not yet - production-ready. Keep Docker deployment documentation independent. +8. Publish the verified bare-metal deployment and operations material under + `/nv/cpp/crowdb-web/site/docs/`, organized by KV cluster, chunk layer, and + data access servers. State that bare-metal is not yet production-ready. + Keep Docker deployment documentation independent and do not add deployment + guides to this repository. 9. Complete container crash diagnostics without changing host-wide collector policy. Respect file-based core patterns, Ubuntu Apport, systemd-coredump and Docker Desktop's Linux VM; document where dumps actually go or why collection - is unavailable. Where file dumps are supported, retain them in a bounded, - private data-volume location. Provide an exact-build source-line + is unavailable. Where relative `core` file dumps are supported, use a + private directory inside the container data volume and retain only the + newest core across child recovery and monitor restart. Bound each dump with + Docker's core ulimit. Provide an exact-build source-line symbolization workflow for child and monitor crashes. Dumps can contain secrets and user data; diagnostics must not expose them in ordinary logs. Optionally ship exact-build debug symbols as a separate GitHub Release asset @@ -148,8 +151,8 @@ not block R187 completion. boundary is explicit, and no link targets the removed combined guide. Invariant: deployment guidance follows its implementation. E2E test. - Given a disposable container on a supported file-based core collector, when - a child or PID 1 crashes, assert the dump has private ownership, bounded - retention and cleanup, and resolves to source lines using exact-build symbols. + a child or PID 1 crashes, assert the dump has private ownership, one-core + retention and size bounds, and resolves to source lines using exact-build symbols. Assert ordinary logs disclose no dump contents or credentials and the container does not change host-wide collector policy. Invariant: private, bounded and reproducible crash diagnostics. E2E test. @@ -181,5 +184,6 @@ Required gates: - This host routes `core_pattern` to Apport, so a container-local directory and core ulimit cannot guarantee a dump in `/opt/crowdb/data`. End-to-end acceptance needs a disposable host with file-based collection or a verified - host-collector export workflow. Bounded volume retention and source-line - symbolization remain unverified. + host-collector export workflow. Private one-core retention has passed local + tests, while collection and source-line symbolization on a real dump remain + unverified. diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 56fb1512..4c36f0eb 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -138,17 +138,19 @@ configuration, documentation and crash-diagnostics tasks below are pending. ## Crash diagnostics follow-up -Transferred from R187 by user request. This work remains pending while R188 -is paused; it does not block the single-node image requirement. +Transferred from R187 by user request. It does not block the single-node image +requirement. -- [ ] **Crash dump location and retention**: document and test how Linux +- [~] **Crash dump location and retention**: document and test how Linux host `core_pattern`, Docker's core ulimit, and the non-root container affect CROWDB child and PID 1 crashes. Cover a plain relative core-file pattern, Ubuntu Apport, systemd-coredump, and Docker Desktop's Linux VM. Choose a bounded, private location under the mounted `/opt/crowdb/data` volume where the host permits file dumps; otherwise report the host collector location and provide explicit setup guidance instead of claiming the volume contains - a core. Verify one disposable child crash end to end, retention/cleanup, + a core. Use a private data-volume directory and retain the newest `core` + file after child recovery and monitor restart. Require Docker's core ulimit + for a per-dump size bound. Verify one disposable child crash end to end, retention/cleanup, secret exposure, and symbolization against the exact binary build. Do not change the host-wide `core_pattern` from inside the container. Files: `container/single-node-container/{Dockerfile,entrypoint.sh,tests/**}`, @@ -157,7 +159,14 @@ is paused; it does not block the single-node image requirement. The single-node README now states the host collector boundary and identifies Apport, systemd-coredump and Docker Desktop lookup paths without promising a volume dump. This host reports an Apport pipe pattern and core ulimit 0. - Volume retention, exact-build symbols and disposable-host acceptance remain. + The container now creates a private crash directory after the bootstrap + manifest is opened, runs the monitor and children there, and retains the + newest regular `core` file after restart or child recovery. A 1 GiB Docker + core ulimit example bounds each dump. Focused retention and monitor suites + pass. The complete container release, image and E2E gate passes, including + startup, crash and hang recovery, persisted-volume restart, exhausted restart + budget and monitor death. Rust fmt and clippy pass. The host's Apport pipe + still prevents file-based end-to-end acceptance. - [ ] **Manual release and optional symbols**: the `tools/` release script now has a read-only dry run, consistent version updates, tag and GitHub Release creation, and workflow dispatch. Optional `--symbols` extracts debug symbols @@ -168,10 +177,10 @@ is paused; it does not block the single-node image requirement. `.github/workflows/release-container.yml`. ## Documentation and completion -- [ ] **Bare-metal documentation**: migrate verified KV, chunk and access - setup into dedicated deployment documentation, state the non-production - boundary, then fix links and remove obsolete combined material. Keep Docker - deployment notes independent. +- [ ] **Bare-metal documentation**: publish verified KV, chunk and access + setup under `/nv/cpp/crowdb-web/site/docs/`, state the non-production + boundary, then fix website links and remove obsolete combined material. + Keep Docker deployment notes independent; do not put these guides in crowdb. - [ ] **Acceptance and cleanup**: run affected integration cases, full console and UI suites, Rust fmt and lint; update the relevant permanent architecture, then remove the requirement, backlog entry and this plan when complete. diff --git a/tools/symbolize-container-core.py b/tools/symbolize-container-core.py new file mode 100644 index 00000000..07a39050 --- /dev/null +++ b/tools/symbolize-container-core.py @@ -0,0 +1,123 @@ +#!/usr/bin/env python3 +# Copyright 2026-present Gian +# Licensed under the Apache License, Version 2.0. + +"""Show source-line stacks from a private core and exact image symbol asset.""" + +import argparse +import hashlib +import json +from pathlib import Path, PurePosixPath +import shutil +import subprocess +import tarfile +import tempfile + + +def output(*command: str) -> str: + return subprocess.check_output(command, text=True).strip() + + +def extract_symbols(archive: Path, destination: Path) -> None: + process = subprocess.Popen(["zstd", "-dc", str(archive)], stdout=subprocess.PIPE) + try: + assert process.stdout is not None + with tarfile.open(fileobj=process.stdout, mode="r|") as bundle: + for member in bundle: + name = member.name.removeprefix("./") + if name in {"", ".", "bin", "lib"} and member.isdir(): + continue + path = PurePosixPath(name) + valid = ( + name in {"VERSION", "SOURCE_REVISION", "RUNTIME_SHA256SUMS"} + or (len(path.parts) == 2 and path.parts[0] in {"bin", "lib"} and path.name.endswith(".debug")) + ) + if not valid or not member.isfile() or path.is_absolute() or ".." in path.parts: + raise ValueError("unexpected symbol archive entry") + target = destination.joinpath(*path.parts) + target.parent.mkdir(exist_ok=True) + source = bundle.extractfile(member) + if source is None: + raise ValueError("missing symbol archive contents") + with source, target.open("wb") as saved: + shutil.copyfileobj(source, saved) + if process.wait() != 0: + raise ValueError("symbol archive decompression failed") + except BaseException: + process.kill() + process.wait() + raise + finally: + if process.stdout is not None: + process.stdout.close() + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--image", required=True, help="Exact Docker image tag or digest") + parser.add_argument("--symbols", type=Path, required=True, help="Matching release symbol archive") + parser.add_argument("--binary", required=True, help="Crashed binary name, such as crowdb-monitor") + parser.add_argument("--core", type=Path, required=True, help="Privately exported core file") + args = parser.parse_args() + if Path(args.binary).name != args.binary or not args.binary.startswith("crowdb-"): + parser.error("--binary must be a CROWDB binary name") + if not args.core.is_file() or not args.symbols.is_file(): + parser.error("core and symbol archive must be readable files") + + labels = json.loads(output("docker", "image", "inspect", args.image))[0]["Config"]["Labels"] + revision = labels["org.opencontainers.image.revision"] + version = labels["org.opencontainers.image.version"] + with tempfile.TemporaryDirectory(prefix="crowdb-core-") as temporary: + root = Path(temporary) + root.chmod(0o700) + runtime = root / "runtime" + runtime.mkdir() + container = output("docker", "create", "--entrypoint", "/bin/true", args.image) + try: + for directory in ("bin", "lib"): + subprocess.run( + ["docker", "cp", f"{container}:/opt/crowdb/{directory}", str(runtime / directory)], + check=True, + ) + finally: + subprocess.run(["docker", "rm", container], check=True, capture_output=True) + symbols = root / "symbols" + symbols.mkdir() + extract_symbols(args.symbols.resolve(), symbols) + if (symbols / "SOURCE_REVISION").read_text().strip() != revision: + raise ValueError("symbol source revision differs from image") + if (symbols / "VERSION").read_text().strip() != version: + raise ValueError("symbol version differs from image") + for line in (symbols / "RUNTIME_SHA256SUMS").read_text().splitlines(): + expected, relative = line.split(maxsplit=1) + relative = relative.lstrip("*") + path = Path(relative) + if path.is_absolute() or path.parts[0] not in {"bin", "lib"} or ".." in path.parts: + raise ValueError("invalid runtime checksum path") + actual = hashlib.sha256((runtime / path).read_bytes()).hexdigest() + if actual != expected: + raise ValueError(f"image binary differs from symbols: {relative}") + binary = runtime / "bin" / args.binary + if not binary.is_file(): + parser.error("crashed binary is absent from the image") + if not (symbols / "bin" / f"{args.binary}.debug").is_file(): + raise ValueError("exact debug symbols for the crashed binary are absent") + for directory in ("bin", "lib"): + for debug in (symbols / directory).glob("*.debug"): + if debug.is_symlink() or not debug.is_file(): + raise ValueError("invalid symbol archive entry") + shutil.copyfile(debug, runtime / directory / debug.name) + print(f"Verified image version {version}, revision {revision}", flush=True) + subprocess.run( + [ + "gdb", "-q", "-batch", "-ex", "set pagination off", + "-ex", "set print frame-arguments none", + "-ex", f"set solib-search-path {runtime / 'lib'}", + "-ex", "thread apply all bt", str(binary), str(args.core.resolve()), + ], + check=True, + ) + + +if __name__ == "__main__": + main() From ea29a0f2312555844046c1b59fb0dda796e0973f Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 10:46:59 +0800 Subject: [PATCH 41/74] Confirm rack and node writes through Group 0 --- .../src/commands/cluster/hardware.rs | 72 +++++++++++-- .../tests/launch_registry_cli_test.rs | 41 +++++++ doc/working/plan-console-authority.md | 10 +- lib/crowdb-console-shared/src/ops/hardware.rs | 100 ++++++++++++++++++ .../src/ops/hardware/authority.rs | 47 ++++++++ .../tests/ops_hardware_authority_test.rs | 75 +++++++++++++ 6 files changed, 333 insertions(+), 12 deletions(-) create mode 100644 lib/crowdb-console-shared/src/ops/hardware/authority.rs create mode 100644 lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs diff --git a/app/crowdb-cli/src/commands/cluster/hardware.rs b/app/crowdb-cli/src/commands/cluster/hardware.rs index 417604a0..4038989f 100644 --- a/app/crowdb-cli/src/commands/cluster/hardware.rs +++ b/app/crowdb-cli/src/commands/cluster/hardware.rs @@ -44,10 +44,17 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { Ok(c) => c, Err(c) => return c, }; - match crowdb_console_shared::ops::hardware::add_rack(&ctx, rack_id, &name).await { + let result = if cli.registry.is_some() { + crowdb_console_shared::ops::hardware::add_rack_to_group0(&ctx, rack_id, &name).await + } else { + crowdb_console_shared::ops::hardware::add_rack(&ctx, rack_id, &name).await + }; + match result { Ok(entry) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; + if cli.registry.is_none() { + if let Err(c) = commit_config(cli, &ctx) { + return c; + } } println!("added rack {}", entry.id); ExitCode::SUCCESS @@ -89,7 +96,17 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { Ok(c) => c, Err(c) => return c, }; - let racks = crowdb_console_shared::ops::hardware::list_racks(&ctx); + let racks = if cli.registry.is_some() { + match crowdb_console_shared::ops::hardware::list_racks_from_group0(&ctx).await { + Ok(racks) => racks, + Err(error) => { + eprintln!("error: list racks: {error}"); + return ExitCode::from(2); + } + } + } else { + crowdb_console_shared::ops::hardware::list_racks(&ctx) + }; if racks.is_empty() { println!("(no racks)"); return ExitCode::SUCCESS; @@ -120,6 +137,8 @@ pub enum NodeVerb { ssh_user: String, #[arg(short = 'k', long)] ssh_key: Option, + #[arg(long)] + ssh_credential_ref: Option, }, Remove { #[arg(short = 'I', long)] @@ -143,7 +162,12 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { ssh_port, ssh_user, ssh_key, + ssh_credential_ref, } => { + if cli.registry.is_some() && ssh_key.is_some() { + eprintln!("error: --ssh-key is local secret material; use --ssh-credential-ref with a launch registry"); + return ExitCode::from(1); + } let node_id: NodeId = match id.parse() { Ok(n) => n, Err(e) => { @@ -166,16 +190,23 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { ssh_user, ssh_key, ssh_password: None, - ssh_credential_ref: None, + ssh_credential_ref, }; let ctx = match op_context(cli) { Ok(c) => c, Err(c) => return c, }; - match crowdb_console_shared::ops::hardware::add_node(&ctx, entry.clone()).await { + let result = if cli.registry.is_some() { + crowdb_console_shared::ops::hardware::add_node_to_group0(&ctx, entry.clone()).await + } else { + crowdb_console_shared::ops::hardware::add_node(&ctx, entry.clone()).await + }; + match result { Ok(e) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; + if cli.registry.is_none() { + if let Err(c) = commit_config(cli, &ctx) { + return c; + } } println!("added node {} (rack {})", e.id, e.rack_id); ExitCode::SUCCESS @@ -217,7 +248,17 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { Ok(c) => c, Err(c) => return c, }; - let nodes = crowdb_console_shared::ops::hardware::list_nodes(&ctx, None); + let nodes = if cli.registry.is_some() { + match crowdb_console_shared::ops::hardware::list_nodes_from_group0(&ctx, None).await { + Ok(nodes) => nodes, + Err(error) => { + eprintln!("error: list nodes: {error}"); + return ExitCode::from(2); + } + } + } else { + crowdb_console_shared::ops::hardware::list_nodes(&ctx, None) + }; print_node_table(&nodes) } NodeVerb::ListRack { rack } => { @@ -232,7 +273,18 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { Ok(c) => c, Err(c) => return c, }; - let nodes = crowdb_console_shared::ops::hardware::list_nodes(&ctx, Some(rack_id)); + let nodes = if cli.registry.is_some() { + match crowdb_console_shared::ops::hardware::list_nodes_from_group0(&ctx, Some(rack_id)).await + { + Ok(nodes) => nodes, + Err(error) => { + eprintln!("error: list nodes: {error}"); + return ExitCode::from(2); + } + } + } else { + crowdb_console_shared::ops::hardware::list_nodes(&ctx, Some(rack_id)) + }; print_node_table(&nodes) } } diff --git a/app/crowdb-cli/tests/launch_registry_cli_test.rs b/app/crowdb-cli/tests/launch_registry_cli_test.rs index c3640d11..264201a2 100644 --- a/app/crowdb-cli/tests/launch_registry_cli_test.rs +++ b/app/crowdb-cli/tests/launch_registry_cli_test.rs @@ -169,3 +169,44 @@ async fn chunk_commands_and_generic_launch_controls_share_process_identity() { } assert_eq!(LaunchRegistry::load(&path).unwrap().launches, records); } + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn registry_hardware_uses_group_zero_across_cli_invocations() { + let g0 = common::direct::spawn_group0() + .await + .expect("KV server binary must be built"); + let dir = tempdir_in_test_data("cli-registry-hardware"); + let path = dir.path().join("launches.toml"); + LaunchRegistry { + version: 1, + launches: Vec::new(), + } + .save(&path) + .unwrap(); + + run_command( + &path, + g0.mgmt_port, + &["cluster", "rack", "add", "--id", "2", "--name", "rack-two"], + ); + assert!(run_command(&path, g0.mgmt_port, &["cluster", "rack", "list"]).contains("rack-two")); + run_command( + &path, + g0.mgmt_port, + &[ + "cluster", + "node", + "add", + "--id", + "2", + "--rack", + "2", + "--host", + "10.0.0.2", + "--ssh-credential-ref", + "ops-key", + ], + ); + assert!(run_command(&path, g0.mgmt_port, &["cluster", "node", "list"]).contains("10.0.0.2")); + assert!(!dir.path().join("invalid-legacy.toml").exists()); +} diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 4c36f0eb..117612c3 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -105,8 +105,14 @@ configuration, documentation and crash-diagnostics tasks below are pending. fields are now part of rack/node records and bootstrap writes the names, management host, SSH port/user and reference without copying secret material. Bare-metal snapshots expose the same Group 0 values to separate consoles. - Exact-value - confirmation and the CLI/Web cutover remain. + Conditional rack/node creation now confirms matching existing values and + rejects conflicting values without changing either console's local topology. + Group 0 rack/node lists expose shared names, management hosts, SSH settings + and credential references but no private key material. Registry-mode CLI + add/list commands use these operations; a real CLI process regression caught + a management-port-as-RPC seed and now refreshes topology before the hardware + write. Complete Console shared and CLI suites, Rust fmt and workspace clippy + pass. Registry-mode Web, deletions, disks and legacy bootstrap still remain. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. diff --git a/lib/crowdb-console-shared/src/ops/hardware.rs b/lib/crowdb-console-shared/src/ops/hardware.rs index 8d88df65..cec0220c 100644 --- a/lib/crowdb-console-shared/src/ops/hardware.rs +++ b/lib/crowdb-console-shared/src/ops/hardware.rs @@ -15,6 +15,106 @@ use crate::config::{DiskEntry, DiskGroupEntry, NodeEntry, RackEntry}; use crate::error::{Error, Result}; use crate::ops::OpContext; +mod authority; + +/// Create a rack only after Group 0 confirms the exact record. +/// +/// # Errors +/// Returns a conflict for a different existing record, or the authority error. +pub async fn add_rack_to_group0(ctx: &OpContext, rack_id: u64, name: &str) -> Result { + let entry = RackEntry { + id: rack_id, + name: name.to_owned(), + }; + authority::create( + ctx, + crowdb_protocol::key::RackKey { rack_id }, + &RackValue { + status: HwStatus::Up as i32, + node_ids: Vec::new(), + name: name.to_owned(), + }, + ) + .await?; + Ok(entry) +} + +/// Create a node only after its rack and exact record are confirmed in Group 0. +/// +/// # Errors +/// Returns a missing rack, conflicting node, or authority error. +pub async fn add_node_to_group0(ctx: &OpContext, entry: NodeEntry) -> Result { + authority::ready(ctx).await?; + if ctx.sysmd().get_rack(entry.rack_id).await?.is_none() { + return Err(Error::NotFound { + kind: "rack".into(), + id: entry.rack_id.to_string(), + }); + } + let value = NodeValue { + status: HwStatus::Up as i32, + management_host: entry.host.clone(), + ssh_port: entry.ssh_port, + ssh_user: entry.ssh_user.clone(), + ssh_credential_ref: entry.ssh_credential_ref.clone(), + ..Default::default() + }; + authority::create( + ctx, + crowdb_protocol::key::NodeKey { + rack_id: entry.rack_id, + node_id: entry.id, + }, + &value, + ) + .await?; + Ok(entry) +} + +/// Read rack names from confirmed Group 0 state. +/// +/// # Errors +/// Returns an authority error; a local topology file is never consulted. +pub async fn list_racks_from_group0(ctx: &OpContext) -> Result> { + authority::ready(ctx).await?; + let mut racks: Vec<_> = ctx + .sysmd() + .list_racks() + .await? + .into_iter() + .map(|(id, value)| RackEntry { id, name: value.name }) + .collect(); + racks.sort_unstable_by_key(|rack| rack.id); + Ok(racks) +} + +/// Read node connection identity from confirmed Group 0 state. +/// +/// # Errors +/// Returns an authority error; private SSH material is never returned. +pub async fn list_nodes_from_group0(ctx: &OpContext, rack_id: Option) -> Result> { + authority::ready(ctx).await?; + let mut nodes: Vec<_> = ctx + .sysmd() + .list_nodes() + .await? + .into_iter() + .filter(|(rack, _, _)| rack_id.is_none() || rack_id == Some(*rack)) + .map(|(rack_id, id, value)| NodeEntry { + id, + rack_id, + host: value.management_host, + ssh_port: value.ssh_port, + ssh_user: value.ssh_user, + ssh_key: None, + ssh_password: None, + ssh_credential_ref: value.ssh_credential_ref, + }) + .collect(); + nodes.sort_unstable_by_key(|node| node.id); + Ok(nodes) +} + // ── rack ──────────────────────────────────────────────────────── /// Add a rack to the local config and group-0 sysdata. diff --git a/lib/crowdb-console-shared/src/ops/hardware/authority.rs b/lib/crowdb-console-shared/src/ops/hardware/authority.rs new file mode 100644 index 00000000..9205b618 --- /dev/null +++ b/lib/crowdb-console-shared/src/ops/hardware/authority.rs @@ -0,0 +1,47 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Conditional Group 0 hardware publication with confirmed outcomes. + +use crowdb_kv_client::{Error as KvError, GetOutcome, ReadMode}; +use crowdb_protocol::key::TextKey; + +use crate::error::{Error, Result}; +use crate::ops::OpContext; + +pub(super) async fn ready(ctx: &OpContext) -> Result<()> { + ctx.kv().refresh_topology().await?; + Ok(()) +} + +pub(super) async fn create(ctx: &OpContext, key: impl TextKey, value: &T) -> Result<()> { + ready(ctx).await?; + let path = key.to_path(); + let intended = serde_json::to_value(value).map_err(|error| Error::Config(error.to_string()))?; + let payload = serde_json::to_vec(value).map_err(|error| Error::Config(error.to_string()))?; + match ctx.kv().put_cas(0, 0, path.as_bytes(), &payload, 0).await { + Ok(_) => Ok(()), + Err(error @ (KvError::CasFailed { .. } | KvError::OutcomeUnknown)) => { + match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, .. } => { + let actual: serde_json::Value = + serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?; + if actual == intended { + Ok(()) + } else { + Err(Error::Conflict { + kind: "hardware".into(), + id: path, + }) + } + } + GetOutcome::NotFound => Err(error.into()), + } + } + Err(error) => Err(error.into()), + } +} diff --git a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs new file mode 100644 index 00000000..8bf459b6 --- /dev/null +++ b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs @@ -0,0 +1,75 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_console_shared::config::{ConsoleConfig, NodeEntry}; +use crowdb_console_shared::error::Error; +use crowdb_console_shared::ops::{cluster as cluster_ops, hardware, OpContext}; +use crowdb_test_harness::cluster::KvCluster; + +#[path = "common/bootstrap_authority.rs"] +mod bootstrap_authority; + +#[tokio::test] +async fn separate_consoles_confirm_matching_hardware_and_reject_conflicts() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + cluster_ops::init(&bootstrap, &[1]).await.unwrap(); + + let first = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + let second = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + + hardware::add_rack_to_group0(&first, 2, "rack-two").await.unwrap(); + hardware::add_rack_to_group0(&second, 2, "rack-two") + .await + .unwrap(); + let conflict = hardware::add_rack_to_group0(&second, 2, "other") + .await + .unwrap_err(); + assert!(matches!(conflict, Error::Conflict { .. }), "{conflict:?}"); + assert_eq!( + second.sysmd().get_rack(2).await.unwrap().unwrap().name, + "rack-two" + ); + assert_eq!( + hardware::list_racks_from_group0(&second).await.unwrap()[1].name, + "rack-two" + ); + assert!(first.config().racks.is_empty()); + assert!(second.config().racks.is_empty()); + + let node = NodeEntry { + id: 2, + rack_id: 2, + host: "10.0.0.2".into(), + ssh_port: 2222, + ssh_user: "operator".into(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: Some("ops-key".into()), + }; + hardware::add_node_to_group0(&first, node.clone()).await.unwrap(); + hardware::add_node_to_group0(&second, node.clone()).await.unwrap(); + let changed = NodeEntry { + host: "10.0.0.3".into(), + ..node + }; + let conflict = hardware::add_node_to_group0(&second, changed).await.unwrap_err(); + assert!(matches!(conflict, Error::Conflict { .. }), "{conflict:?}"); + let actual = second.sysmd().get_node(2, 2).await.unwrap().unwrap(); + assert_eq!(actual.management_host, "10.0.0.2"); + assert_eq!(actual.ssh_credential_ref.as_deref(), Some("ops-key")); + let listed = hardware::list_nodes_from_group0(&second, Some(2)).await.unwrap(); + assert_eq!(listed.len(), 1); + assert_eq!(listed[0].host, "10.0.0.2"); + assert!(listed[0].ssh_key.is_none()); + assert!(listed[0].ssh_password.is_none()); + assert!(second.config().nodes.is_empty()); +} From 3bb237b53414f7f737dbf4b3e2a87f13e1716949 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 10:50:48 +0800 Subject: [PATCH 42/74] Verify hardware publication after lost responses --- doc/working/plan-console-authority.md | 4 +++- .../tests/ops_hardware_authority_test.rs | 18 ++++++++++++++++++ 2 files changed, 21 insertions(+), 1 deletion(-) diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 117612c3..b61b21b5 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -112,7 +112,9 @@ configuration, documentation and crash-diagnostics tasks below are pending. add/list commands use these operations; a real CLI process regression caught a management-port-as-RPC seed and now refreshes topology before the hardware write. Complete Console shared and CLI suites, Rust fmt and workspace clippy - pass. Registry-mode Web, deletions, disks and legacy bootstrap still remain. + pass. A real RPC proxy drops the committed rack write reply; a confirmed + linearizable read recovers the successful outcome. Registry-mode Web, + deletions, disks and legacy bootstrap still remain. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. diff --git a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs index 8bf459b6..c11e9887 100644 --- a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs +++ b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs @@ -5,9 +5,12 @@ use crowdb_console_shared::config::{ConsoleConfig, NodeEntry}; use crowdb_console_shared::error::Error; use crowdb_console_shared::ops::{cluster as cluster_ops, hardware, OpContext}; use crowdb_test_harness::cluster::KvCluster; +use std::sync::atomic::Ordering; #[path = "common/bootstrap_authority.rs"] mod bootstrap_authority; +#[path = "common/rpc_response_proxy.rs"] +mod rpc_response_proxy; #[tokio::test] async fn separate_consoles_confirm_matching_hardware_and_reject_conflicts() { @@ -73,3 +76,18 @@ async fn separate_consoles_confirm_matching_hardware_and_reject_conflicts() { assert!(listed[0].ssh_password.is_none()); assert!(second.config().nodes.is_empty()); } + +#[tokio::test] +async fn committed_rack_survives_a_lost_conditional_write_response() { + let cluster = KvCluster::start().await; + let proxy = rpc_response_proxy::TestResponseProxy::start(cluster.group0_leader_endpoint.clone()).await; + let ctx = OpContext::new( + proxy.endpoint.clone(), + vec![proxy.management_endpoint.clone()], + ConsoleConfig::default(), + ); + proxy.armed.store(true, Ordering::SeqCst); + hardware::add_rack_to_group0(&ctx, 9, "rack-nine").await.unwrap(); + assert_eq!(proxy.dropped.load(Ordering::SeqCst), 1); + assert_eq!(ctx.sysmd().get_rack(9).await.unwrap().unwrap().name, "rack-nine"); +} From 07dc46713637b67d376f140e189a14d5b9122970 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 11:03:43 +0800 Subject: [PATCH 43/74] Seal bootstrap intent before Group 0 publication --- doc/working/plan-console-authority.md | 8 + .../src/bootstrap_intent.rs | 292 ++++++++++++++++++ lib/crowdb-console-shared/src/lib.rs | 1 + lib/crowdb-console-shared/src/ops/cluster.rs | 2 +- .../src/ops/cluster/bootstrap.rs | 36 +++ .../tests/bootstrap_intent_test.rs | 184 +++++++++++ 6 files changed, 522 insertions(+), 1 deletion(-) create mode 100644 lib/crowdb-console-shared/src/bootstrap_intent.rs create mode 100644 lib/crowdb-console-shared/tests/bootstrap_intent_test.rs diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index b61b21b5..b187f1bc 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -135,6 +135,14 @@ configuration, documentation and crash-diagnostics tasks below are pending. recording membership. Four focused failure cases, complete shared tests, Web deploy/restart/migration suites, fmt and clippy pass. Logs: `/tmp/crowdb-bootstrap-replay-*.log`. Durable intent/cutover remain pending. + A separate versioned bootstrap intent now captures rack/node identity, KV + management endpoints and selected member order without process PIDs, binary + paths or inline SSH secrets. It is atomically sealed with mode 0600; + interrupted retries restore a fresh in-memory context, reject changed + topology before mutating Group 0, then delete the intent only after + confirmed publication. Focused real Group 0 tests pass. CLI, Web and S3 + mini-cluster callers still need to use this path before removing the mixed + persisted config. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with diff --git a/lib/crowdb-console-shared/src/bootstrap_intent.rs b/lib/crowdb-console-shared/src/bootstrap_intent.rs new file mode 100644 index 00000000..7405468f --- /dev/null +++ b/lib/crowdb-console-shared/src/bootstrap_intent.rs @@ -0,0 +1,292 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Immutable pre-Group-0 topology intent for interrupted bootstrap recovery. + +use std::collections::HashSet; +use std::fs::{self, OpenOptions}; +use std::io::Write; +use std::os::unix::fs::OpenOptionsExt; +use std::path::Path; +use std::sync::atomic::{AtomicU64, Ordering}; + +use serde::{Deserialize, Serialize}; + +use crate::config::{ConsoleConfig, NodeEntry, RackEntry, ServerEntry, ServiceType}; +use crate::error::{Error, Result}; + +const VERSION: u32 = 1; +static NEXT_TEMP: AtomicU64 = AtomicU64::new(0); + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct BootstrapIntent { + version: u32, + members: Vec, + racks: Vec, + nodes: Vec, + servers: Vec, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct IntentRack { + id: u64, + name: String, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct IntentNode { + id: u64, + rack_id: u64, + host: String, + ssh_port: u16, + ssh_user: String, + ssh_credential_ref: Option, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct IntentServer { + id: String, + node_id: u64, + management_url: String, + rpc_url: Option, +} + +impl BootstrapIntent { + /// Capture immutable hardware and KV endpoint identities before bootstrap. + /// + /// # Errors + /// Rejects missing members, missing management endpoints and inline SSH secrets. + pub fn capture(config: &ConsoleConfig, members: &[u64]) -> Result { + if members.is_empty() + || members + .iter() + .enumerate() + .any(|(index, node)| members[..index].contains(node)) + { + return Err(Error::Config( + "bootstrap members must be nonempty and distinct".into(), + )); + } + let nodes: Vec<_> = config + .nodes + .iter() + .map(|node| { + if node.ssh_key.is_some() || node.ssh_password.is_some() { + return Err(Error::Config( + "bootstrap intent cannot store inline SSH secrets".into(), + )); + } + Ok(IntentNode { + id: node.id, + rack_id: node.rack_id, + host: node.host.clone(), + ssh_port: node.ssh_port, + ssh_user: node.ssh_user.clone(), + ssh_credential_ref: node.ssh_credential_ref.clone(), + }) + }) + .collect::>()?; + let servers: Vec<_> = config + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| { + let node_id = server + .node_id + .ok_or_else(|| Error::Config("bootstrap KV server has no node id".into()))?; + Ok(IntentServer { + id: server.id.clone(), + node_id, + management_url: server.url.clone(), + rpc_url: server.rpc_url.clone(), + }) + }) + .collect::>()?; + let mut rack_ids = HashSet::new(); + if config.racks.iter().any(|rack| !rack_ids.insert(rack.id)) { + return Err(Error::Config("duplicate bootstrap rack id".into())); + } + let mut node_ids = HashSet::new(); + if nodes + .iter() + .any(|node| !node_ids.insert(node.id) || !rack_ids.contains(&node.rack_id)) + { + return Err(Error::Config("duplicate or orphan bootstrap node".into())); + } + let mut endpoint_nodes = HashSet::new(); + if servers.iter().any(|server| { + !endpoint_nodes.insert(server.node_id) + || !node_ids.contains(&server.node_id) + || server.management_url.is_empty() + }) { + return Err(Error::Config("duplicate or invalid bootstrap KV endpoint".into())); + } + for member in members { + if !nodes.iter().any(|node| node.id == *member) + || !servers.iter().any(|server| server.node_id == *member) + { + return Err(Error::Config(format!( + "bootstrap member {member} lacks node or KV endpoint" + ))); + } + } + Ok(Self { + version: VERSION, + members: members.to_vec(), + racks: config + .racks + .iter() + .map(|rack| IntentRack { + id: rack.id, + name: rack.name.clone(), + }) + .collect(), + nodes, + servers, + }) + } + + #[must_use] + pub fn members(&self) -> &[u64] { + &self.members + } + + /// Create the intent once, accepting an identical interrupted attempt. + /// + /// # Errors + /// Rejects changed intent, symlinks, invalid content or I/O errors. + pub fn seal(&self, path: &Path) -> Result<()> { + if path.exists() || fs::symlink_metadata(path).is_ok() { + return if Self::load(path)? == *self { + Ok(()) + } else { + Err(Error::Conflict { + kind: "bootstrap intent".into(), + id: path.display().to_string(), + }) + }; + } + let parent = path + .parent() + .ok_or_else(|| Error::Config("bootstrap intent has no parent".into()))?; + fs::create_dir_all(parent)?; + let data = toml::to_string(self).map_err(|error| Error::Config(error.to_string()))?; + let temporary = path.with_extension(format!( + "bootstrap-tmp-{}-{}", + std::process::id(), + NEXT_TEMP.fetch_add(1, Ordering::Relaxed) + )); + let result = (|| { + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&temporary)?; + file.write_all(data.as_bytes())?; + file.sync_all()?; + match fs::hard_link(&temporary, path) { + Ok(()) => { + fs::File::open(parent)?.sync_all()?; + Ok(()) + } + Err(error) if error.kind() == std::io::ErrorKind::AlreadyExists => { + if Self::load(path)? == *self { + Ok(()) + } else { + Err(Error::Conflict { + kind: "bootstrap intent".into(), + id: path.display().to_string(), + }) + } + } + Err(error) => Err(Error::Io(error)), + } + })(); + let _ = fs::remove_file(&temporary); + result + } + + /// Read a previously sealed intent without accepting a legacy console file. + /// + /// # Errors + /// Rejects symlinks, unknown fields, invalid version or I/O errors. + pub fn load(path: &Path) -> Result { + if !fs::symlink_metadata(path)?.file_type().is_file() { + return Err(Error::Config("bootstrap intent is not a regular file".into())); + } + let intent: Self = + toml::from_str(&fs::read_to_string(path)?).map_err(|error| Error::Config(error.to_string()))?; + if intent.version != VERSION { + return Err(Error::Config("unsupported bootstrap intent version".into())); + } + if Self::capture(&intent.to_config(), &intent.members)? != intent { + return Err(Error::Config("bootstrap intent contents are inconsistent".into())); + } + Ok(intent) + } + + /// Restore only the bootstrap inputs into a fresh in-memory console context. + #[must_use] + pub fn to_config(&self) -> ConsoleConfig { + ConsoleConfig { + racks: self + .racks + .iter() + .map(|rack| RackEntry { + id: rack.id, + name: rack.name.clone(), + }) + .collect(), + nodes: self + .nodes + .iter() + .map(|node| NodeEntry { + id: node.id, + rack_id: node.rack_id, + host: node.host.clone(), + ssh_port: node.ssh_port, + ssh_user: node.ssh_user.clone(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: node.ssh_credential_ref.clone(), + }) + .collect(), + servers: self + .servers + .iter() + .map(|server| ServerEntry { + id: server.id.clone(), + url: server.management_url.clone(), + node_id: Some(server.node_id), + rpc_url: server.rpc_url.clone(), + rest_port: None, + rpc_port: None, + auto_start: false, + binary: None, + election_profile: None, + pid: None, + service_type: ServiceType::Kv, + rpc_workers: None, + no_fsync: false, + }) + .collect(), + ..Default::default() + } + } + + /// Remove the intent only after Group 0 contents were verified. + /// + /// # Errors + /// Returns an I/O error if removal fails. + pub(crate) fn clear_verified(path: &Path) -> Result<()> { + fs::remove_file(path)?; + if let Some(parent) = path.parent() { + fs::File::open(parent)?.sync_all()?; + } + Ok(()) + } +} diff --git a/lib/crowdb-console-shared/src/lib.rs b/lib/crowdb-console-shared/src/lib.rs index e31cacbb..83c49681 100644 --- a/lib/crowdb-console-shared/src/lib.rs +++ b/lib/crowdb-console-shared/src/lib.rs @@ -12,6 +12,7 @@ #![cfg_attr(not(test), allow(dead_code))] #![allow(clippy::mod_module_files)] +pub mod bootstrap_intent; pub mod clients; pub mod cluster; pub mod cluster_deployer; diff --git a/lib/crowdb-console-shared/src/ops/cluster.rs b/lib/crowdb-console-shared/src/ops/cluster.rs index 5391b531..6ad6fbfb 100644 --- a/lib/crowdb-console-shared/src/ops/cluster.rs +++ b/lib/crowdb-console-shared/src/ops/cluster.rs @@ -27,7 +27,7 @@ use crate::ops::hardware::{self, AddDiskInput}; use crate::ops::OpContext; mod bootstrap; -pub use bootstrap::{init, InitSummary}; +pub use bootstrap::{init, init_with_intent, InitSummary}; fn server_client(ctx: &OpContext, node_id: u64) -> Result { let url = ctx.node_mgmt_url(node_id)?; diff --git a/lib/crowdb-console-shared/src/ops/cluster/bootstrap.rs b/lib/crowdb-console-shared/src/ops/cluster/bootstrap.rs index 2ecd3e29..7356991c 100644 --- a/lib/crowdb-console-shared/src/ops/cluster/bootstrap.rs +++ b/lib/crowdb-console-shared/src/ops/cluster/bootstrap.rs @@ -4,10 +4,12 @@ //! Initialize system-group processes and publish bootstrap metadata. use super::server_client; +use crate::bootstrap_intent::BootstrapIntent; use crate::config::{ReplicaEntry, ServiceType}; use crate::error::{Error, Result}; use crate::ops::OpContext; use std::collections::{HashMap, HashSet}; +use std::path::Path; mod leader; mod nodes; @@ -21,6 +23,40 @@ pub struct InitSummary { pub nodes: Vec<(u64, u64)>, } +/// Initialize from a sealed pre-Group-0 intent, then delete it after verified +/// publication. A retry can restore an empty in-memory console context. +/// +/// # Errors +/// Rejects changed member or topology identity and propagates bootstrap errors. +pub async fn init_with_intent(ctx: &OpContext, nodes: &[u64], path: &Path) -> Result { + let intent = if std::fs::symlink_metadata(path).is_ok() { + let sealed = BootstrapIntent::load(path)?; + if sealed.members() != nodes { + return Err(Error::Conflict { + kind: "bootstrap members".into(), + id: path.display().to_string(), + }); + } + let current = ctx.config().clone(); + if current.racks.is_empty() && current.nodes.is_empty() && current.servers.is_empty() { + *ctx.config_mut() = sealed.to_config(); + } else if BootstrapIntent::capture(¤t, nodes)? != sealed { + return Err(Error::Conflict { + kind: "bootstrap topology".into(), + id: path.display().to_string(), + }); + } + sealed + } else { + let captured = BootstrapIntent::capture(&ctx.config(), nodes)?; + captured.seal(path)?; + captured + }; + let result = init(ctx, intent.members()).await?; + BootstrapIntent::clear_verified(path)?; + Ok(result) +} + /// Initialize the cluster by bootstrapping group 0 on the listed nodes. /// /// # Errors diff --git a/lib/crowdb-console-shared/tests/bootstrap_intent_test.rs b/lib/crowdb-console-shared/tests/bootstrap_intent_test.rs new file mode 100644 index 00000000..50247e4f --- /dev/null +++ b/lib/crowdb-console-shared/tests/bootstrap_intent_test.rs @@ -0,0 +1,184 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::os::unix::fs::{symlink, PermissionsExt}; + +use crowdb_console_shared::bootstrap_intent::BootstrapIntent; +use crowdb_console_shared::config::{ConsoleConfig, NodeEntry, RackEntry, ServerEntry, ServiceType}; +use crowdb_console_shared::error::Error; +use crowdb_console_shared::ops::{cluster as cluster_ops, OpContext}; +use crowdb_test_harness::cluster::KvCluster; +use crowdb_test_harness::test_dirs::tempdir_in_test_data; + +#[path = "common/bootstrap_authority.rs"] +mod bootstrap_authority; + +fn config() -> ConsoleConfig { + let mut config = ConsoleConfig::default(); + config.racks.push(RackEntry { + id: 1, + name: "rack-one".into(), + }); + config.nodes.push(NodeEntry { + id: 1, + rack_id: 1, + host: "127.0.0.1".into(), + ssh_port: 22, + ssh_user: "operator".into(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: Some("ops-key".into()), + }); + config.servers.push(ServerEntry { + id: "kv-1".into(), + url: "http://127.0.0.1:10000".into(), + node_id: Some(1), + rpc_url: Some("127.0.0.1:10100".into()), + rest_port: Some(10000), + rpc_port: Some(10100), + auto_start: true, + binary: Some("/private/bin/crowdb-kv-server".into()), + election_profile: None, + pid: Some(42), + service_type: ServiceType::Kv, + rpc_workers: None, + no_fsync: false, + }); + config +} + +#[test] +fn sealed_intent_is_private_immutable_and_restores_only_bootstrap_inputs() { + let dir = tempdir_in_test_data("bootstrap-intent"); + let path = dir.path().join("intent.toml"); + let intent = BootstrapIntent::capture(&config(), &[1]).unwrap(); + intent.seal(&path).unwrap(); + intent.seal(&path).unwrap(); + assert_eq!( + std::fs::metadata(&path).unwrap().permissions().mode() & 0o777, + 0o600 + ); + assert_eq!(BootstrapIntent::load(&path).unwrap(), intent); + let body = std::fs::read_to_string(&path).unwrap(); + assert!(!body.contains("/private/bin")); + assert!(!body.contains("pid")); + let restored = intent.to_config(); + assert_eq!(restored.racks[0].name, "rack-one"); + assert_eq!(restored.nodes[0].ssh_credential_ref.as_deref(), Some("ops-key")); + assert!(restored.servers[0].pid.is_none()); + assert!(restored.servers[0].binary.is_none()); + + let mut changed = config(); + changed.racks[0].name = "other".into(); + let error = BootstrapIntent::capture(&changed, &[1]) + .unwrap() + .seal(&path) + .unwrap_err(); + assert!(matches!(error, Error::Conflict { .. })); + assert_eq!(BootstrapIntent::load(&path).unwrap(), intent); +} + +#[test] +fn intent_rejects_inline_secrets_symlinks_and_legacy_fields() { + let mut config = config(); + config.nodes[0].ssh_key = Some("/private/id_ed25519".into()); + assert!(BootstrapIntent::capture(&config, &[1]).is_err()); + config.nodes[0].ssh_key = None; + assert!(BootstrapIntent::capture(&config, &[1, 1]).is_err()); + + let dir = tempdir_in_test_data("bootstrap-intent-invalid"); + let target = dir.path().join("target.toml"); + std::fs::write(&target, "version = 1\nlegacy = true\n").unwrap(); + let link = dir.path().join("intent.toml"); + symlink(&target, &link).unwrap(); + assert!(BootstrapIntent::load(&link).is_err()); + assert!(BootstrapIntent::capture(&config, &[1]) + .unwrap() + .seal(&link) + .is_err()); + assert!(BootstrapIntent::load(&target).is_err()); + + let strict = dir.path().join("strict.toml"); + BootstrapIntent::capture(&config, &[1]) + .unwrap() + .seal(&strict) + .unwrap(); + let body = std::fs::read_to_string(&strict).unwrap(); + let changed = body.replace("name = \"rack-one\"", "name = \"rack-one\"\nlegacy = true"); + assert_ne!(changed, body); + std::fs::write(&strict, changed).unwrap(); + assert!(BootstrapIntent::load(&strict).is_err()); +} + +#[tokio::test] +async fn interrupted_bootstrap_restores_identity_then_clears_verified_intent() { + let cluster = KvCluster::start().await; + let source = bootstrap_authority::context(&cluster).await; + let dir = tempdir_in_test_data("bootstrap-intent-resume"); + let path = dir.path().join("intent.toml"); + BootstrapIntent::capture(&source.config(), &[1]) + .unwrap() + .seal(&path) + .unwrap(); + + let resumed = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + cluster_ops::init_with_intent(&resumed, &[1], &path) + .await + .unwrap(); + assert!(!path.exists()); + assert_eq!( + resumed.sysmd().get_store(0).await.unwrap().unwrap().node_ids, + vec![1] + ); + assert_eq!(resumed.config().racks.len(), 1); + assert_eq!(resumed.config().nodes.len(), 1); +} + +#[tokio::test] +async fn committed_group_zero_is_verified_before_stale_intent_is_removed() { + let cluster = KvCluster::start().await; + let source = bootstrap_authority::context(&cluster).await; + let dir = tempdir_in_test_data("bootstrap-intent-committed"); + let path = dir.path().join("intent.toml"); + BootstrapIntent::capture(&source.config(), &[1]) + .unwrap() + .seal(&path) + .unwrap(); + cluster_ops::init(&source, &[1]).await.unwrap(); + assert!(path.exists()); + + let resumed = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + cluster_ops::init_with_intent(&resumed, &[1], &path) + .await + .unwrap(); + assert!(!path.exists()); + assert_eq!( + resumed.sysmd().list_replicas_in_group(0, 0).await.unwrap().len(), + 1 + ); +} + +#[tokio::test] +async fn changed_bootstrap_topology_is_rejected_before_group_zero_mutation() { + let cluster = KvCluster::start().await; + let ctx = bootstrap_authority::context(&cluster).await; + let dir = tempdir_in_test_data("bootstrap-intent-conflict"); + let path = dir.path().join("intent.toml"); + BootstrapIntent::capture(&ctx.config(), &[1]) + .unwrap() + .seal(&path) + .unwrap(); + ctx.config_mut().racks[0].name = "changed".into(); + let result = cluster_ops::init_with_intent(&ctx, &[1], &path).await; + assert!(matches!(result, Err(Error::Conflict { .. })), "{result:?}"); + assert!(ctx.sysmd().get_store(0).await.unwrap().is_none()); + assert!(path.exists()); +} From 9f8032ed528fd4a07b5a62592252e34e4572b11c Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 11:13:12 +0800 Subject: [PATCH 44/74] Resume persistent console bootstrap from sealed intent --- app/crowdb-cli/src/commands.rs | 2 +- app/crowdb-cli/src/commands/cluster.rs | 10 +++++++-- app/crowdb-cli/tests/cluster_cli_test.rs | 1 + app/crowdb-web/src/mgmt/cluster_init.rs | 10 ++++++--- app/crowdb-web/tests/mgmt_routes_test.rs | 28 +++++++++++++++++++++++- doc/working/plan-console-authority.md | 9 +++++--- 6 files changed, 50 insertions(+), 10 deletions(-) diff --git a/app/crowdb-cli/src/commands.rs b/app/crowdb-cli/src/commands.rs index c9129ff1..adc41892 100644 --- a/app/crowdb-cli/src/commands.rs +++ b/app/crowdb-cli/src/commands.rs @@ -95,7 +95,7 @@ pub(crate) fn load_config(cli: &Cli) -> Result std::path::PathBuf { +pub(crate) fn config_path() -> std::path::PathBuf { std::env::var_os("CROWDB_CLI_STATE").map_or_else( || { crowdb_protocol::port::namespace::runtime_root() diff --git a/app/crowdb-cli/src/commands/cluster.rs b/app/crowdb-cli/src/commands/cluster.rs index 356c0bf0..655acc38 100644 --- a/app/crowdb-cli/src/commands/cluster.rs +++ b/app/crowdb-cli/src/commands/cluster.rs @@ -15,7 +15,7 @@ use std::process::ExitCode; use clap::Subcommand; -use crate::commands::{commit_config, op_context}; +use crate::commands::{commit_config, config_path, op_context}; use crate::Cli; #[derive(Subcommand, Debug)] @@ -192,7 +192,13 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { return ExitCode::from(1); } }; - match crowdb_console_shared::ops::cluster::init(&ctx, &node_ids).await { + let result = if cli.registry.is_some() { + crowdb_console_shared::ops::cluster::init(&ctx, &node_ids).await + } else { + let intent_path = config_path().with_extension("bootstrap.toml"); + crowdb_console_shared::ops::cluster::init_with_intent(&ctx, &node_ids, &intent_path).await + }; + match result { Ok(summary) => { if let Err(c) = commit_config(cli, &ctx) { return c; diff --git a/app/crowdb-cli/tests/cluster_cli_test.rs b/app/crowdb-cli/tests/cluster_cli_test.rs index 23efa74c..317be07e 100644 --- a/app/crowdb-cli/tests/cluster_cli_test.rs +++ b/app/crowdb-cli/tests/cluster_cli_test.rs @@ -81,6 +81,7 @@ async fn cluster_status_topology_via_direct_group0() { &["cluster", "init", "-n", "1"], ); assert_eq!(code, 0, "cluster init stderr={stderr}"); + assert!(!g0.config_path.with_extension("bootstrap.toml").exists()); // status — lists stores from group-0 sysdata. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "status"]); diff --git a/app/crowdb-web/src/mgmt/cluster_init.rs b/app/crowdb-web/src/mgmt/cluster_init.rs index 79f8274b..35de565f 100644 --- a/app/crowdb-web/src/mgmt/cluster_init.rs +++ b/app/crowdb-web/src/mgmt/cluster_init.rs @@ -38,9 +38,13 @@ pub(crate) async fn http_cluster_init( Json(body): Json, ) -> Result<(StatusCode, Json), (StatusCode, Json)> { let ctx = state.op_context().await.map_err(|e| err_502(format!("{e}")))?; - let summary = ops::cluster::init(&ctx, &body.nodes) - .await - .map_err(map_config_err)?; + let summary = if state.config_engine.is_some() || state.web_mode.is_some() { + let path = state.runtime_root.join("bootstrap-intent.toml"); + ops::cluster::init_with_intent(&ctx, &body.nodes, &path).await + } else { + ops::cluster::init(&ctx, &body.nodes).await + } + .map_err(map_config_err)?; state.commit_op_context(&ctx).map_err(map_persist_err)?; // The cluster is now live — re-seed the shared kv_client with the diff --git a/app/crowdb-web/tests/mgmt_routes_test.rs b/app/crowdb-web/tests/mgmt_routes_test.rs index 69096003..4cb8afe1 100644 --- a/app/crowdb-web/tests/mgmt_routes_test.rs +++ b/app/crowdb-web/tests/mgmt_routes_test.rs @@ -84,6 +84,13 @@ async fn spawn_upstream() -> Option { } async fn spawn_web(upstream: &Upstream) -> SocketAddr { + spawn_web_with_config_path(upstream, None).await +} + +async fn spawn_web_with_config_path( + upstream: &Upstream, + config_path: Option, +) -> SocketAddr { let listener = tokio::net::TcpListener::bind(SocketAddr::from(([127, 0, 0, 1], 0))) .await .expect("bind"); @@ -119,7 +126,7 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { no_fsync: false, }) .unwrap(); - let state = AppState::with_config(cfg, None); + let state = AppState::with_config(cfg, config_path); // Register the upstream's pid so `refresh_node_cache` (which skips // nodes with no tracked runtime pid) refreshes after mutations. state.set_runtime_pid(1, upstream.pid); @@ -130,6 +137,25 @@ async fn spawn_web(upstream: &Upstream) -> SocketAddr { addr } +#[tokio::test] +async fn persistent_web_bootstrap_clears_verified_intent() { + let Some(upstream) = spawn_upstream().await else { + eprintln!("skipping: crowdb-kv-server binary not built"); + return; + }; + let config_path = upstream.workspace.join("console.toml"); + let web = spawn_web_with_config_path(&upstream, Some(config_path.clone())).await; + let response = reqwest::Client::new() + .post(format!("http://{web}/api/cluster/init")) + .json(&json!({"nodes": [1]})) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 201, "{}", response.text().await.unwrap()); + assert!(config_path.exists()); + assert!(!upstream.workspace.join("bootstrap-intent.toml").exists()); +} + #[tokio::test] #[allow(clippy::too_many_lines)] async fn full_mgmt_cycle_through_web_routes() { diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index b187f1bc..fcf3af93 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -140,9 +140,12 @@ configuration, documentation and crash-diagnostics tasks below are pending. paths or inline SSH secrets. It is atomically sealed with mode 0600; interrupted retries restore a fresh in-memory context, reject changed topology before mutating Group 0, then delete the intent only after - confirmed publication. Focused real Group 0 tests pass. CLI, Web and S3 - mini-cluster callers still need to use this path before removing the mixed - persisted config. + confirmed publication. Persistent CLI and Web cluster-init callers, including + versioned Web process mode, now use this path. Real Web and CLI regressions + confirm the intent is removed after successful Group 0 publication. The old + mixed console file is still written + afterward; the launch-registry bootstrap path and S3 mini-cluster caller + still need cutover before that file can be removed. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with From 520a6604a5239e20e5e624b9e57274b4ceb9a666 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 11:21:35 +0800 Subject: [PATCH 45/74] Publish node membership with conditional rack update --- doc/working/plan-console-authority.md | 4 + lib/crowdb-console-shared/src/ops/hardware.rs | 32 +++-- .../src/ops/hardware/authority.rs | 119 +++++++++++++++++- .../tests/ops_hardware_authority_test.rs | 62 +++++++++ 4 files changed, 198 insertions(+), 19 deletions(-) diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index fcf3af93..f2211485 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -115,6 +115,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. pass. A real RPC proxy drops the committed rack write reply; a confirmed linearizable read recovers the successful outcome. Registry-mode Web, deletions, disks and legacy bootstrap still remain. + Node creation now conditionally updates rack membership and creates the node + in one Group 0 batch. Two concurrent consoles retain both child IDs; a + repeated rack add preserves its existing children. Retried node creation + compares immutable connection identity while preserving live status fields. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. diff --git a/lib/crowdb-console-shared/src/ops/hardware.rs b/lib/crowdb-console-shared/src/ops/hardware.rs index cec0220c..00c3212d 100644 --- a/lib/crowdb-console-shared/src/ops/hardware.rs +++ b/lib/crowdb-console-shared/src/ops/hardware.rs @@ -26,7 +26,7 @@ pub async fn add_rack_to_group0(ctx: &OpContext, rack_id: u64, name: &str) -> Re id: rack_id, name: name.to_owned(), }; - authority::create( + let created = authority::create( ctx, crowdb_protocol::key::RackKey { rack_id }, &RackValue { @@ -35,7 +35,18 @@ pub async fn add_rack_to_group0(ctx: &OpContext, rack_id: u64, name: &str) -> Re name: name.to_owned(), }, ) - .await?; + .await; + if let Err(Error::Conflict { .. }) = &created { + if ctx + .sysmd() + .get_rack(rack_id) + .await? + .is_some_and(|rack| rack.name == name) + { + return Ok(entry); + } + } + created?; Ok(entry) } @@ -44,13 +55,6 @@ pub async fn add_rack_to_group0(ctx: &OpContext, rack_id: u64, name: &str) -> Re /// # Errors /// Returns a missing rack, conflicting node, or authority error. pub async fn add_node_to_group0(ctx: &OpContext, entry: NodeEntry) -> Result { - authority::ready(ctx).await?; - if ctx.sysmd().get_rack(entry.rack_id).await?.is_none() { - return Err(Error::NotFound { - kind: "rack".into(), - id: entry.rack_id.to_string(), - }); - } let value = NodeValue { status: HwStatus::Up as i32, management_host: entry.host.clone(), @@ -59,15 +63,7 @@ pub async fn add_node_to_group0(ctx: &OpContext, entry: NodeEntry) -> Result(ctx: &OpContext, key: impl TextK Err(error) => Err(error.into()), } } + +/// Update the rack membership and create its node in one conditional Group 0 write. +pub(super) async fn create_node( + ctx: &OpContext, + rack_id: u64, + node_id: u64, + value: &NodeValue, +) -> Result<()> { + ready(ctx).await?; + let rack_path = RackKey { rack_id }.to_path(); + let node_path = NodeKey { rack_id, node_id }.to_path(); + for attempt in 0..10u64 { + let (mut rack, revision) = match ctx + .kv() + .get(0, 0, rack_path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, revision, .. } => ( + serde_json::from_slice::(&value) + .map_err(|error| Error::Config(error.to_string()))?, + revision, + ), + GetOutcome::NotFound => { + return Err(Error::NotFound { + kind: "rack".into(), + id: rack_id.to_string(), + }) + } + }; + let existing = ctx + .kv() + .get(0, 0, node_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + let node_exists = matches!(existing, GetOutcome::Found { .. }); + if let GetOutcome::Found { value: stored, .. } = existing { + let actual: NodeValue = + serde_json::from_slice(&stored).map_err(|error| Error::Config(error.to_string()))?; + if !same_node_connection(&actual, value) { + return Err(Error::Conflict { + kind: "node".into(), + id: node_id.to_string(), + }); + } + if rack.node_ids.contains(&node_id) { + return Ok(()); + } + } + if !rack.node_ids.contains(&node_id) { + rack.node_ids.push(node_id); + rack.node_ids.sort_unstable(); + } + let rack_bytes = serde_json::to_vec(&rack).map_err(|error| Error::Config(error.to_string()))?; + let mut ops = vec![BatchOp::Put { + key: rack_path.as_bytes().to_vec().into(), + value: rack_bytes.into(), + }]; + if !node_exists { + let node_bytes = serde_json::to_vec(value).map_err(|error| Error::Config(error.to_string()))?; + ops.push(BatchOp::Put { + key: node_path.as_bytes().to_vec().into(), + value: node_bytes.into(), + }); + } + match ctx + .kv() + .batch_write_cas(0, 0, &ops, rack_path.as_bytes(), revision) + .await + { + Ok(_) => return Ok(()), + Err(KvError::CasFailed { .. } | KvError::CasBusy) => { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1) * 5)).await; + } + Err(KvError::OutcomeUnknown) => { + let confirmed = ctx + .kv() + .get(0, 0, node_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + if let GetOutcome::Found { value: stored, .. } = confirmed { + let actual: NodeValue = + serde_json::from_slice(&stored).map_err(|error| Error::Config(error.to_string()))?; + if same_node_connection(&actual, value) { + let rack = ctx + .kv() + .get(0, 0, rack_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + if let GetOutcome::Found { value, .. } = rack { + let rack: RackValue = serde_json::from_slice(&value) + .map_err(|error| Error::Config(error.to_string()))?; + if rack.node_ids.contains(&node_id) { + return Ok(()); + } + } + return Err(KvError::OutcomeUnknown.into()); + } + return Err(Error::Conflict { + kind: "node".into(), + id: node_id.to_string(), + }); + } + return Err(KvError::OutcomeUnknown.into()); + } + Err(error) => return Err(error.into()), + } + } + Err(KvError::CasBusy.into()) +} + +fn same_node_connection(actual: &NodeValue, intended: &NodeValue) -> bool { + actual.management_host == intended.management_host + && actual.ssh_port == intended.ssh_port + && actual.ssh_user == intended.ssh_user + && actual.ssh_credential_ref == intended.ssh_credential_ref +} diff --git a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs index c11e9887..99272b8c 100644 --- a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs +++ b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs @@ -60,6 +60,20 @@ async fn separate_consoles_confirm_matching_hardware_and_reject_conflicts() { }; hardware::add_node_to_group0(&first, node.clone()).await.unwrap(); hardware::add_node_to_group0(&second, node.clone()).await.unwrap(); + second + .sysmd() + .set_node_status(2, 2, crowdb_protocol::common::HwStatus::Maintenance) + .await + .unwrap(); + hardware::add_node_to_group0(&first, node.clone()).await.unwrap(); + assert_eq!( + first.sysmd().get_node(2, 2).await.unwrap().unwrap().status, + crowdb_protocol::common::HwStatus::Maintenance as i32 + ); + assert_eq!( + second.sysmd().get_rack(2).await.unwrap().unwrap().node_ids, + vec![2] + ); let changed = NodeEntry { host: "10.0.0.3".into(), ..node @@ -75,6 +89,54 @@ async fn separate_consoles_confirm_matching_hardware_and_reject_conflicts() { assert!(listed[0].ssh_key.is_none()); assert!(listed[0].ssh_password.is_none()); assert!(second.config().nodes.is_empty()); + + // A repeated rack create preserves its confirmed child membership. + hardware::add_rack_to_group0(&first, 2, "rack-two").await.unwrap(); + assert_eq!( + first.sysmd().get_rack(2).await.unwrap().unwrap().node_ids, + vec![2] + ); +} + +#[tokio::test] +async fn concurrent_node_creates_preserve_both_rack_memberships() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + cluster_ops::init(&bootstrap, &[1]).await.unwrap(); + let first = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + let second = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + hardware::add_rack_to_group0(&first, 7, "rack-seven") + .await + .unwrap(); + let node = |id| NodeEntry { + id, + rack_id: 7, + host: format!("10.0.0.{id}"), + ssh_port: 22, + ssh_user: String::new(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: None, + }; + let (a, b) = tokio::join!( + hardware::add_node_to_group0(&first, node(8)), + hardware::add_node_to_group0(&second, node(9)) + ); + a.unwrap(); + b.unwrap(); + assert_eq!( + first.sysmd().get_rack(7).await.unwrap().unwrap().node_ids, + vec![8, 9] + ); + assert_eq!(first.sysmd().list_nodes_in_rack(7).await.unwrap().len(), 2); } #[tokio::test] From 38fb0add1be3aa1dcdaad2ba51f3b738523e53c4 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 11:23:58 +0800 Subject: [PATCH 46/74] Confirm empty rack removal in Group 0 --- .../src/commands/cluster/hardware.rs | 13 ++++- .../tests/launch_registry_cli_test.rs | 7 +++ doc/working/plan-console-authority.md | 2 + lib/crowdb-console-shared/src/ops/hardware.rs | 8 +++ .../src/ops/hardware/authority.rs | 56 +++++++++++++++++++ .../tests/ops_hardware_authority_test.rs | 8 +++ 6 files changed, 91 insertions(+), 3 deletions(-) diff --git a/app/crowdb-cli/src/commands/cluster/hardware.rs b/app/crowdb-cli/src/commands/cluster/hardware.rs index 4038989f..a82a139d 100644 --- a/app/crowdb-cli/src/commands/cluster/hardware.rs +++ b/app/crowdb-cli/src/commands/cluster/hardware.rs @@ -77,10 +77,17 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { Ok(c) => c, Err(c) => return c, }; - match crowdb_console_shared::ops::hardware::remove_rack(&ctx, rack_id).await { + let result = if cli.registry.is_some() { + crowdb_console_shared::ops::hardware::remove_rack_from_group0(&ctx, rack_id).await + } else { + crowdb_console_shared::ops::hardware::remove_rack(&ctx, rack_id).await + }; + match result { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; + if cli.registry.is_none() { + if let Err(c) = commit_config(cli, &ctx) { + return c; + } } println!("removed rack {id}"); ExitCode::SUCCESS diff --git a/app/crowdb-cli/tests/launch_registry_cli_test.rs b/app/crowdb-cli/tests/launch_registry_cli_test.rs index 264201a2..1e67f3af 100644 --- a/app/crowdb-cli/tests/launch_registry_cli_test.rs +++ b/app/crowdb-cli/tests/launch_registry_cli_test.rs @@ -208,5 +208,12 @@ async fn registry_hardware_uses_group_zero_across_cli_invocations() { ], ); assert!(run_command(&path, g0.mgmt_port, &["cluster", "node", "list"]).contains("10.0.0.2")); + run_command( + &path, + g0.mgmt_port, + &["cluster", "rack", "add", "--id", "3", "--name", "empty"], + ); + run_command(&path, g0.mgmt_port, &["cluster", "rack", "remove", "--id", "3"]); + assert!(!run_command(&path, g0.mgmt_port, &["cluster", "rack", "list"]).contains("empty")); assert!(!dir.path().join("invalid-legacy.toml").exists()); } diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index f2211485..46d40122 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -119,6 +119,8 @@ configuration, documentation and crash-diagnostics tasks below are pending. in one Group 0 batch. Two concurrent consoles retain both child IDs; a repeated rack add preserves its existing children. Retried node creation compares immutable connection identity while preserving live status fields. + Registry CLI rack removal now conditionally deletes only a confirmed empty + rack; shared and real CLI regressions cover child conflict and absence. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. diff --git a/lib/crowdb-console-shared/src/ops/hardware.rs b/lib/crowdb-console-shared/src/ops/hardware.rs index 00c3212d..e48639c3 100644 --- a/lib/crowdb-console-shared/src/ops/hardware.rs +++ b/lib/crowdb-console-shared/src/ops/hardware.rs @@ -111,6 +111,14 @@ pub async fn list_nodes_from_group0(ctx: &OpContext, rack_id: Option) -> Re Ok(nodes) } +/// Remove an empty rack after Group 0 confirms no node belongs to it. +/// +/// # Errors +/// Returns a conflict if children remain, or the authority error. +pub async fn remove_rack_from_group0(ctx: &OpContext, rack_id: u64) -> Result<()> { + authority::remove_empty_rack(ctx, rack_id).await +} + // ── rack ──────────────────────────────────────────────────────── /// Add a rack to the local config and group-0 sysdata. diff --git a/lib/crowdb-console-shared/src/ops/hardware/authority.rs b/lib/crowdb-console-shared/src/ops/hardware/authority.rs index 724ac7d2..d25b3c40 100644 --- a/lib/crowdb-console-shared/src/ops/hardware/authority.rs +++ b/lib/crowdb-console-shared/src/ops/hardware/authority.rs @@ -162,3 +162,59 @@ fn same_node_connection(actual: &NodeValue, intended: &NodeValue) -> bool { && actual.ssh_user == intended.ssh_user && actual.ssh_credential_ref == intended.ssh_credential_ref } + +/// Delete an empty rack only while its confirmed revision still matches. +pub(super) async fn remove_empty_rack(ctx: &OpContext, rack_id: u64) -> Result<()> { + ready(ctx).await?; + let path = RackKey { rack_id }.to_path(); + for attempt in 0..10u64 { + let (rack, revision) = match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, revision } => ( + serde_json::from_slice::(&value) + .map_err(|error| Error::Config(error.to_string()))?, + revision, + ), + GetOutcome::NotFound => { + return Err(Error::NotFound { + kind: "rack".into(), + id: rack_id.to_string(), + }) + } + }; + if !rack.node_ids.is_empty() || !ctx.sysmd().list_nodes_in_rack(rack_id).await?.is_empty() { + return Err(Error::Conflict { + kind: "rack with nodes".into(), + id: rack_id.to_string(), + }); + } + let ops = [BatchOp::Delete { + key: path.as_bytes().to_vec().into(), + }]; + match ctx + .kv() + .batch_write_cas(0, 0, &ops, path.as_bytes(), revision) + .await + { + Ok(_) => return Ok(()), + Err(KvError::CasFailed { .. } | KvError::CasBusy) => { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1) * 5)).await; + } + Err(KvError::OutcomeUnknown) => { + return match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::NotFound => Ok(()), + GetOutcome::Found { .. } => Err(KvError::OutcomeUnknown.into()), + }; + } + Err(error) => return Err(error.into()), + } + } + Err(KvError::CasBusy.into()) +} diff --git a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs index 99272b8c..e1508d7a 100644 --- a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs +++ b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs @@ -90,6 +90,14 @@ async fn separate_consoles_confirm_matching_hardware_and_reject_conflicts() { assert!(listed[0].ssh_password.is_none()); assert!(second.config().nodes.is_empty()); + let occupied = hardware::remove_rack_from_group0(&second, 2).await.unwrap_err(); + assert!(matches!(occupied, Error::Conflict { .. }), "{occupied:?}"); + assert!(second.sysmd().get_rack(2).await.unwrap().is_some()); + + hardware::add_rack_to_group0(&first, 3, "empty").await.unwrap(); + hardware::remove_rack_from_group0(&second, 3).await.unwrap(); + assert!(first.sysmd().get_rack(3).await.unwrap().is_none()); + // A repeated rack create preserves its confirmed child membership. hardware::add_rack_to_group0(&first, 2, "rack-two").await.unwrap(); assert_eq!( From 8ef3532f3ba0b93d314dd76bdd8fc48152ce17c2 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 11:30:52 +0800 Subject: [PATCH 47/74] Expose confirmed hardware routes in bare metal Web --- app/crowdb-web/src/lib.rs | 20 ++- app/crowdb-web/src/managed_hardware.rs | 89 ++++++++++++ app/crowdb-web/src/managed_logical.rs | 2 +- .../tests/bare_metal_authority_test.rs | 128 +++++++++++++++++- doc/working/plan-console-authority.md | 17 ++- 5 files changed, 247 insertions(+), 9 deletions(-) create mode 100644 app/crowdb-web/src/managed_hardware.rs diff --git a/app/crowdb-web/src/lib.rs b/app/crowdb-web/src/lib.rs index 9a8986b9..35793221 100644 --- a/app/crowdb-web/src/lib.rs +++ b/app/crowdb-web/src/lib.rs @@ -19,6 +19,7 @@ pub mod kv; mod launch; pub mod lifecycle; mod managed; +mod managed_hardware; mod managed_logical; pub mod mgmt; pub mod owner_assignment; @@ -75,7 +76,24 @@ pub fn router(state: AppState) -> axum::Router { .route("/api/*path", any(health::managed_api_unavailable)) .fallback(spa::spa_fallback); let managed = if state.web_mode == Some(crowdb_console_shared::config::web::WebMode::BareMetal) { - managed.merge(launch::routes().route_layer(authorization)) + let hardware = axum::Router::new() + .route( + "/api/racks", + get(managed_hardware::list_racks) + .merge(post(managed_hardware::add_rack).route_layer(authorization.clone())), + ) + .route( + "/api/racks/:rack_id", + delete(managed_hardware::remove_rack).route_layer(authorization.clone()), + ) + .route( + "/api/nodes", + get(managed_hardware::list_nodes) + .merge(post(managed_hardware::add_node).route_layer(authorization.clone())), + ); + managed + .merge(hardware) + .merge(launch::routes().route_layer(authorization)) } else { managed }; diff --git a/app/crowdb-web/src/managed_hardware.rs b/app/crowdb-web/src/managed_hardware.rs new file mode 100644 index 00000000..eae11761 --- /dev/null +++ b/app/crowdb-web/src/managed_hardware.rs @@ -0,0 +1,89 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Bare-metal hardware routes backed by confirmed Group 0 state. + +use axum::extract::{Path, Query, State}; +use axum::http::StatusCode; +use axum::Json; +use crowdb_console_shared::config::{NodeEntry, RackEntry}; +use crowdb_console_shared::ops::hardware; +use serde::Deserialize; + +use crate::error::ErrorBody; +use crate::managed_logical::api_error; +use crate::state::AppState; + +type ApiError = (StatusCode, Json); + +#[derive(Deserialize)] +pub(crate) struct CreateRack { + id: u64, + #[serde(default)] + name: String, +} + +#[derive(Deserialize)] +pub(crate) struct NodeFilter { + rack_id: Option, +} + +pub(crate) async fn list_racks(State(state): State) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_racks_from_group0(&ctx) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn add_rack( + State(state): State, + Json(body): Json, +) -> Result<(StatusCode, Json), ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + let rack = hardware::add_rack_to_group0(&ctx, body.id, &body.name) + .await + .map_err(api_error)?; + Ok((StatusCode::CREATED, Json(rack))) +} + +pub(crate) async fn remove_rack( + State(state): State, + Path(id): Path, +) -> Result { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::remove_rack_from_group0(&ctx, id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} + +pub(crate) async fn list_nodes( + State(state): State, + Query(filter): Query, +) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_nodes_from_group0(&ctx, filter.rack_id) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn add_node( + State(state): State, + Json(node): Json, +) -> Result<(StatusCode, Json), ApiError> { + if node.ssh_key.is_some() || node.ssh_password.is_some() { + return Err(( + StatusCode::BAD_REQUEST, + Json(ErrorBody { + error: "inline SSH secrets are not accepted; use ssh_credential_ref".into(), + }), + )); + } + let ctx = state.op_context().await.map_err(api_error)?; + let node = hardware::add_node_to_group0(&ctx, node) + .await + .map_err(api_error)?; + Ok((StatusCode::CREATED, Json(node))) +} diff --git a/app/crowdb-web/src/managed_logical.rs b/app/crowdb-web/src/managed_logical.rs index 648d9bd3..f412dc7b 100644 --- a/app/crowdb-web/src/managed_logical.rs +++ b/app/crowdb-web/src/managed_logical.rs @@ -13,7 +13,7 @@ use crate::state::AppState; type ApiError = (StatusCode, Json); #[allow(clippy::needless_pass_by_value)] -fn api_error(error: Error) -> ApiError { +pub(crate) fn api_error(error: Error) -> ApiError { let status = match error { Error::NotFound { .. } => StatusCode::NOT_FOUND, Error::Conflict { .. } => StatusCode::CONFLICT, diff --git a/app/crowdb-web/tests/bare_metal_authority_test.rs b/app/crowdb-web/tests/bare_metal_authority_test.rs index dab6eadc..cab0ddfe 100644 --- a/app/crowdb-web/tests/bare_metal_authority_test.rs +++ b/app/crowdb-web/tests/bare_metal_authority_test.rs @@ -95,7 +95,127 @@ fn application(cluster: &KvCluster) -> axum::Router { log_max_files: 5, request_timeout_ms: Some(500), }; - router(AppState::default().with_process_config(&config)) + router( + AppState::default() + .with_process_config(&config) + .with_management_token("bare-metal-test-token-123456789012345".into()) + .unwrap(), + ) +} + +async fn hardware_request( + app: &axum::Router, + method: axum::http::Method, + path: &str, + body: Option, + authenticated: bool, +) -> (StatusCode, serde_json::Value) { + let mut request = Request::builder().method(method).uri(path); + if authenticated { + request = request.header("authorization", "Bearer bare-metal-test-token-123456789012345"); + } + let request = if let Some(body) = body { + request + .header("content-type", "application/json") + .body(Body::from(body.to_string())) + .unwrap() + } else { + request.body(Body::empty()).unwrap() + }; + let response = app.clone().oneshot(request).await.unwrap(); + let status = response.status(); + let bytes = axum::body::to_bytes(response.into_body(), 1024 * 1024) + .await + .unwrap(); + let value = if bytes.is_empty() { + serde_json::Value::Null + } else { + serde_json::from_slice(&bytes).unwrap() + }; + (status, value) +} + +#[tokio::test] +async fn bare_metal_hardware_routes_share_confirmed_group_zero_state() { + let cluster = KvCluster::start().await; + initialized_authority(&cluster).await; + let first = application(&cluster); + let second = application(&cluster); + let rack = serde_json::json!({"id": 8, "name": "rack-eight"}); + assert_eq!( + hardware_request( + &first, + axum::http::Method::POST, + "/api/racks", + Some(rack.clone()), + false + ) + .await + .0, + StatusCode::UNAUTHORIZED + ); + assert_eq!( + hardware_request(&first, axum::http::Method::POST, "/api/racks", Some(rack), true) + .await + .0, + StatusCode::CREATED + ); + let (_, racks) = hardware_request(&second, axum::http::Method::GET, "/api/racks", None, false).await; + assert!(racks + .as_array() + .unwrap() + .iter() + .any(|rack| rack["id"] == 8 && rack["name"] == "rack-eight")); + let node = serde_json::json!({"id": 9, "rack_id": 8, "host": "node-nine.example", "ssh_port": 2222, "ssh_user": "operator", "ssh_credential_ref": "ops-key"}); + let mut secret_node = node.clone(); + secret_node["ssh_key"] = serde_json::json!("/tmp/private-key"); + assert_eq!( + hardware_request( + &first, + axum::http::Method::POST, + "/api/nodes", + Some(secret_node), + true + ) + .await + .0, + StatusCode::BAD_REQUEST + ); + assert_eq!( + hardware_request(&first, axum::http::Method::POST, "/api/nodes", Some(node), true) + .await + .0, + StatusCode::CREATED + ); + let (_, nodes) = hardware_request( + &second, + axum::http::Method::GET, + "/api/nodes?rack_id=8", + None, + false, + ) + .await; + assert_eq!(nodes[0]["host"], "node-nine.example"); + assert_eq!(nodes[0]["ssh_credential_ref"], "ops-key"); + assert!(nodes[0].get("ssh_key").is_none()); + assert_eq!( + hardware_request(&second, axum::http::Method::DELETE, "/api/racks/8", None, true) + .await + .0, + StatusCode::CONFLICT + ); + assert_eq!( + hardware_request( + &second, + axum::http::Method::POST, + "/api/racks", + Some(serde_json::json!({"id": 8, "name": "changed"})), + true + ) + .await + .0, + StatusCode::CONFLICT + ); } async fn unavailable(app: &axum::Router) { @@ -154,6 +274,12 @@ async fn bare_metal_snapshot_requires_live_authority_without_a_docker_monitor() drop(cluster); unavailable(&app).await; + assert_eq!( + hardware_request(&app, axum::http::Method::GET, "/api/racks", None, false) + .await + .0, + StatusCode::SERVICE_UNAVAILABLE + ); } #[tokio::test] diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 46d40122..f54d0897 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -121,6 +121,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. compares immutable connection identity while preserving live status fields. Registry CLI rack removal now conditionally deletes only a confirmed empty rack; shared and real CLI regressions cover child conflict and absence. + Bare-metal Web now exposes authenticated Group 0 rack creation/removal and + node creation, with public confirmed rack/node reads. Two Web instances + observe the same records; inline SSH material is rejected. Node deletion, + disks and remaining legacy routes still need conversion. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. @@ -146,12 +150,13 @@ configuration, documentation and crash-diagnostics tasks below are pending. paths or inline SSH secrets. It is atomically sealed with mode 0600; interrupted retries restore a fresh in-memory context, reject changed topology before mutating Group 0, then delete the intent only after - confirmed publication. Persistent CLI and Web cluster-init callers, including - versioned Web process mode, now use this path. Real Web and CLI regressions - confirm the intent is removed after successful Group 0 publication. The old - mixed console file is still written - afterward; the launch-registry bootstrap path and S3 mini-cluster caller - still need cutover before that file can be removed. + confirmed publication. Persistent CLI and legacy Web cluster-init callers + now use this path. Real Web and CLI regressions confirm the intent is removed + after successful Group 0 publication. The versioned Web handler is prepared + to use the intent, but its managed router does not expose cluster init yet. + The old mixed console file is still written afterward; the launch-registry + bootstrap path and S3 mini-cluster caller still need cutover before that + file can be removed. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with From 90dd7e569b1f3d016d93ece5ec1c7d87263babd8 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 11:33:42 +0800 Subject: [PATCH 48/74] Bootstrap registry clusters from sealed topology input --- app/crowdb-cli/src/commands/cluster.rs | 44 +++++++++++++++--- .../tests/launch_registry_cli_test.rs | 45 +++++++++++++++++++ doc/working/plan-console-authority.md | 8 ++-- 3 files changed, 89 insertions(+), 8 deletions(-) diff --git a/app/crowdb-cli/src/commands/cluster.rs b/app/crowdb-cli/src/commands/cluster.rs index 655acc38..d81c4909 100644 --- a/app/crowdb-cli/src/commands/cluster.rs +++ b/app/crowdb-cli/src/commands/cluster.rs @@ -24,6 +24,9 @@ pub enum ClusterVerb { Init { #[arg(short = 'n', long, value_delimiter = ',')] nodes: Vec, + /// Versioned bootstrap topology input for the first registry-mode init. + #[arg(long, value_name = "PATH")] + bootstrap_file: Option, }, /// Deploy a local N-node KV cluster on 127.0.0.1 (forks /// `crowdb-kv-server` on each node, bootstraps group 0). @@ -180,7 +183,10 @@ pub enum ClusterVerb { #[allow(clippy::too_many_lines)] pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { match verb { - ClusterVerb::Init { nodes } => { + ClusterVerb::Init { + nodes, + bootstrap_file, + } => { let ctx = match op_context(cli) { Ok(c) => c, Err(c) => return c, @@ -192,16 +198,44 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { return ExitCode::from(1); } }; - let result = if cli.registry.is_some() { - crowdb_console_shared::ops::cluster::init(&ctx, &node_ids).await + let result = if let Some(registry) = &cli.registry { + let sealed_path = registry.with_extension("bootstrap-intent.toml"); + if let Some(source) = bootstrap_file { + let intent = match crowdb_console_shared::bootstrap_intent::BootstrapIntent::load(&source) + { + Ok(intent) => intent, + Err(error) => { + eprintln!("error: load bootstrap file: {error}"); + return ExitCode::from(2); + } + }; + if intent.members() != node_ids.as_slice() { + eprintln!("error: bootstrap file members differ from --nodes"); + return ExitCode::from(1); + } + if let Err(error) = intent.seal(&sealed_path) { + eprintln!("error: seal bootstrap intent: {error}"); + return ExitCode::from(2); + } + } else if !sealed_path.exists() { + eprintln!("error: --bootstrap-file is required for the first registry-mode init"); + return ExitCode::from(1); + } + crowdb_console_shared::ops::cluster::init_with_intent(&ctx, &node_ids, &sealed_path).await } else { + if bootstrap_file.is_some() { + eprintln!("error: --bootstrap-file requires --registry"); + return ExitCode::from(1); + } let intent_path = config_path().with_extension("bootstrap.toml"); crowdb_console_shared::ops::cluster::init_with_intent(&ctx, &node_ids, &intent_path).await }; match result { Ok(summary) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; + if cli.registry.is_none() { + if let Err(c) = commit_config(cli, &ctx) { + return c; + } } println!( "cluster initialized: store {}, group {}, {} nodes", diff --git a/app/crowdb-cli/tests/launch_registry_cli_test.rs b/app/crowdb-cli/tests/launch_registry_cli_test.rs index 1e67f3af..96167f4b 100644 --- a/app/crowdb-cli/tests/launch_registry_cli_test.rs +++ b/app/crowdb-cli/tests/launch_registry_cli_test.rs @@ -10,6 +10,7 @@ use std::path::Path; use std::process::Command; use std::time::Duration; +use crowdb_console_shared::bootstrap_intent::BootstrapIntent; use crowdb_console_shared::config::web::{LaunchRecord, LaunchRegistry}; use crowdb_console_shared::launch::LaunchRuntime; use crowdb_console_shared::lifecycle; @@ -217,3 +218,47 @@ async fn registry_hardware_uses_group_zero_across_cli_invocations() { assert!(!run_command(&path, g0.mgmt_port, &["cluster", "rack", "list"]).contains("empty")); assert!(!dir.path().join("invalid-legacy.toml").exists()); } + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn registry_bootstrap_uses_sealed_intent_without_legacy_topology_file() { + let g0 = common::direct::spawn_group0() + .await + .expect("KV server binary must be built"); + let dir = tempdir_in_test_data("cli-registry-bootstrap"); + let path = dir.path().join("launches.toml"); + LaunchRegistry { + version: 1, + launches: Vec::new(), + } + .save(&path) + .unwrap(); + let source = dir.path().join("bootstrap-source.toml"); + let config = crowdb_console_shared::ConsoleConfig::load(&g0.config_path).unwrap(); + let intent = BootstrapIntent::capture(&config, &[1]).unwrap(); + intent.seal(&source).unwrap(); + std::fs::write(dir.path().join("invalid-legacy.toml"), "invalid legacy config").unwrap(); + + run_command( + &path, + g0.mgmt_port, + &[ + "cluster", + "init", + "--nodes", + "1", + "--bootstrap-file", + source.to_str().unwrap(), + ], + ); + assert!(!path.with_extension("bootstrap-intent.toml").exists()); + intent + .seal(&path.with_extension("bootstrap-intent.toml")) + .unwrap(); + run_command(&path, g0.mgmt_port, &["cluster", "init", "--nodes", "1"]); + assert!(!path.with_extension("bootstrap-intent.toml").exists()); + assert_eq!( + std::fs::read_to_string(dir.path().join("invalid-legacy.toml")).unwrap(), + "invalid legacy config" + ); + assert!(run_command(&path, g0.mgmt_port, &["cluster", "rack", "list"]).contains("rack-1")); +} diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index f54d0897..48b0145a 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -154,9 +154,11 @@ configuration, documentation and crash-diagnostics tasks below are pending. now use this path. Real Web and CLI regressions confirm the intent is removed after successful Group 0 publication. The versioned Web handler is prepared to use the intent, but its managed router does not expose cluster init yet. - The old mixed console file is still written afterward; the launch-registry - bootstrap path and S3 mini-cluster caller still need cutover before that - file can be removed. + Registry CLI now accepts a versioned bootstrap input, seals an immutable + retry copy beside the launch registry, runs the same confirmation path and + deletes that copy after success. It does not write the mixed console file. + The legacy CLI/Web path still writes that file; S3 mini-cluster and + versioned Web bootstrap still need cutover before it can be removed. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with From c028a9539298dd4f73d8d1404109a663d6226bde Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 11:43:29 +0800 Subject: [PATCH 49/74] Enable sealed bootstrap in bare metal Web --- app/crowdb-web/src/lib.rs | 4 ++ app/crowdb-web/src/mgmt/cluster_init.rs | 21 +++++++ app/crowdb-web/tests/mgmt_routes_test.rs | 72 +++++++++++++++++++++--- doc/working/plan-console-authority.md | 11 ++-- 4 files changed, 94 insertions(+), 14 deletions(-) diff --git a/app/crowdb-web/src/lib.rs b/app/crowdb-web/src/lib.rs index 35793221..fd1d5e2d 100644 --- a/app/crowdb-web/src/lib.rs +++ b/app/crowdb-web/src/lib.rs @@ -77,6 +77,10 @@ pub fn router(state: AppState) -> axum::Router { .fallback(spa::spa_fallback); let managed = if state.web_mode == Some(crowdb_console_shared::config::web::WebMode::BareMetal) { let hardware = axum::Router::new() + .route( + "/api/cluster/init", + post(mgmt::http_cluster_init).route_layer(authorization.clone()), + ) .route( "/api/racks", get(managed_hardware::list_racks) diff --git a/app/crowdb-web/src/mgmt/cluster_init.rs b/app/crowdb-web/src/mgmt/cluster_init.rs index 35de565f..946c2021 100644 --- a/app/crowdb-web/src/mgmt/cluster_init.rs +++ b/app/crowdb-web/src/mgmt/cluster_init.rs @@ -19,6 +19,9 @@ pub(crate) struct ClusterInitBody { /// Must be non-empty. For a single node, group 0 self-elects. /// For multiple nodes, remotes are wired and election starts after. pub nodes: Vec, + /// Optional versioned bootstrap topology file for a first bare-metal init. + #[serde(default)] + pub bootstrap_file: Option, } /// `POST /api/cluster/init` — initialize the cluster by bootstrapping @@ -40,6 +43,24 @@ pub(crate) async fn http_cluster_init( let ctx = state.op_context().await.map_err(|e| err_502(format!("{e}")))?; let summary = if state.config_engine.is_some() || state.web_mode.is_some() { let path = state.runtime_root.join("bootstrap-intent.toml"); + if state.web_mode == Some(crowdb_console_shared::config::web::WebMode::BareMetal) { + if let Some(source) = &body.bootstrap_file { + let intent = crowdb_console_shared::bootstrap_intent::BootstrapIntent::load(source) + .map_err(map_config_err)?; + if intent.members() != body.nodes.as_slice() { + return Err(map_config_err(crowdb_console_shared::error::Error::Validation { + field: "nodes".into(), + message: "bootstrap file members differ from requested nodes".into(), + })); + } + intent.seal(&path).map_err(map_config_err)?; + } else if !path.exists() { + return Err(map_config_err(crowdb_console_shared::error::Error::Validation { + field: "bootstrap_file".into(), + message: "required for the first bare-metal cluster init".into(), + })); + } + } ops::cluster::init_with_intent(&ctx, &body.nodes, &path).await } else { ops::cluster::init(&ctx, &body.nodes).await diff --git a/app/crowdb-web/tests/mgmt_routes_test.rs b/app/crowdb-web/tests/mgmt_routes_test.rs index 4cb8afe1..8d65b5b7 100644 --- a/app/crowdb-web/tests/mgmt_routes_test.rs +++ b/app/crowdb-web/tests/mgmt_routes_test.rs @@ -11,7 +11,9 @@ use std::net::SocketAddr; use std::path::PathBuf; use std::time::Duration; +use crowdb_console_shared::bootstrap_intent::BootstrapIntent; use crowdb_console_shared::cluster::{NodeHealth, NodeStore}; +use crowdb_console_shared::config::web::{WebMode, WebProcessConfig}; use crowdb_console_shared::config::{NodeEntry, RackEntry, ServerEntry, ServiceType}; use crowdb_console_shared::lifecycle::{self, crowdb_kv_server_bin, stop_pid_with_timeout, DeployRequest}; use crowdb_console_shared::monitor::NodeRecord; @@ -95,6 +97,19 @@ async fn spawn_web_with_config_path( .await .expect("bind"); let addr = listener.local_addr().expect("local_addr"); + let cfg = config_for_upstream(upstream); + let state = AppState::with_config(cfg, config_path); + // Register the upstream's pid so `refresh_node_cache` (which skips + // nodes with no tracked runtime pid) refreshes after mutations. + state.set_runtime_pid(1, upstream.pid); + tokio::spawn(async move { + axum::serve(listener, router(state)).await.unwrap(); + }); + tokio::time::sleep(Duration::from_millis(50)).await; + addr +} + +fn config_for_upstream(upstream: &Upstream) -> ConsoleConfig { let mut cfg = ConsoleConfig::default(); cfg.racks.push(RackEntry { id: 1, @@ -126,15 +141,7 @@ async fn spawn_web_with_config_path( no_fsync: false, }) .unwrap(); - let state = AppState::with_config(cfg, config_path); - // Register the upstream's pid so `refresh_node_cache` (which skips - // nodes with no tracked runtime pid) refreshes after mutations. - state.set_runtime_pid(1, upstream.pid); - tokio::spawn(async move { - axum::serve(listener, router(state)).await.unwrap(); - }); - tokio::time::sleep(Duration::from_millis(50)).await; - addr + cfg } #[tokio::test] @@ -156,6 +163,53 @@ async fn persistent_web_bootstrap_clears_verified_intent() { assert!(!upstream.workspace.join("bootstrap-intent.toml").exists()); } +#[tokio::test] +async fn bare_metal_web_bootstrap_consumes_sealed_topology_input() { + let Some(upstream) = spawn_upstream().await else { + eprintln!("skipping: crowdb-kv-server binary not built"); + return; + }; + let source = upstream.workspace.join("bootstrap-source.toml"); + BootstrapIntent::capture(&config_for_upstream(&upstream), &[1]) + .unwrap() + .seal(&source) + .unwrap(); + let config = WebProcessConfig { + version: 1, + mode: WebMode::BareMetal, + bind: "127.0.0.1".into(), + port: 14000, + group0_management_seeds: vec![upstream.mgmt_url.clone()], + ui_root: "/tmp".into(), + monitor_status: None, + log_dir: "/tmp".into(), + log_max_file_mb: 30, + log_max_files: 5, + request_timeout_ms: Some(500), + }; + let state = AppState::with_config_engine(ConsoleConfig::default(), None, upstream.workspace.clone()) + .with_process_config(&config) + .with_management_token("bare-metal-bootstrap-test-token-12345".into()) + .unwrap(); + let listener = tokio::net::TcpListener::bind(SocketAddr::from(([127, 0, 0, 1], 0))) + .await + .unwrap(); + let addr = listener.local_addr().unwrap(); + tokio::spawn(async move { + axum::serve(listener, router(state)).await.unwrap(); + }); + let response = reqwest::Client::new() + .post(format!("http://{addr}/api/cluster/init")) + .bearer_auth("bare-metal-bootstrap-test-token-12345") + .json(&json!({"nodes": [1], "bootstrap_file": source})) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 201, "{}", response.text().await.unwrap()); + assert!(!upstream.workspace.join("bootstrap-intent.toml").exists()); + assert!(!upstream.workspace.join("console.toml").exists()); +} + #[tokio::test] #[allow(clippy::too_many_lines)] async fn full_mgmt_cycle_through_web_routes() { diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 48b0145a..79de22b1 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -152,13 +152,14 @@ configuration, documentation and crash-diagnostics tasks below are pending. topology before mutating Group 0, then delete the intent only after confirmed publication. Persistent CLI and legacy Web cluster-init callers now use this path. Real Web and CLI regressions confirm the intent is removed - after successful Group 0 publication. The versioned Web handler is prepared - to use the intent, but its managed router does not expose cluster init yet. - Registry CLI now accepts a versioned bootstrap input, seals an immutable + after successful Group 0 publication. Versioned bare-metal Web now exposes + an authenticated cluster-init route and accepts the same independent + bootstrap input without writing a mixed console file. Registry CLI accepts + a versioned bootstrap input, seals an immutable retry copy beside the launch registry, runs the same confirmation path and deletes that copy after success. It does not write the mixed console file. - The legacy CLI/Web path still writes that file; S3 mini-cluster and - versioned Web bootstrap still need cutover before it can be removed. + The legacy CLI/Web path still writes that file; S3 mini-cluster still needs + cutover before it can be removed. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with From c99e0e20f6c1af882679ced3d6a007866d950159 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 12:27:16 +0800 Subject: [PATCH 50/74] Clarify crash collector limits and release scope --- container/single-node-container/README.md | 20 +++++++-- doc/backlog/R188-console-group0-authority.md | 47 ++++++++++---------- doc/working/plan-console-authority.md | 26 +++++------ 3 files changed, 53 insertions(+), 40 deletions(-) diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index 646a9898..38663caf 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -72,10 +72,22 @@ make some stack frames unavailable. Core collection can also be suppressed by the host's dumpability policy, including for executables with file capabilities. The container never changes `core_pattern` or the host's dumpability policy. -- On a systemd-coredump host, use `coredumpctl list` and `coredumpctl dump` - on the host to locate and export a captured dump. -- On an Ubuntu Apport host, use the host's Apport report and core extraction - workflow. A pipe pattern does not create a volume file. +- On a systemd-coredump host, use `coredumpctl list crowdb-kv-server` to find + the host report, then `coredumpctl --output=/private/core dump + crowdb-kv-server` as an authorized host user to export it. Check that the + result is readable and nonempty before symbolization. +- On an Ubuntu Apport host, find the matching report in `/var/crash` on the + Docker host. Create a private directory, then run `sudo apport-unpack + /var/crash/REPORT.crash /private/crowdb-core/unpacked`. The extracted + `CoreDump` is the file to pass to the symbolizer. Apport reports may be + readable only by the host administrator; preserve the private permissions + when granting the debugging user access. A pipe pattern does not create a + volume file. Apport may fail to resolve a CROWDB executable because its + `/opt/crowdb/bin` path exists only inside the container; it may also ignore + executables outside host distribution packages. Check the host's Apport log + when no report appears. + If the report or `CoreDump` is absent, collection is unavailable for that + crash; do not substitute a log or an unrelated dump. - On Docker Desktop, inspect the Linux VM's collector. The desktop host's native crash directory is not the container's core directory. diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md index 8c68cd92..9a01f305 100644 --- a/doc/backlog/R188-console-group0-authority.md +++ b/doc/backlog/R188-console-group0-authority.md @@ -68,9 +68,11 @@ not block R187 completion. confirmed Group 0 state. If nonmember KV processes were launched before Group 0 exists, propagate usable Group 0 seed hints after initialization before treating their registration as live; seed hints are not topology. -7. Audit the S3 mini-cluster's local `console.toml` and restart path under the +7. Replace the S3 mini-cluster's local `console.toml` and restart path under the same authority boundary. Retain only launch inputs and bootstrap seeds - locally after Group 0 cutover; do not replay a local topology copy. + locally after Group 0 cutover; do not replay a local topology copy. This + configuration has not been released, so no migration or compatibility path + is needed. 8. Publish the verified bare-metal deployment and operations material under `/nv/cpp/crowdb-web/site/docs/`, organized by KV cluster, chunk layer, and data access servers. State that bare-metal is not yet production-ready. @@ -88,10 +90,12 @@ not block R187 completion. Optionally ship exact-build debug symbols as a separate GitHub Release asset generated from the same staged runtime as the image, indexed by version and source revision. Omit this large asset by default so its upload cannot block - image publication. The release preparation script in `tools/` runs manually, shows a - dry-run plan, updates versions, creates the tag and GitHub Release, then - dispatches the existing verified DockerHub publication workflow. Host - acceptance remains open; no host configuration change is assumed. + image publication. The release preparation script in `tools/` runs manually, + shows a dry-run plan, updates versions, creates the tag and GitHub Release, + then dispatches the existing verified DockerHub publication workflow. Its + actual use is deferred to the operator's later release; no release execution + is required for this requirement. Host crash acceptance remains open; no + host configuration change is assumed. ## Dependencies @@ -142,9 +146,10 @@ not block R187 completion. Group 0 is created and seed hints are propagated, assert each process registers exactly one live node identity before logical operations use it. Invariant: registration readiness. E2E test. -- Given a persisted S3 mini-cluster and a Group 0 outage, when it restarts or - tears down, assert local launch data cannot recreate or mask old cluster - topology. Invariant: no secondary authority. Integration test. +- Given an S3 mini-cluster started with the new launch-only configuration and + a Group 0 outage, when it restarts or tears down, assert local launch data + cannot recreate or mask cluster topology. Invariant: no secondary authority. + Integration test. - Given the two deployment guides and a reader following bare-metal steps, when the reader deploys KV, chunk services, and Iceberg or S3 access servers, assert each layer has a verified setup and health check, the non-production @@ -161,14 +166,6 @@ not block R187 completion. locates the dump or explicitly reports unsupported collection, without claiming an absent data-volume core. Invariant: truthful collector boundary. Integration test. -- Given a clean main checkout and a version bump, when the release tool runs in - dry-run mode, assert it shows every version change and no file or remote is - modified. When run for a release, assert the tag and GitHub Release identify - the same verified revision and image publication succeeds without symbol - upload. When symbols are requested, assert the asset contains source-line - information and GNU debuglink CRCs and SHA-256 hashes match the image's - stripped binaries. Invariant: optional released symbols come from the image - build and remain available after a build host changes. E2E test. Required gates: @@ -181,9 +178,13 @@ Required gates: ## Open Issues -- This host routes `core_pattern` to Apport, so a container-local directory and - core ulimit cannot guarantee a dump in `/opt/crowdb/data`. End-to-end - acceptance needs a disposable host with file-based collection or a verified - host-collector export workflow. Private one-core retention has passed local - tests, while collection and source-line symbolization on a real dump remain - unverified. +- This host routes `core_pattern` to Apport. A disposable container KV child + aborted and the monitor recovered it, but Apport did not create a CROWDB + report: its log says `/opt/crowdb/bin/crowdb-kv-server` does not exist on the + host. A packaged host program did produce an exportable `CoreDump`, proving + the extraction procedure without proving CROWDB collection. Exact-image + hashes, GNU debuglinks and source-line symbolization passed with a + debugger-generated CROWDB monitor core. Private one-core retention passed + local tests. File-based collection, real crash-core symbolization and + retention still need a disposable host with a relative `core_pattern`; no + change to this host's collector is assumed. diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 79de22b1..d9cc3026 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -90,11 +90,12 @@ configuration, documentation and crash-diagnostics tasks below are pending. Three-service lifecycle regressions and the complete CLI suite, fmt and clippy pass. Logs: `/tmp/crowdb-chunk-launch-*.log`. Removal of legacy startup/restore paths remains coupled to bootstrap cutover below. -- [~] **Remove mixed persistence**: remove the unreleased `ConsoleConfig` +- [ ] **Remove mixed persistence**: remove the unreleased `ConsoleConfig` parser/writer, inline SSH secrets, topology restoration and fixtures after the launch lifecycle and replay-safe bootstrap paths are wired. Preserve bootstrap intent independently until verified cutover. Update CLI commands, - Web persistence and S3 mini-cluster callers together; no compatibility reader. + Web persistence and S3 mini-cluster callers together; no compatibility reader + or migration path because the old configuration was never released. - [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through shared Group 0 hardware operations; preserve conflicts and uncertain writes without local-first commits. Docker keeps its hardware restrictions. Hardware @@ -138,7 +139,8 @@ configuration, documentation and crash-diagnostics tasks below are pending. - [ ] **Replay-safe bootstrap cutover**: persist bootstrap identity, verify committed records, write only safely missing content, reject conflicts and delete topology intent after verified transfer. Clean/destroy use confirmed - authority. Audit S3 mini-cluster persistence against the same contract. + authority. Replace S3 mini-cluster persistence against the same contract, + without a migration path for its unreleased mixed configuration. System initialization now confirms an existing replica's identity after a conflict or lost response, preserves groups for retry after peer failures, and requires every peer endpoint and remote-wiring request to succeed before @@ -198,16 +200,14 @@ requirement. core ulimit example bounds each dump. Focused retention and monitor suites pass. The complete container release, image and E2E gate passes, including startup, crash and hang recovery, persisted-volume restart, exhausted restart - budget and monitor death. Rust fmt and clippy pass. The host's Apport pipe - still prevents file-based end-to-end acceptance. -- [ ] **Manual release and optional symbols**: the `tools/` release script now - has a read-only dry run, consistent version updates, tag and GitHub Release - creation, and workflow dispatch. Optional `--symbols` extracts debug symbols - from the same staged ELF files as the image and uploads the named archive; - the default release skips that large asset. Local symbol identity checks pass. - A real release and source-line lookup remain to verify. Files: - `tools/release.py`, `container/single-node-container/{build.sh,collect-libs.sh}`, - `.github/workflows/release-container.yml`. + budget and monitor death. Rust fmt and clippy pass. This host's Apport pipe + can export a packaged program's real dump, but a CROWDB KV child abort left + no report because Apport cannot resolve its container-only executable path. + The monitor recovered the child. A symbols-enabled image and archive for the + same revision passed hashes, debuglink CRCs and `.debug_line` checks; the + symbolizer resolved a debugger-generated CROWDB monitor core to + `container/crowdb-monitor/src/main.rs:55`. File-based collection and + retention of a real CROWDB crash remain for a disposable file-collector host. ## Documentation and completion - [ ] **Bare-metal documentation**: publish verified KV, chunk and access From 65799bcb6d138cded13d5bc087a5a4afbd174d15 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 12:48:35 +0800 Subject: [PATCH 51/74] Document crash debugging and host core setup --- container/single-node-container/README.md | 8 +- doc/backlog/R188-console-group0-authority.md | 33 ++-- doc/dev/crash_debugging.md | 168 +++++++++++++++++++ doc/doc_index.md | 3 +- doc/working/plan-console-authority.md | 17 +- 5 files changed, 200 insertions(+), 29 deletions(-) create mode 100644 doc/dev/crash_debugging.md diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index 38663caf..57dfe7bf 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -50,6 +50,10 @@ remains available and the workflow reports a warning. ## Crash collection boundary +For host configuration, restoring its collector, and GDB commands for both +container and bare-metal cores, see the +[crash debugging guide](../../doc/dev/crash_debugging.md). + The image does not configure the host's Linux core collector. Inspect `/proc/sys/kernel/core_pattern` on the Docker host before expecting a dump in the mounted data volume. A leading `|` sends a crash to a host-side collector; @@ -108,8 +112,8 @@ directory, verifies source revision, version and SHA-256 hashes, then runs `gdb` without printing frame arguments. Use the crashed child binary instead of `crowdb-monitor` for a child core. The temporary binaries are removed after the stack is shown; the core stays at the path supplied by the operator. -The exact-build source-line check on a supported file-based collector remains -tracked by R188. +The operator will validate collection and source-line output when a real +crash is available. No host collector change is required by the image build. Collector behavior follows the [Linux core pattern documentation](https://docs.kernel.org/admin-guide/sysctl/kernel.html), [systemd-coredump manual](https://www.freedesktop.org/software/systemd/man/250/systemd-coredump.socket.html), diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md index 9a01f305..e6cc33e4 100644 --- a/doc/backlog/R188-console-group0-authority.md +++ b/doc/backlog/R188-console-group0-authority.md @@ -94,8 +94,10 @@ not block R187 completion. shows a dry-run plan, updates versions, creates the tag and GitHub Release, then dispatches the existing verified DockerHub publication workflow. Its actual use is deferred to the operator's later release; no release execution - is required for this requirement. Host crash acceptance remains open; no - host configuration change is assumed. + is required for this requirement. Document host crash collection, GDB and + exact-build symbols in `doc/dev/crash_debugging.md`. The user will validate + a real core when a future crash occurs; this requirement does not change + the host collector or require a new crash test. ## Dependencies @@ -155,17 +157,11 @@ not block R187 completion. assert each layer has a verified setup and health check, the non-production boundary is explicit, and no link targets the removed combined guide. Invariant: deployment guidance follows its implementation. E2E test. -- Given a disposable container on a supported file-based core collector, when - a child or PID 1 crashes, assert the dump has private ownership, one-core - retention and size bounds, and resolves to source lines using exact-build symbols. - Assert ordinary logs disclose no dump contents or credentials and the - container does not change host-wide collector policy. Invariant: private, - bounded and reproducible crash diagnostics. E2E test. -- Given Apport, systemd-coredump or Docker Desktop collector policies, when - crash collection is attempted, assert the documented host export workflow - locates the dump or explicitly reports unsupported collection, without - claiming an absent data-volume core. Invariant: truthful collector boundary. - Integration test. +- Crash debugging documentation is complete when `doc/dev/crash_debugging.md` + explains host collector selection and rollback, private core location and + limits, exact-image symbols and GDB, and direct GDB use for unstripped + bare-metal binaries. The user will verify a real core during a future + incident; no crash test or host configuration change is required now. Required gates: @@ -181,10 +177,7 @@ Required gates: - This host routes `core_pattern` to Apport. A disposable container KV child aborted and the monitor recovered it, but Apport did not create a CROWDB report: its log says `/opt/crowdb/bin/crowdb-kv-server` does not exist on the - host. A packaged host program did produce an exportable `CoreDump`, proving - the extraction procedure without proving CROWDB collection. Exact-image - hashes, GNU debuglinks and source-line symbolization passed with a - debugger-generated CROWDB monitor core. Private one-core retention passed - local tests. File-based collection, real crash-core symbolization and - retention still need a disposable host with a relative `core_pattern`; no - change to this host's collector is assumed. + host. Exact-build symbolization passed with a debugger-generated monitor + core. The user accepted the crash debugging guide as completion and will + validate file collection and source lines when a future real crash occurs. + The host collector was not changed. diff --git a/doc/dev/crash_debugging.md b/doc/dev/crash_debugging.md new file mode 100644 index 00000000..68027c94 --- /dev/null +++ b/doc/dev/crash_debugging.md @@ -0,0 +1,168 @@ + + + +# Debugging a CROWDB Crash + +Use the core from the crashed process and the exact executable build that +produced it. Core files can contain credentials and user data. Keep them in a +private directory, do not attach them to ordinary logs or issues, and remove +them when the investigation is complete. + +## 1. Find the collector + +On the machine running Docker or a bare-metal service, inspect: + +```sh +cat /proc/sys/kernel/core_pattern +cat /proc/sys/fs/suid_dumpable +``` + +- A relative name such as `core.%e.%p.%t` writes in the crashing process's + working directory. The name must start with `core` for the single-node + container's retention rule to recognize it. +- A leading `|` sends the dump to a host-side program. Ubuntu Apport usually + writes a report under `/var/crash`; `apport-unpack REPORT.crash OUTPUT_DIR` + extracts its `CoreDump`. `systemd-coredump` uses `coredumpctl list` and + `coredumpctl --output=FILE dump`. A collector may reject a container crash. + If there is no report, there is no core to extract. +- An absolute file pattern uses the crashing process's mount namespace and + root. Check the actual destination and access policy on that host. + +The container's data volume does not override a host collector. During the +2026-09-29 Ubuntu development-host check, Apport received the crash but could +not resolve `/opt/crowdb/bin/crowdb-kv-server` in the host filesystem. +The image contains the executable; the failed lookup happens in Apport. No +CROWDB core was saved by that test. + +## 2. Configure a file-based collector on a dedicated development host + +This is a **host-wide change**, not a Docker setting. It replaces Apport's +automatic crash reports for all processes on that host. Other programs with +a nonzero core limit may write private core files in their own working +directories. Do not apply it to a shared or production host without an +operator decision. Run these commands on the Docker host, never inside the +container. Record the original `core_pattern` and Apport service state first. + +On an Ubuntu host where `apport.service` owns `core_pattern`, a persistent +file-based setup is: + +```sh +cat /proc/sys/kernel/core_pattern +systemctl is-enabled apport.service +sudo systemctl disable --now apport.service +printf 'kernel.core_pattern=core.%%e.%%p.%%t\nfs.suid_dumpable=0\n' | + sudo tee /etc/sysctl.d/99-crowdb-core.conf +sudo sysctl -w 'kernel.core_pattern=core.%e.%p.%t' +sudo sysctl -w fs.suid_dumpable=0 +cat /proc/sys/kernel/core_pattern +``` + +`apport.service` sets the pipe pattern when it starts, so a sysctl file alone +does not keep the file pattern after a restart. Verify `core_pattern` again +after reboot. `fs.suid_dumpable=0` permits an ordinary process to write a +relative core; executables with file capabilities may still be excluded. +The container currently gives file capabilities to `crowdb-iceberg` and +`crowdb-access-server` for low ports, so do not assume those two will produce +cores under this setting. Verify the particular crashed service. + +For a one-time investigation, stop Apport and apply the two `sysctl -w` +commands without creating the sysctl file or disabling the service. Restart +Apport after the investigation to restore its collector. For the persistent +setup above, restore the prior Ubuntu behavior with: + +```sh +sudo unlink /etc/sysctl.d/99-crowdb-core.conf +sudo systemctl enable --now apport.service +cat /proc/sys/kernel/core_pattern +``` + +If this host used another collector originally, restore its recorded service +state and exact original pattern instead of starting Apport. + +## 3. Bound and locate the core + +For the single-node container, add a nonzero per-process bound when running +Docker: + +```sh +--ulimit core=1073741824:1073741824 +``` + +The monitor and managed children run from the private mounted directory +`/opt/crowdb/data/crash` (mode `0700`). With a relative `core.*` pattern, a +dump lands there. The monitor keeps the newest regular `core` file after +startup or child recovery. Inspect the directory with `docker exec`; export +the selected core to a private host directory with `docker cp` for GDB. A +1 GiB limit can truncate a larger dump. A piped collector ignores this +`RLIMIT_CORE` bound and uses its own policy. + +For a bare-metal process, set a writable private working directory and a +nonzero core limit in the launcher. A shell launch can use +`ulimit -c 1048576` (KiB); a systemd unit can use `WorkingDirectory=` and +`LimitCORE=1G`. Verify the actual process limit in `/proc/PID/limits`. +There is no container monitor retention rule for bare-metal cores. + +## 4. Open a container core with exact-build symbols + +Use the image that ran the crashed process and its matching optional symbol +archive. From the CROWDB checkout: + +```sh +pixi run -- python tools/symbolize-container-core.py \ + --image 'docker.io/crowdb/crowdb-iceberg:' \ + --symbols '/private/crowdb-symbols--git--linux-amd64.tar.zst' \ + --binary crowdb-kv-server \ + --core /private/core.crowdb-kv-ser.PID.TIME +``` + +The tool checks the image revision, version and binary SHA-256 hashes against +the archive, places the `.debug` files beside the matching stripped ELF files, +and starts GDB with the image's shared libraries. GNU debuglinks let GDB load +the separate symbols. Replace `--binary` with the actual crashed CROWDB +binary. If the archive was not released, build the exact Git tag with +`CROWDB_PACKAGE_SYMBOLS=1 pixi run build-single-node-container` and package +its local `target/container-symbols` directory for the helper: + +```sh +pixi run -- bash -c 'tar -C target/container-symbols -cf - . | zstd -q -o /private/crowdb-symbols-local.tar.zst' +``` + +Use that archive with the image produced by the same build. A different +commit or build is not a safe substitute. + +For an interactive session, keep private copies of the same image's `bin/` +and `lib/`, put each matching `.debug` file beside its stripped binary or +library, and run: + +```sh +pixi run -- gdb -q /private/runtime/bin/crowdb-kv-server /private/core +(gdb) set solib-search-path /private/runtime/lib +(gdb) sharedlibrary +(gdb) set print frame-arguments none +(gdb) thread apply all bt +``` + +Do not use a newly built binary against an older core just because its version +string is unchanged. + +## 5. Open a bare-metal core + +Bare-metal deployment keeps the binary's debug information. Use the exact +binary and shared libraries that were running when the core was made; no +separate container symbol archive is needed: + +```sh +pixi run -- gdb -q /path/to/exact/crowdb-kv-server /private/core +(gdb) set solib-search-path /path/to/exact/lib +(gdb) sharedlibrary +(gdb) set print frame-arguments none +(gdb) thread apply all bt +``` + +If the executable or a shared library has been replaced since the crash, +recover the original build before trusting the stack. A core from one build +must not be interpreted with symbols from another. + +References: [Linux core dump rules](https://man7.org/linux/man-pages/man5/core.5.html), +[kernel `core_pattern`](https://docs.kernel.org/admin-guide/sysctl/kernel.html), +and [Ubuntu Apport](https://documentation.ubuntu.com/project/contributors/debugging/apport/). diff --git a/doc/doc_index.md b/doc/doc_index.md index 8c54e702..f57984c7 100644 --- a/doc/doc_index.md +++ b/doc/doc_index.md @@ -47,9 +47,10 @@ Temporary plans live under `doc/working/`; flow analyses live under | Doc | When to read | | ---------------------------- | ---------------------------------------------------------------------- | +| `doc/dev/crash_debugging.md` | Core collection, host setup, GDB, and exact-build symbols. | | `doc/dev/env_setup.md` | Benchmark commands, sentinels, prerequisites, and perf-counter setup. | | `doc/dev/hyper_fork.md` | Hyper fork branches, submodule, build, sync, validation, and recovery. | -| `tools/README.md` | Tool directories, Pixi task entry points, CI checks, and suite timing. | +| `tools/README.md` | Tool directories, Pixi task entry points, CI checks, and suite timing. | ## Project Files (repo root) diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index d9cc3026..f17c75c8 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -90,7 +90,7 @@ configuration, documentation and crash-diagnostics tasks below are pending. Three-service lifecycle regressions and the complete CLI suite, fmt and clippy pass. Logs: `/tmp/crowdb-chunk-launch-*.log`. Removal of legacy startup/restore paths remains coupled to bootstrap cutover below. -- [ ] **Remove mixed persistence**: remove the unreleased `ConsoleConfig` +- [~] **Remove mixed persistence**: remove the unreleased `ConsoleConfig` parser/writer, inline SSH secrets, topology restoration and fixtures after the launch lifecycle and replay-safe bootstrap paths are wired. Preserve bootstrap intent independently until verified cutover. Update CLI commands, @@ -176,7 +176,7 @@ configuration, documentation and crash-diagnostics tasks below are pending. Transferred from R187 by user request. It does not block the single-node image requirement. -- [~] **Crash dump location and retention**: document and test how Linux +- [x] **Crash dump location and retention**: document how Linux host `core_pattern`, Docker's core ulimit, and the non-root container affect CROWDB child and PID 1 crashes. Cover a plain relative core-file pattern, Ubuntu Apport, systemd-coredump, and Docker Desktop's Linux VM. Choose a @@ -185,8 +185,8 @@ requirement. and provide explicit setup guidance instead of claiming the volume contains a core. Use a private data-volume directory and retain the newest `core` file after child recovery and monitor restart. Require Docker's core ulimit - for a per-dump size bound. Verify one disposable child crash end to end, retention/cleanup, - secret exposure, and symbolization against the exact binary build. Do not + for a per-dump size bound. Document how to inspect a real child crash, + retention, secret exposure, and exact-build symbolization. Do not change the host-wide `core_pattern` from inside the container. Files: `container/single-node-container/{Dockerfile,entrypoint.sh,tests/**}`, `container/crowdb-monitor/src/**`, @@ -206,8 +206,13 @@ requirement. The monitor recovered the child. A symbols-enabled image and archive for the same revision passed hashes, debuglink CRCs and `.debug_line` checks; the symbolizer resolved a debugger-generated CROWDB monitor core to - `container/crowdb-monitor/src/main.rs:55`. File-based collection and - retention of a real CROWDB crash remain for a disposable file-collector host. + `container/crowdb-monitor/src/main.rs:55`. The developer guide at + `doc/dev/crash_debugging.md` now covers host configuration and rollback, + private core handling, GDB with exact-image symbols, and bare-metal GDB with + unstripped binaries. The user accepted documentation as completion and + deferred live core verification until a future incident; no new crash test + or host configuration change is required in this task. + ## Documentation and completion - [ ] **Bare-metal documentation**: publish verified KV, chunk and access From 49307e8dc4ad181c1cf8317e407d2967d0677e76 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 13:24:30 +0800 Subject: [PATCH 52/74] Confirm Group 0 node removal --- .../src/commands/cluster/hardware.rs | 13 +- .../tests/launch_registry_cli_test.rs | 3 + app/crowdb-web/src/lib.rs | 4 + app/crowdb-web/src/managed_hardware.rs | 11 ++ .../tests/bare_metal_authority_test.rs | 31 +++++ doc/working/plan-console-authority.md | 6 +- lib/crowdb-console-shared/src/ops/hardware.rs | 9 ++ .../src/ops/hardware/authority.rs | 128 ++++++++++++++++++ .../tests/ops_hardware_authority_test.rs | 15 ++ 9 files changed, 215 insertions(+), 5 deletions(-) diff --git a/app/crowdb-cli/src/commands/cluster/hardware.rs b/app/crowdb-cli/src/commands/cluster/hardware.rs index a82a139d..01162a28 100644 --- a/app/crowdb-cli/src/commands/cluster/hardware.rs +++ b/app/crowdb-cli/src/commands/cluster/hardware.rs @@ -236,10 +236,17 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { Ok(c) => c, Err(c) => return c, }; - match crowdb_console_shared::ops::hardware::remove_node(&ctx, node_id).await { + let result = if cli.registry.is_some() { + crowdb_console_shared::ops::hardware::remove_node_from_group0(&ctx, node_id).await + } else { + crowdb_console_shared::ops::hardware::remove_node(&ctx, node_id).await + }; + match result { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; + if cli.registry.is_none() { + if let Err(c) = commit_config(cli, &ctx) { + return c; + } } println!("removed node {id}"); ExitCode::SUCCESS diff --git a/app/crowdb-cli/tests/launch_registry_cli_test.rs b/app/crowdb-cli/tests/launch_registry_cli_test.rs index 96167f4b..a48e362f 100644 --- a/app/crowdb-cli/tests/launch_registry_cli_test.rs +++ b/app/crowdb-cli/tests/launch_registry_cli_test.rs @@ -216,6 +216,9 @@ async fn registry_hardware_uses_group_zero_across_cli_invocations() { ); run_command(&path, g0.mgmt_port, &["cluster", "rack", "remove", "--id", "3"]); assert!(!run_command(&path, g0.mgmt_port, &["cluster", "rack", "list"]).contains("empty")); + run_command(&path, g0.mgmt_port, &["cluster", "node", "remove", "--id", "2"]); + assert!(!run_command(&path, g0.mgmt_port, &["cluster", "node", "list"]).contains("10.0.0.2")); + run_command(&path, g0.mgmt_port, &["cluster", "rack", "remove", "--id", "2"]); assert!(!dir.path().join("invalid-legacy.toml").exists()); } diff --git a/app/crowdb-web/src/lib.rs b/app/crowdb-web/src/lib.rs index fd1d5e2d..5ee0816d 100644 --- a/app/crowdb-web/src/lib.rs +++ b/app/crowdb-web/src/lib.rs @@ -94,6 +94,10 @@ pub fn router(state: AppState) -> axum::Router { "/api/nodes", get(managed_hardware::list_nodes) .merge(post(managed_hardware::add_node).route_layer(authorization.clone())), + ) + .route( + "/api/nodes/:id", + delete(managed_hardware::remove_node).route_layer(authorization.clone()), ); managed .merge(hardware) diff --git a/app/crowdb-web/src/managed_hardware.rs b/app/crowdb-web/src/managed_hardware.rs index eae11761..daed7381 100644 --- a/app/crowdb-web/src/managed_hardware.rs +++ b/app/crowdb-web/src/managed_hardware.rs @@ -87,3 +87,14 @@ pub(crate) async fn add_node( .map_err(api_error)?; Ok((StatusCode::CREATED, Json(node))) } + +pub(crate) async fn remove_node( + State(state): State, + Path(id): Path, +) -> Result { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::remove_node_from_group0(&ctx, id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} diff --git a/app/crowdb-web/tests/bare_metal_authority_test.rs b/app/crowdb-web/tests/bare_metal_authority_test.rs index cab0ddfe..66e8cdd0 100644 --- a/app/crowdb-web/tests/bare_metal_authority_test.rs +++ b/app/crowdb-web/tests/bare_metal_authority_test.rs @@ -216,6 +216,37 @@ async fn bare_metal_hardware_routes_share_confirmed_group_zero_state() { .0, StatusCode::CONFLICT ); + verify_hardware_deletion(&first, &second).await; +} + +async fn verify_hardware_deletion(first: &axum::Router, second: &axum::Router) { + assert_eq!( + hardware_request(second, axum::http::Method::DELETE, "/api/nodes/9", None, false) + .await + .0, + StatusCode::UNAUTHORIZED + ); + assert_eq!( + hardware_request(second, axum::http::Method::DELETE, "/api/nodes/9", None, true) + .await + .0, + StatusCode::NO_CONTENT + ); + let (_, nodes) = hardware_request( + first, + axum::http::Method::GET, + "/api/nodes?rack_id=8", + None, + false, + ) + .await; + assert!(nodes.as_array().unwrap().is_empty()); + assert_eq!( + hardware_request(first, axum::http::Method::DELETE, "/api/racks/8", None, true) + .await + .0, + StatusCode::NO_CONTENT + ); } async fn unavailable(app: &axum::Router) { diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index f17c75c8..b50b80cb 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -124,8 +124,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. rack; shared and real CLI regressions cover child conflict and absence. Bare-metal Web now exposes authenticated Group 0 rack creation/removal and node creation, with public confirmed rack/node reads. Two Web instances - observe the same records; inline SSH material is rejected. Node deletion, - disks and remaining legacy routes still need conversion. + observe the same records; inline SSH material is rejected. Registry CLI and + bare-metal Web now remove only unused nodes through one conditional Group 0 + rack-membership/node deletion; occupied nodes and unauthenticated Web writes + fail. Disk groups, disks and remaining legacy routes still need conversion. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. diff --git a/lib/crowdb-console-shared/src/ops/hardware.rs b/lib/crowdb-console-shared/src/ops/hardware.rs index e48639c3..84d74ff0 100644 --- a/lib/crowdb-console-shared/src/ops/hardware.rs +++ b/lib/crowdb-console-shared/src/ops/hardware.rs @@ -119,6 +119,15 @@ pub async fn remove_rack_from_group0(ctx: &OpContext, rack_id: u64) -> Result<() authority::remove_empty_rack(ctx, rack_id).await } +/// Remove an unused node and its rack membership in one confirmed Group 0 write. +/// +/// # Errors +/// Rejects a node with disk groups or KV replicas, a missing node, or an +/// uncertain authority result. +pub async fn remove_node_from_group0(ctx: &OpContext, node_id: u64) -> Result<()> { + authority::remove_empty_node(ctx, node_id).await +} + // ── rack ──────────────────────────────────────────────────────── /// Add a rack to the local config and group-0 sysdata. diff --git a/lib/crowdb-console-shared/src/ops/hardware/authority.rs b/lib/crowdb-console-shared/src/ops/hardware/authority.rs index d25b3c40..6cedce19 100644 --- a/lib/crowdb-console-shared/src/ops/hardware/authority.rs +++ b/lib/crowdb-console-shared/src/ops/hardware/authority.rs @@ -218,3 +218,131 @@ pub(super) async fn remove_empty_rack(ctx: &OpContext, rack_id: u64) -> Result<( } Err(KvError::CasBusy.into()) } + +/// Remove a node only after confirmed authority shows no children or KV replicas. +pub(super) async fn remove_empty_node(ctx: &OpContext, node_id: u64) -> Result<()> { + ready(ctx).await?; + for attempt in 0..10u64 { + let (rack_id, node) = ctx + .sysmd() + .list_nodes() + .await? + .into_iter() + .find_map(|(rack_id, id, value)| (id == node_id).then_some((rack_id, value))) + .ok_or_else(|| Error::NotFound { + kind: "node".into(), + id: node_id.to_string(), + })?; + if node_is_used(ctx, rack_id, node_id, &node).await? { + return Err(Error::Conflict { + kind: "node with children".into(), + id: node_id.to_string(), + }); + } + let rack_path = RackKey { rack_id }.to_path(); + let node_path = NodeKey { rack_id, node_id }.to_path(); + let (mut rack, revision) = match ctx + .kv() + .get(0, 0, rack_path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, revision } => ( + serde_json::from_slice::(&value) + .map_err(|error| Error::Config(error.to_string()))?, + revision, + ), + GetOutcome::NotFound => { + return Err(Error::Conflict { + kind: "node without rack".into(), + id: node_id.to_string(), + }) + } + }; + if !rack.node_ids.contains(&node_id) { + return Err(Error::Conflict { + kind: "node missing from rack membership".into(), + id: node_id.to_string(), + }); + } + rack.node_ids.retain(|id| *id != node_id); + let rack_bytes = serde_json::to_vec(&rack).map_err(|error| Error::Config(error.to_string()))?; + let ops = [ + BatchOp::Put { + key: rack_path.as_bytes().to_vec().into(), + value: rack_bytes.into(), + }, + BatchOp::Delete { + key: node_path.as_bytes().to_vec().into(), + }, + ]; + match ctx + .kv() + .batch_write_cas(0, 0, &ops, rack_path.as_bytes(), revision) + .await + { + Ok(_) => return Ok(()), + Err(KvError::CasFailed { .. } | KvError::CasBusy) => { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1) * 5)).await; + } + Err(KvError::OutcomeUnknown) => { + if node_removal_confirmed(ctx, &rack_path, &node_path, node_id).await? { + return Ok(()); + } + return Err(KvError::OutcomeUnknown.into()); + } + Err(error) => return Err(error.into()), + } + } + Err(KvError::CasBusy.into()) +} + +async fn node_is_used(ctx: &OpContext, rack_id: u64, node_id: u64, node: &NodeValue) -> Result { + if !node.disk_group_ids.is_empty() + || !ctx + .sysmd() + .list_disk_groups_on_node(rack_id, node_id) + .await? + .is_empty() + || ctx + .sysmd() + .list_all_replicas() + .await? + .iter() + .any(|replica| replica.node_id == node_id) + { + return Ok(true); + } + Ok(ctx + .sysmd() + .read_all_kv_server_instances() + .await? + .into_iter() + .any(|(_, instance)| { + instance + .extra + .and_then(|extra| extra.kv_server) + .and_then(|server| server.node_id) + == Some(node_id) + })) +} + +async fn node_removal_confirmed( + ctx: &OpContext, + rack_path: &str, + node_path: &str, + node_id: u64, +) -> Result { + let rack = ctx + .kv() + .get(0, 0, rack_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + let node = ctx + .kv() + .get(0, 0, node_path.as_bytes(), ReadMode::Linearizable, None) + .await?; + let (GetOutcome::Found { value, .. }, GetOutcome::NotFound) = (rack, node) else { + return Ok(false); + }; + let rack: RackValue = serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?; + Ok(!rack.node_ids.contains(&node_id)) +} diff --git a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs index e1508d7a..fa763535 100644 --- a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs +++ b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs @@ -104,6 +104,21 @@ async fn separate_consoles_confirm_matching_hardware_and_reject_conflicts() { first.sysmd().get_rack(2).await.unwrap().unwrap().node_ids, vec![2] ); + + let occupied = hardware::remove_node_from_group0(&second, 1).await.unwrap_err(); + assert!(matches!(occupied, Error::Conflict { .. }), "{occupied:?}"); + hardware::remove_node_from_group0(&second, 2).await.unwrap(); + assert!(first.sysmd().get_node(2, 2).await.unwrap().is_none()); + assert!(first + .sysmd() + .get_rack(2) + .await + .unwrap() + .unwrap() + .node_ids + .is_empty()); + hardware::remove_rack_from_group0(&first, 2).await.unwrap(); + assert!(second.sysmd().get_rack(2).await.unwrap().is_none()); } #[tokio::test] From e5fad5da646f2fb698fd29769523fccd8ef68683 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 13:41:32 +0800 Subject: [PATCH 53/74] Move S3 mini cluster to local launch state --- doc/working/plan-console-authority.md | 10 +- lib/crowdb-console-shared/src/ops/cluster.rs | 11 +- lib/crowdb-console-shared/src/ops/s3.rs | 141 ++++++++++++++---- .../src/ops/s3/local_state.rs | 123 +++++++++++++++ .../tests/s3_mini_cluster_test.rs | 36 ++++- 5 files changed, 280 insertions(+), 41 deletions(-) create mode 100644 lib/crowdb-console-shared/src/ops/s3/local_state.rs diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index b50b80cb..4316e051 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -96,6 +96,11 @@ configuration, documentation and crash-diagnostics tasks below are pending. bootstrap intent independently until verified cutover. Update CLI commands, Web persistence and S3 mini-cluster callers together; no compatibility reader or migration path because the old configuration was never released. + S3 mini-clusters now persist versioned local process/seed state rather than + `console.toml`; restored KV launch nodes are ephemeral process inputs and the + bundled Web uses `WebProcessConfig`. The full persistent S3 stop/restart and + range-read E2E passes. Remaining CLI/Web legacy config paths and the shared + parser/writer still need removal. - [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through shared Group 0 hardware operations; preserve conflicts and uncertain writes without local-first commits. Docker keeps its hardware restrictions. Hardware @@ -163,7 +168,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. retry copy beside the launch registry, runs the same confirmation path and deletes that copy after success. It does not write the mixed console file. The legacy CLI/Web path still writes that file; S3 mini-cluster still needs - cutover before it can be removed. + bootstrap-interruption replay before the old format can be removed. Its + completed cluster now restarts from launch-only local state and Group 0 + seeds; no local topology is loaded after publication. A full persistent S3 + stop/restart and range-read E2E passes. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with diff --git a/lib/crowdb-console-shared/src/ops/cluster.rs b/lib/crowdb-console-shared/src/ops/cluster.rs index 6ad6fbfb..b00e11b2 100644 --- a/lib/crowdb-console-shared/src/ops/cluster.rs +++ b/lib/crowdb-console-shared/src/ops/cluster.rs @@ -874,7 +874,7 @@ pub async fn local_deploy_diskdb( workspace: &std::path::Path, cfg: &LocalDiskdbDeployConfig, ) -> Result { - let nodes = validate_diskdb_deploy(ctx, cfg)?; + let nodes = validate_diskdb_deploy(ctx, cfg).await?; ensure_diskdb_hardware(ctx, &nodes).await?; let (disk_group_count, disk_count) = provision_diskdb_topology(ctx, &nodes, cfg).await?; let ports = alloc_diskdb_ports(workspace, nodes.len())?; @@ -932,7 +932,7 @@ async fn ensure_diskdb_hardware(ctx: &OpContext, nodes: &[NodeEntry]) -> Result< Ok(()) } -fn validate_diskdb_deploy(ctx: &OpContext, cfg: &LocalDiskdbDeployConfig) -> Result> { +async fn validate_diskdb_deploy(ctx: &OpContext, cfg: &LocalDiskdbDeployConfig) -> Result> { if cfg.disk_groups_per_node == 0 || cfg.disks_per_group == 0 || cfg.data_groups.is_empty() { return Err(Error::Validation { field: "diskdb_topology".into(), @@ -947,12 +947,9 @@ fn validate_diskdb_deploy(ctx: &OpContext, cfg: &LocalDiskdbDeployConfig) -> Res message: "deploy the KV cluster before DiskDB".into(), }); } - let configured_groups = ctx.config().groups.clone(); + ctx.kv().refresh_topology().await?; for group_id in &cfg.data_groups { - if !configured_groups - .iter() - .any(|group| group.store_id == 0 && group.group_id == *group_id) - { + if ctx.sysmd().get_group(0, *group_id).await?.is_none() { return Err(Error::NotFound { kind: "kv_group".into(), id: format!("0:{group_id}"), diff --git a/lib/crowdb-console-shared/src/ops/s3.rs b/lib/crowdb-console-shared/src/ops/s3.rs index 3763d64b..8a38a1cd 100644 --- a/lib/crowdb-console-shared/src/ops/s3.rs +++ b/lib/crowdb-console-shared/src/ops/s3.rs @@ -12,13 +12,14 @@ use crowdb_protocol::port::namespace::{assign_process_ports, RuntimeNamespace}; use crowdb_protocol::ServicePort; use serde::Serialize; -use crate::config::{ConsoleConfig, LocalLaunchSpec, ServerEntry, ServiceType}; +use crate::config::{ConsoleConfig, LocalLaunchSpec, NodeEntry, RackEntry, ServerEntry, ServiceType}; use crate::error::{Error, Result}; use crate::lifecycle; use crate::ops::cluster::{self, KvDeployTunables, LocalChunkdbDeployConfig, LocalDiskdbDeployConfig}; use crate::ops::OpContext; -const CONFIG_FILE: &str = "console.toml"; +mod local_state; + const MARKER_FILE: &str = "s3-mini-cluster.json"; const INITIALIZING_FILE: &str = "s3-mini-cluster.initializing.json"; const MASTER_KEY: &str = "1111111111111111111111111111111111111111111111111111111111111111"; @@ -58,10 +59,10 @@ pub struct MiniClusterStatus { #[must_use] pub fn config_path(data_dir: &Path) -> PathBuf { - data_dir.join(CONFIG_FILE) + local_state::path(data_dir) } -/// Load a persisted mini-cluster record and console configuration. +/// Load a persisted mini-cluster record and local process state. /// /// # Errors /// Returns an error for an unrecognized directory or invalid persisted data. @@ -71,7 +72,8 @@ pub fn load(data_dir: &Path) -> Result<(ConsoleConfig, MiniClusterRecord)> { message: format!("{} is not a CROWDB S3 mini-cluster: {error}", data_dir.display()), })?; let record = serde_json::from_slice(&marker).map_err(|error| Error::Config(error.to_string()))?; - Ok((ConsoleConfig::load(&config_path(data_dir))?, record)) + let (config, _) = local_state::load(data_dir)?; + Ok((config, record)) } /// Create or restart a persistent local S3 mini-cluster. @@ -187,7 +189,7 @@ async fn start_with_profile( Ok(endpoints) => endpoints, Err(error) => { stop_config_processes(&mut ctx.config_mut()); - let _ = ctx.config().save(&config_path(data_dir)); + let _ = local_state::save(data_dir, &ctx.config()); return Err(error); } }; @@ -247,14 +249,14 @@ async fn initialize_new( } } let seeds = management_seeds(&ctx.config()); - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; let chunk_kv = spawn_chunk_kv(data_dir, &seeds).await?; add_service(ctx, chunk_kv)?; - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; let access = spawn_access(data_dir, &seeds).await?; let endpoint = access.entry.url.clone(); add_service(ctx, access)?; - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; let (web_endpoint, web_pid) = spawn_web(data_dir).await?; Ok(StartedEndpoints { s3: endpoint, @@ -264,7 +266,8 @@ async fn initialize_new( } async fn restart(data_dir: &Path) -> Result { - let (config, mut record) = load(data_dir)?; + let (mut config, mut record) = load(data_dir)?; + restore_launch_nodes(&mut config)?; if let Some(pid) = record.web_pid.take() { let _ = lifecycle::stop_pid_with_timeout(pid, Duration::from_secs(5)); save_record(&data_dir.join(MARKER_FILE), &record)?; @@ -288,7 +291,7 @@ async fn restart(data_dir: &Path) -> Result { crate::ops::kv_server::restart(&ctx, node_id, Some(&server_dir), None, &seeds).await?; } cluster::restart_storage_services(&ctx).await?; - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; for kind in [ServiceType::ChunkKv, ServiceType::AccessServer] { let server = ctx .config() @@ -306,7 +309,7 @@ async fn restart(data_dir: &Path) -> Result { record.endpoint.clone_from(&spawned.entry.url); } add_service(&ctx, spawned)?; - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; continue; }; let mut launch = ctx @@ -329,9 +332,9 @@ async fn restart(data_dir: &Path) -> Result { { entry.pid = Some(pid); } - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; } - ctx.config().save(&config_path(data_dir))?; + local_state::save(data_dir, &ctx.config())?; let (web_endpoint, web_pid) = spawn_web(data_dir).await?; record.web_endpoint = web_endpoint; record.web_pid = Some(web_pid); @@ -367,7 +370,7 @@ pub fn stop(data_dir: &Path) -> Result { let _ = lifecycle::stop_pid_with_timeout(pid, Duration::from_secs(5)); } stop_config_processes(&mut config); - config.save(&config_path(data_dir))?; + local_state::save(data_dir, &config)?; save_record(&data_dir.join(MARKER_FILE), &record)?; Ok(status_from(data_dir, false, &config, &record)) } @@ -421,6 +424,36 @@ fn management_seeds(config: &ConsoleConfig) -> Vec { .collect() } +fn restore_launch_nodes(config: &mut ConsoleConfig) -> Result<()> { + config.add_rack(RackEntry { + id: 1, + name: "local-launch".into(), + })?; + let node_ids: Vec<_> = config + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| { + server + .node_id + .ok_or_else(|| Error::Config("S3 KV launch has no node id".into())) + }) + .collect::>()?; + for id in node_ids { + config.add_node(NodeEntry { + id, + rack_id: 1, + host: "127.0.0.1".into(), + ssh_port: 22, + ssh_user: String::new(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: None, + })?; + } + Ok(()) +} + struct SpawnedService { entry: ServerEntry, launch: LocalLaunchSpec, @@ -508,33 +541,85 @@ async fn spawn_access(data_dir: &Path, seeds: &[String]) -> Result Result<(String, u32)> { + use crate::config::web::{WebMode, WebProcessConfig}; + let binary = find_binary("CROWDB_WEB_BIN", "crowdb-web")?; let port = assign_cluster_port(data_dir, ServicePort::Web, "web-1")?; - let workdir = data_dir.join("services/web-1"); + let root = std::fs::canonicalize(data_dir)?; + let workdir = root.join("services/web-1"); let log_dir = workdir.join("log"); std::fs::create_dir_all(&log_dir)?; + let ui_root = root.join("ui"); + std::fs::create_dir_all(&ui_root)?; let endpoint = format!("http://127.0.0.1:{port}"); + let (_, seeds) = local_state::load(data_dir)?; + let config = WebProcessConfig { + version: 1, + mode: WebMode::BareMetal, + bind: "127.0.0.1".into(), + port, + group0_management_seeds: seeds, + ui_root, + monitor_status: None, + log_dir, + log_max_file_mb: 30, + log_max_files: 5, + request_timeout_ms: Some(5_000), + }; + config.validate()?; + let config_path = root.join("s3-web.toml"); + std::fs::write( + &config_path, + toml::to_string_pretty(&config).map_err(|error| Error::Config(error.to_string()))?, + )?; + let mut env = BTreeMap::new(); + env.insert("CROWDB_ICEBERG_MANAGE_TOKEN".into(), web_token(&root)?); let launch = LocalLaunchSpec { program: binary.to_string_lossy().into_owned(), - args: vec![ - "--bind".into(), - "127.0.0.1".into(), - "--port".into(), - port.to_string(), - "--config".into(), - config_path(data_dir).to_string_lossy().into_owned(), - "--skip-startup-restore".into(), - "--log-dir".into(), - log_dir.to_string_lossy().into_owned(), - ], + args: vec!["--config".into(), config_path.to_string_lossy().into_owned()], workdir: workdir.to_string_lossy().into_owned(), - env: BTreeMap::new(), + env, readiness_url: Some(format!("{endpoint}/healthz")), }; let pid = spawn(&launch, "web-1").await?; Ok((endpoint, pid)) } +fn web_token(root: &Path) -> Result { + use std::io::{Read, Write}; + use std::os::unix::fs::{OpenOptionsExt, PermissionsExt}; + + let path = root.join("s3-web-manage.token"); + if let Ok(metadata) = std::fs::symlink_metadata(&path) { + if !metadata.file_type().is_file() || metadata.permissions().mode() & 0o077 != 0 { + return Err(Error::Config( + "S3 Web management token file is not private".into(), + )); + } + let token = std::fs::read_to_string(path)?; + if token.len() != 64 || !token.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(Error::Config("S3 Web management token is invalid".into())); + } + return Ok(token); + } + let mut bytes = [0u8; 32]; + std::fs::File::open("/dev/urandom")?.read_exact(&mut bytes)?; + let mut token = String::with_capacity(64); + for byte in bytes { + const HEX: &[u8; 16] = b"0123456789abcdef"; + token.push(char::from(HEX[usize::from(byte >> 4)])); + token.push(char::from(HEX[usize::from(byte & 0x0f)])); + } + let mut file = std::fs::OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(path)?; + file.write_all(token.as_bytes())?; + file.sync_all()?; + Ok(token) +} + fn assign_cluster_port(data_dir: &Path, service: ServicePort, identity: &str) -> Result { if data_dir.join("namespace.json").is_file() { return RuntimeNamespace::persistent(data_dir, NAMESPACE_ID) diff --git a/lib/crowdb-console-shared/src/ops/s3/local_state.rs b/lib/crowdb-console-shared/src/ops/s3/local_state.rs new file mode 100644 index 00000000..2bdc97c6 --- /dev/null +++ b/lib/crowdb-console-shared/src/ops/s3/local_state.rs @@ -0,0 +1,123 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Local S3 mini-cluster process inputs and runtime identity, without topology. + +use std::collections::{BTreeMap, HashSet}; +use std::fs::{self, OpenOptions}; +use std::io::Write; +use std::os::unix::fs::OpenOptionsExt; +use std::path::{Path, PathBuf}; +use std::sync::atomic::{AtomicU64, Ordering}; + +use serde::{Deserialize, Serialize}; + +use crate::config::{ConsoleConfig, LocalLaunchSpec, ServerEntry, ServiceType}; +use crate::error::{Error, Result}; + +const FILE: &str = "s3-local-state.toml"; +const VERSION: u32 = 1; +static NEXT_TEMP: AtomicU64 = AtomicU64::new(0); + +#[derive(Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct LocalState { + version: u32, + group0_seeds: Vec, + #[serde(rename = "service")] + services: Vec, + #[serde(default)] + local_launches: BTreeMap, +} + +pub(super) fn path(data_dir: &Path) -> PathBuf { + data_dir.join(FILE) +} + +pub(super) fn load(data_dir: &Path) -> Result<(ConsoleConfig, Vec)> { + let body = fs::read_to_string(path(data_dir))?; + let state: LocalState = toml::from_str(&body).map_err(|error| Error::Config(error.to_string()))?; + state.validate()?; + Ok(( + ConsoleConfig { + servers: state.services, + local_launches: state.local_launches, + ..ConsoleConfig::default() + }, + state.group0_seeds, + )) +} + +pub(super) fn save(data_dir: &Path, config: &ConsoleConfig) -> Result<()> { + let group0_seeds: Vec<_> = config + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| server.url.clone()) + .collect(); + let state = LocalState { + version: VERSION, + group0_seeds, + services: config.servers.clone(), + local_launches: config.local_launches.clone(), + }; + state.validate()?; + let body = toml::to_string_pretty(&state).map_err(|error| Error::Config(error.to_string()))?; + let destination = path(data_dir); + let temporary = destination.with_extension(format!( + "tmp.{}.{}", + std::process::id(), + NEXT_TEMP.fetch_add(1, Ordering::Relaxed) + )); + let result = (|| { + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&temporary)?; + file.write_all(body.as_bytes())?; + file.sync_all()?; + fs::rename(&temporary, &destination)?; + fs::File::open(data_dir)?.sync_all()?; + Ok(()) + })(); + if result.is_err() { + let _ = fs::remove_file(temporary); + } + result +} + +impl LocalState { + fn validate(&self) -> Result<()> { + let expected_seeds: Vec<_> = self + .services + .iter() + .filter(|service| service.service_type == ServiceType::Kv) + .map(|service| service.url.clone()) + .collect(); + if self.version != VERSION + || expected_seeds.is_empty() + || expected_seeds != self.group0_seeds + || self.group0_seeds.iter().any(String::is_empty) + { + return Err(Error::Config( + "S3 local state version or Group 0 seeds are invalid".into(), + )); + } + let mut identities = HashSet::new(); + if self + .services + .iter() + .any(|service| !identities.insert(&service.id)) + || self + .local_launches + .values() + .any(|launch| launch.env.contains_key("CROWDB_S3_MASTER_KEY")) + { + return Err(Error::Config( + "S3 local state has duplicate services or inline secrets".into(), + )); + } + Ok(()) + } +} diff --git a/lib/crowdb-console-shared/tests/s3_mini_cluster_test.rs b/lib/crowdb-console-shared/tests/s3_mini_cluster_test.rs index 9a79d512..a695d99e 100644 --- a/lib/crowdb-console-shared/tests/s3_mini_cluster_test.rs +++ b/lib/crowdb-console-shared/tests/s3_mini_cluster_test.rs @@ -26,6 +26,28 @@ fn incomplete_marker_fails_closed() { assert!(error.to_string().contains("config error")); } +#[test] +fn local_launch_state_rejects_topology_and_legacy_console_file() { + let dir = TestDir::new("s3-mini-local-only").expect("create test directory"); + std::fs::write( + dir.path().join("s3-mini-cluster.json"), + r#"{"version":1,"endpoint":"http://127.0.0.1:16000","tenant":"local"}"#, + ) + .unwrap(); + std::fs::write(dir.path().join("console.toml"), "[[rack]]\nid = 1\n").unwrap(); + assert!( + s3::status(dir.path()).is_err(), + "legacy topology must not be loaded" + ); + std::fs::write( + dir.path().join("s3-local-state.toml"), + "version = 1\ngroup0_seeds = ['http://127.0.0.1:10000']\n[[rack]]\nid = 1\n", + ) + .unwrap(); + let error = s3::status(dir.path()).expect_err("local state cannot contain topology"); + assert!(error.to_string().contains("unknown field"), "{error}"); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] #[ignore = "starts the complete local storage and S3 process stack"] async fn persistent_cluster_survives_stop_restart_and_range_read() { @@ -36,13 +58,13 @@ async fn persistent_cluster_survives_stop_restart_and_range_read() { .await .expect("web health request"); assert!(health.status().is_success()); - let servers = reqwest::get(format!("{}/api/servers", started.web_endpoint)) + let preview = reqwest::get(format!("{}/api/preview", started.web_endpoint)) .await - .expect("web server-list request") + .expect("web preview request") .text() .await - .expect("web server-list body"); - assert!(servers.contains("access-server-1")); + .expect("web preview body"); + assert!(preview.contains("group0")); let client = s3::S3HttpClient::from_data_dir(dir.path()).expect("S3 client"); client .request(Method::PUT, Some("durable-bucket"), None, &[], None, None) @@ -60,9 +82,13 @@ async fn persistent_cluster_survives_stop_restart_and_range_read() { .await .expect("put object"); let marker = std::fs::read_to_string(dir.path().join("s3-mini-cluster.json")).expect("marker"); - let config = std::fs::read_to_string(dir.path().join("console.toml")).expect("config"); + let config = std::fs::read_to_string(dir.path().join("s3-local-state.toml")).expect("local state"); assert!(!marker.contains("1111111111111111")); assert!(!config.contains("1111111111111111")); + assert!(!config.contains("[[rack]]")); + assert!(!config.contains("[[node]]")); + assert!(!config.contains("[[store]]")); + assert!(!dir.path().join("console.toml").exists()); let stopped = s3::stop(dir.path()).expect("stop cluster"); assert_eq!(stopped.running_services, 0); From 26b3be4b205fdb36623ff99100f3ec4d6ef0f1a0 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 13:55:31 +0800 Subject: [PATCH 54/74] Confirm Group 0 storage hardware mutations --- .../tests/common/iceberg_stack.rs | 1 + .../tests/s3_full_stack_test.rs | 1 + app/crowdb-chunkdb/tests/common/cluster.rs | 1 + app/crowdb-chunkdb/tests/e2e_test.rs | 1 + app/crowdb-chunkdb/tests/selector_test.rs | 1 + app/crowdb-chunkdb/tests/topology_test.rs | 1 + .../src/commands/cluster/hardware.rs | 195 +++++++- app/crowdb-diskdb/tests/diskdb_e2e_test.rs | 1 + app/crowdb-diskdb/tests/recovery_test.rs | 1 + .../tests/relocation_journal_test.rs | 1 + app/crowdb-diskdb/tests/scanner_test.rs | 1 + app/crowdb-web/src/lib.rs | 18 + app/crowdb-web/src/managed_hardware.rs | 77 +++- .../tests/bare_metal_authority_test.rs | 87 ++++ .../crowdb-monitor/src/bootstrap/hardware.rs | 1 + doc/working/plan-console-authority.md | 14 +- .../tests/production_restart_e2e.rs | 1 + lib/crowdb-console-shared/src/ops/hardware.rs | 9 +- .../src/ops/hardware/authority_storage.rs | 429 ++++++++++++++++++ .../tests/ops_hardware_authority_test.rs | 119 +++++ .../tests/disk_io_group0_sync_test.rs | 1 + lib/crowdb-protocol/src/types/diskdb.rs | 2 + lib/crowdb-test-harness/src/hardware.rs | 1 + 23 files changed, 954 insertions(+), 10 deletions(-) create mode 100644 lib/crowdb-console-shared/src/ops/hardware/authority_storage.rs diff --git a/app/crowdb-access-server/tests/common/iceberg_stack.rs b/app/crowdb-access-server/tests/common/iceberg_stack.rs index e589c041..96b98e29 100644 --- a/app/crowdb-access-server/tests/common/iceberg_stack.rs +++ b/app/crowdb-access-server/tests/common/iceberg_stack.rs @@ -195,6 +195,7 @@ async fn seed(cluster: &KvCluster) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: vec![disk], + name: String::new(), }, ) .await diff --git a/app/crowdb-access-server/tests/s3_full_stack_test.rs b/app/crowdb-access-server/tests/s3_full_stack_test.rs index f2e3900f..7c7f2176 100644 --- a/app/crowdb-access-server/tests/s3_full_stack_test.rs +++ b/app/crowdb-access-server/tests/s3_full_stack_test.rs @@ -804,6 +804,7 @@ async fn seed_compact_hardware(hardware: &HardwareClient) -> Vec TopologyCache { value: DiskGroupValue { status: HwStatus::Up as i32, disk_ids: vec![], + name: String::new(), }, }); } diff --git a/app/crowdb-chunkdb/tests/selector_test.rs b/app/crowdb-chunkdb/tests/selector_test.rs index 90b0bfd2..19c51709 100644 --- a/app/crowdb-chunkdb/tests/selector_test.rs +++ b/app/crowdb-chunkdb/tests/selector_test.rs @@ -22,6 +22,7 @@ fn make_dg_entry(dg_id: u64, rack: u64, node: u64, status: HwStatus) -> DiskGrou value: DiskGroupValue { status: status as i32, disk_ids: vec![], + name: String::new(), }, } } diff --git a/app/crowdb-chunkdb/tests/topology_test.rs b/app/crowdb-chunkdb/tests/topology_test.rs index 3c23aef1..b7fa5d9f 100644 --- a/app/crowdb-chunkdb/tests/topology_test.rs +++ b/app/crowdb-chunkdb/tests/topology_test.rs @@ -19,6 +19,7 @@ fn make_dg_entry(dg_id: u64, rack: u64, node: u64, status: HwStatus) -> DiskGrou value: DiskGroupValue { status: status as i32, disk_ids: vec![], + name: String::new(), }, } } diff --git a/app/crowdb-cli/src/commands/cluster/hardware.rs b/app/crowdb-cli/src/commands/cluster/hardware.rs index 01162a28..644f17ba 100644 --- a/app/crowdb-cli/src/commands/cluster/hardware.rs +++ b/app/crowdb-cli/src/commands/cluster/hardware.rs @@ -341,9 +341,83 @@ pub enum DiskGroupVerb { } pub async fn run_disk_group_verb(cli: &Cli, verb: DiskGroupVerb) -> ExitCode { - let _ = (cli, verb); - eprintln!("disk-group commands not yet wired to ops (Phase 3)"); - ExitCode::from(1) + use crowdb_console_shared::ops::hardware; + let ctx = match op_context(cli) { + Ok(ctx) => ctx, + Err(code) => return code, + }; + let result = match verb { + DiskGroupVerb::Add { id, rack, node, name } => { + let (Ok(id), Ok(rack), Ok(node)) = (id.parse::(), rack.parse::(), node.parse::()) + else { + eprintln!("error: disk-group, rack and node IDs must be integers"); + return ExitCode::from(1); + }; + match hardware::list_nodes_from_group0(&ctx, Some(rack)).await { + Ok(nodes) if nodes.iter().any(|entry| entry.id == node) => {} + Ok(_) => { + eprintln!("error: node {node} is not in rack {rack}"); + return ExitCode::from(2); + } + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(2); + } + } + hardware::add_disk_group_to_group0(&ctx, node, id, &name) + .await + .map(|entry| { + println!("added disk group {} on node {}", entry.id, node); + }) + } + DiskGroupVerb::Remove { id } => { + let id = match id.parse::() { + Ok(id) => id, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(1); + } + }; + let groups = match ctx.sysmd().list_disk_groups().await { + Ok(groups) => groups, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(2); + } + }; + let matches: Vec<_> = groups.into_iter().filter(|group| group.dg_id == id).collect(); + if matches.len() != 1 { + eprintln!( + "error: disk group {id} has {} matches; specify a unique ID", + matches.len() + ); + return ExitCode::from(2); + } + hardware::remove_disk_group_from_group0(&ctx, matches[0].node_id, id) + .await + .map(|()| println!("removed disk group {id}")) + } + DiskGroupVerb::List => match ctx.sysmd().list_disk_groups().await { + Ok(mut groups) => { + groups.sort_unstable_by_key(|group| (group.rack_id, group.node_id, group.dg_id)); + for group in groups { + println!( + "{}\t{}\t{}\t{}", + group.rack_id, group.node_id, group.dg_id, group.value.name + ); + } + Ok(()) + } + Err(error) => Err(error.into()), + }, + }; + match result { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("error: {error}"); + ExitCode::from(2) + } + } } // ── disk ───────────────────────────────────────────────────────── @@ -377,8 +451,117 @@ pub enum DiskVerb { List, } +#[allow(clippy::too_many_lines)] pub async fn run_disk_verb(cli: &Cli, verb: DiskVerb) -> ExitCode { - let _ = (cli, verb); - eprintln!("disk commands not yet wired to ops (Phase 3)"); - ExitCode::from(1) + use crowdb_console_shared::ops::hardware::{self, AddDiskInput}; + use crowdb_protocol::DiskIdExt; + let ctx = match op_context(cli) { + Ok(ctx) => ctx, + Err(code) => return code, + }; + let result = match verb { + DiskVerb::Add { + id, + rack, + node, + group, + disk_type, + capacity, + zone_size, + unit_size, + device_path, + } => { + let (Ok(rack), Ok(node), Ok(group), Ok(capacity_bytes), Ok(zone_size_bytes), Ok(unit_size_bytes)) = ( + rack.parse::(), + node.parse::(), + group.parse::(), + capacity.parse::(), + zone_size.parse::(), + unit_size.parse::(), + ) else { + eprintln!("error: rack, node, group and size arguments must be integers"); + return ExitCode::from(1); + }; + let nodes = match hardware::list_nodes_from_group0(&ctx, Some(rack)).await { + Ok(nodes) => nodes, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(2); + } + }; + if !nodes.iter().any(|entry| entry.id == node) { + eprintln!("error: node {node} is not in rack {rack}"); + return ExitCode::from(2); + } + let input = AddDiskInput { + disk_id: id, + disk_type, + capacity_bytes, + zone_size_bytes, + unit_size_bytes, + device_path, + }; + hardware::add_disk_to_group0(&ctx, node, group, &input) + .await + .map(|entry| println!("added disk {}", entry.disk_id)) + } + DiskVerb::Remove { id } => { + let disk_id = match crowdb_protocol::common::DiskId::from_display_string(&id) { + Ok(id) => id, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(1); + } + }; + let disks = match ctx.sysmd().list_all_disks().await { + Ok(disks) => disks, + Err(error) => { + eprintln!("error: {error}"); + return ExitCode::from(2); + } + }; + let matches: Vec<_> = disks.into_iter().filter(|disk| disk.disk_id == disk_id).collect(); + if matches.len() != 1 { + eprintln!( + "error: disk {id} has {} matches; specify a unique ID", + matches.len() + ); + return ExitCode::from(2); + } + hardware::remove_disk_from_group0(&ctx, matches[0].node_id, matches[0].disk_group_id, &id) + .await + .map(|_| println!("removed disk {id}")) + } + DiskVerb::List => match ctx.sysmd().list_all_disks().await { + Ok(mut disks) => { + disks.sort_unstable_by_key(|disk| { + ( + disk.rack_id, + disk.node_id, + disk.disk_group_id, + disk.disk_id.high, + disk.disk_id.low, + ) + }); + for disk in disks { + println!( + "{}\t{}\t{}\t{}", + disk.rack_id, + disk.node_id, + disk.disk_group_id, + disk.disk_id.to_display_string() + ); + } + Ok(()) + } + Err(error) => Err(error.into()), + }, + }; + match result { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("error: {error}"); + ExitCode::from(2) + } + } } diff --git a/app/crowdb-diskdb/tests/diskdb_e2e_test.rs b/app/crowdb-diskdb/tests/diskdb_e2e_test.rs index 6dea06f5..bd0761a4 100644 --- a/app/crowdb-diskdb/tests/diskdb_e2e_test.rs +++ b/app/crowdb-diskdb/tests/diskdb_e2e_test.rs @@ -107,6 +107,7 @@ async fn seed_hardware(hw: &HardwareClient) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.clone(), + name: String::new(), }, ) .await diff --git a/app/crowdb-diskdb/tests/recovery_test.rs b/app/crowdb-diskdb/tests/recovery_test.rs index b1298752..934a2679 100644 --- a/app/crowdb-diskdb/tests/recovery_test.rs +++ b/app/crowdb-diskdb/tests/recovery_test.rs @@ -84,6 +84,7 @@ async fn seed_hardware(hw: &HardwareClient) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.clone(), + name: String::new(), }, ) .await diff --git a/app/crowdb-diskdb/tests/relocation_journal_test.rs b/app/crowdb-diskdb/tests/relocation_journal_test.rs index f414d9e3..f11ace9f 100644 --- a/app/crowdb-diskdb/tests/relocation_journal_test.rs +++ b/app/crowdb-diskdb/tests/relocation_journal_test.rs @@ -136,6 +136,7 @@ async fn seed_hardware(hw: &HardwareClient) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.clone(), + name: String::new(), }, ) .await diff --git a/app/crowdb-diskdb/tests/scanner_test.rs b/app/crowdb-diskdb/tests/scanner_test.rs index 07ac061e..ef96f859 100644 --- a/app/crowdb-diskdb/tests/scanner_test.rs +++ b/app/crowdb-diskdb/tests/scanner_test.rs @@ -94,6 +94,7 @@ async fn seed_hardware(hw: &HardwareClient) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.clone(), + name: String::new(), }, ) .await diff --git a/app/crowdb-web/src/lib.rs b/app/crowdb-web/src/lib.rs index 5ee0816d..24757e6a 100644 --- a/app/crowdb-web/src/lib.rs +++ b/app/crowdb-web/src/lib.rs @@ -98,6 +98,24 @@ pub fn router(state: AppState) -> axum::Router { .route( "/api/nodes/:id", delete(managed_hardware::remove_node).route_layer(authorization.clone()), + ) + .route( + "/api/nodes/:id/disk-groups", + get(managed_hardware::list_disk_groups) + .merge(post(managed_hardware::add_disk_group).route_layer(authorization.clone())), + ) + .route( + "/api/nodes/:id/disk-groups/:dg_id", + delete(managed_hardware::remove_disk_group).route_layer(authorization.clone()), + ) + .route( + "/api/nodes/:id/disk-groups/:dg_id/disks", + get(managed_hardware::list_disks) + .merge(post(managed_hardware::add_disk).route_layer(authorization.clone())), + ) + .route( + "/api/nodes/:id/disk-groups/:dg_id/disks/:disk_id", + delete(managed_hardware::remove_disk).route_layer(authorization.clone()), ); managed .merge(hardware) diff --git a/app/crowdb-web/src/managed_hardware.rs b/app/crowdb-web/src/managed_hardware.rs index daed7381..ca88d926 100644 --- a/app/crowdb-web/src/managed_hardware.rs +++ b/app/crowdb-web/src/managed_hardware.rs @@ -6,7 +6,7 @@ use axum::extract::{Path, Query, State}; use axum::http::StatusCode; use axum::Json; -use crowdb_console_shared::config::{NodeEntry, RackEntry}; +use crowdb_console_shared::config::{DiskEntry, DiskGroupEntry, NodeEntry, RackEntry}; use crowdb_console_shared::ops::hardware; use serde::Deserialize; @@ -28,6 +28,13 @@ pub(crate) struct NodeFilter { rack_id: Option, } +#[derive(Deserialize)] +pub(crate) struct CreateDiskGroup { + id: u64, + #[serde(default)] + name: String, +} + pub(crate) async fn list_racks(State(state): State) -> Result>, ApiError> { let ctx = state.op_context().await.map_err(api_error)?; hardware::list_racks_from_group0(&ctx) @@ -98,3 +105,71 @@ pub(crate) async fn remove_node( .map_err(api_error)?; Ok(StatusCode::NO_CONTENT) } + +pub(crate) async fn list_disk_groups( + State(state): State, + Path(node_id): Path, +) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_disk_groups_from_group0(&ctx, node_id) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn add_disk_group( + State(state): State, + Path(node_id): Path, + Json(body): Json, +) -> Result<(StatusCode, Json), ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + let group = hardware::add_disk_group_to_group0(&ctx, node_id, body.id, &body.name) + .await + .map_err(api_error)?; + Ok((StatusCode::CREATED, Json(group))) +} + +pub(crate) async fn remove_disk_group( + State(state): State, + Path((node_id, dg_id)): Path<(u64, u64)>, +) -> Result { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::remove_disk_group_from_group0(&ctx, node_id, dg_id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} + +pub(crate) async fn list_disks( + State(state): State, + Path((node_id, dg_id)): Path<(u64, u64)>, +) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_disks_from_group0(&ctx, node_id, dg_id) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn add_disk( + State(state): State, + Path((node_id, dg_id)): Path<(u64, u64)>, + Json(body): Json, +) -> Result<(StatusCode, Json), ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + let disk = hardware::add_disk_to_group0(&ctx, node_id, dg_id, &body) + .await + .map_err(api_error)?; + Ok((StatusCode::CREATED, Json(disk))) +} + +pub(crate) async fn remove_disk( + State(state): State, + Path((node_id, dg_id, disk_id)): Path<(u64, u64, String)>, +) -> Result { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::remove_disk_from_group0(&ctx, node_id, dg_id, &disk_id) + .await + .map_err(api_error)?; + Ok(StatusCode::NO_CONTENT) +} diff --git a/app/crowdb-web/tests/bare_metal_authority_test.rs b/app/crowdb-web/tests/bare_metal_authority_test.rs index 66e8cdd0..70110102 100644 --- a/app/crowdb-web/tests/bare_metal_authority_test.rs +++ b/app/crowdb-web/tests/bare_metal_authority_test.rs @@ -216,9 +216,96 @@ async fn bare_metal_hardware_routes_share_confirmed_group_zero_state() { .0, StatusCode::CONFLICT ); + verify_storage_hardware(&first, &second).await; verify_hardware_deletion(&first, &second).await; } +async fn verify_storage_hardware(first: &axum::Router, second: &axum::Router) { + let groups = "/api/nodes/9/disk-groups"; + let group = serde_json::json!({"id": 4, "name": "hot"}); + assert_eq!( + hardware_request( + first, + axum::http::Method::POST, + groups, + Some(group.clone()), + false + ) + .await + .0, + StatusCode::UNAUTHORIZED + ); + assert_eq!( + hardware_request(first, axum::http::Method::POST, groups, Some(group), true) + .await + .0, + StatusCode::CREATED + ); + assert_eq!( + hardware_request(second, axum::http::Method::GET, groups, None, false) + .await + .1[0]["name"], + "hot" + ); + let disks = "/api/nodes/9/disk-groups/4/disks"; + let disk = serde_json::json!({ + "disk_id": "0000000000000000-0000000000000009", "disk_type": "Ssd", + "capacity_bytes": 4096, "zone_size_bytes": 4096, "unit_size_bytes": 4096, + "device_path": "/dev/test", + }); + assert_eq!( + hardware_request(second, axum::http::Method::POST, disks, Some(disk), true) + .await + .0, + StatusCode::CREATED + ); + assert_eq!( + hardware_request(first, axum::http::Method::GET, disks, None, false) + .await + .1 + .as_array() + .unwrap() + .len(), + 1 + ); + assert_eq!( + hardware_request( + first, + axum::http::Method::DELETE, + "/api/nodes/9/disk-groups/4", + None, + true + ) + .await + .0, + StatusCode::CONFLICT + ); + assert_eq!( + hardware_request( + second, + axum::http::Method::DELETE, + "/api/nodes/9/disk-groups/4/disks/0000000000000000-0000000000000009", + None, + true + ) + .await + .0, + StatusCode::NO_CONTENT + ); + assert_eq!( + hardware_request( + first, + axum::http::Method::DELETE, + "/api/nodes/9/disk-groups/4", + None, + true + ) + .await + .0, + StatusCode::NO_CONTENT + ); +} + async fn verify_hardware_deletion(first: &axum::Router, second: &axum::Router) { assert_eq!( hardware_request(second, axum::http::Method::DELETE, "/api/nodes/9", None, false) diff --git a/container/crowdb-monitor/src/bootstrap/hardware.rs b/container/crowdb-monitor/src/bootstrap/hardware.rs index 7e835871..77f7a3d2 100644 --- a/container/crowdb-monitor/src/bootstrap/hardware.rs +++ b/container/crowdb-monitor/src/bootstrap/hardware.rs @@ -416,6 +416,7 @@ fn expected(profile: &DeploymentProfile) -> Result Vec { // ── disk ──────────────────────────────────────────────────────── /// Input for adding a disk. Mirrors the web handler's `AddDiskBody`. -#[derive(Debug, Clone)] +#[derive(Debug, Clone, serde::Deserialize)] pub struct AddDiskInput { pub disk_id: String, pub disk_type: String, diff --git a/lib/crowdb-console-shared/src/ops/hardware/authority_storage.rs b/lib/crowdb-console-shared/src/ops/hardware/authority_storage.rs new file mode 100644 index 00000000..3bd5d043 --- /dev/null +++ b/lib/crowdb-console-shared/src/ops/hardware/authority_storage.rs @@ -0,0 +1,429 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Rack-fenced disk-group and disk mutations in Group 0. + +use crowdb_kv_client::{BatchOp, Error as KvError, GetOutcome, ReadMode}; +use crowdb_protocol::common::{DiskId, HwStatus, NodeValue, RackValue}; +use crowdb_protocol::diskdb::rpc::{DiskGroupValue, DiskValue}; +use crowdb_protocol::key::{DiskGroupKey, DiskKey, NodeKey, RackKey, TextKey}; +use crowdb_protocol::DiskIdExt; + +use crate::config::{DiskEntry, DiskGroupEntry}; +use crate::error::{Error, Result}; +use crate::ops::hardware::{authority, validate_disk_input, AddDiskInput}; +use crate::ops::OpContext; + +const RETRIES: u64 = 10; + +async fn rack_revision(ctx: &OpContext, rack_id: u64) -> Result<(String, RackValue, u64)> { + let path = RackKey { rack_id }.to_path(); + let (value, revision) = required::(ctx, &path, "rack", rack_id.to_string()).await?; + Ok((path, value, revision)) +} + +async fn required( + ctx: &OpContext, + path: &str, + kind: &str, + id: String, +) -> Result<(T, u64)> { + match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, revision, .. } => Ok(( + serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?, + revision, + )), + GetOutcome::NotFound => Err(Error::NotFound { + kind: kind.into(), + id, + }), + } +} + +async fn optional(ctx: &OpContext, path: &str) -> Result> { + match ctx + .kv() + .get(0, 0, path.as_bytes(), ReadMode::Linearizable, None) + .await? + { + GetOutcome::Found { value, .. } => Ok(Some( + serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?, + )), + GetOutcome::NotFound => Ok(None), + } +} + +fn put(path: &str, value: &T) -> Result { + let encoded = serde_json::to_vec(value).map_err(|error| Error::Config(error.to_string()))?; + Ok(BatchOp::Put { + key: path.as_bytes().to_vec().into(), + value: encoded.into(), + }) +} + +fn delete(path: &str) -> BatchOp { + BatchOp::Delete { + key: path.as_bytes().to_vec().into(), + } +} + +async fn fenced( + ctx: &OpContext, + rack_path: &str, + rack: &RackValue, + revision: u64, + mut ops: Vec, +) -> Result { + ops.insert(0, put(rack_path, rack)?); + match ctx + .kv() + .batch_write_cas(0, 0, &ops, rack_path.as_bytes(), revision) + .await + { + Ok(_) => Ok(true), + Err(KvError::CasFailed { .. } | KvError::CasBusy | KvError::OutcomeUnknown) => Ok(false), + Err(error) => Err(error.into()), + } +} + +async fn pause(attempt: u64) { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1) * 5)).await; +} + +fn node_location(nodes: &[(u64, u64, NodeValue)], node_id: u64) -> Result { + nodes + .iter() + .find_map(|(rack, id, _)| (*id == node_id).then_some(*rack)) + .ok_or_else(|| Error::NotFound { + kind: "node".into(), + id: node_id.to_string(), + }) +} + +/// Create a disk group and update its node membership in one confirmed write. +/// +/// # Errors +/// Returns an authority error, a missing node, or a conflicting group. +pub async fn add_disk_group_to_group0( + ctx: &OpContext, + node_id: u64, + dg_id: u64, + name: &str, +) -> Result { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let entry = DiskGroupEntry { + id: dg_id, + rack_id, + node_id, + name: name.into(), + }; + let node_path = NodeKey { rack_id, node_id }.to_path(); + let group_path = DiskGroupKey { + rack_id, + node_id, + disk_group_id: dg_id, + } + .to_path(); + for attempt in 0..RETRIES { + let (rack_path, rack, revision) = rack_revision(ctx, rack_id).await?; + let (mut node, _) = required::(ctx, &node_path, "node", node_id.to_string()).await?; + let current = optional::(ctx, &group_path).await?; + if let Some(group) = current { + if group.name != name || !node.disk_group_ids.contains(&dg_id) { + return Err(Error::Conflict { + kind: "disk_group".into(), + id: dg_id.to_string(), + }); + } + return Ok(entry); + } + if node.disk_group_ids.contains(&dg_id) { + return Err(Error::Conflict { + kind: "disk_group membership".into(), + id: dg_id.to_string(), + }); + } + node.disk_group_ids.push(dg_id); + node.disk_group_ids.sort_unstable(); + node.last_used_dg_id = node.last_used_dg_id.max(dg_id); + let group = DiskGroupValue { + status: HwStatus::Up as i32, + disk_ids: Vec::new(), + name: name.into(), + }; + if fenced( + ctx, + &rack_path, + &rack, + revision, + vec![put(&node_path, &node)?, put(&group_path, &group)?], + ) + .await? + { + return Ok(entry); + } + pause(attempt).await; + } + Err(KvError::OutcomeUnknown.into()) +} + +/// Return confirmed disk groups on one node. +/// +/// # Errors +/// Returns an authority error or a missing node. +pub async fn list_disk_groups_from_group0(ctx: &OpContext, node_id: u64) -> Result> { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let mut groups: Vec<_> = ctx + .sysmd() + .list_disk_groups_on_node(rack_id, node_id) + .await? + .into_iter() + .map(|group| DiskGroupEntry { + id: group.dg_id, + rack_id, + node_id, + name: group.value.name, + }) + .collect(); + groups.sort_unstable_by_key(|group| group.id); + Ok(groups) +} + +/// Delete an empty, unowned disk group and remove node membership atomically. +/// +/// # Errors +/// Returns an authority error, missing record, or child/assignment conflict. +pub async fn remove_disk_group_from_group0(ctx: &OpContext, node_id: u64, dg_id: u64) -> Result<()> { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let node_path = NodeKey { rack_id, node_id }.to_path(); + let group_path = DiskGroupKey { + rack_id, + node_id, + disk_group_id: dg_id, + } + .to_path(); + let mut uncertain = false; + for attempt in 0..RETRIES { + let (rack_path, rack, revision) = rack_revision(ctx, rack_id).await?; + let (mut node, _) = required::(ctx, &node_path, "node", node_id.to_string()).await?; + let group = optional::(ctx, &group_path).await?; + if group.is_none() && uncertain && !node.disk_group_ids.contains(&dg_id) { + return Ok(()); + } + let group = group.ok_or_else(|| Error::NotFound { + kind: "disk_group".into(), + id: dg_id.to_string(), + })?; + if !group.disk_ids.is_empty() + || !ctx + .sysmd() + .list_disks_in_group(rack_id, node_id, dg_id) + .await? + .is_empty() + || ctx.sysmd().get_owner(rack_id, node_id, dg_id).await?.is_some() + || ctx.sysmd().get_bind(rack_id, node_id, dg_id).await?.is_some() + { + return Err(Error::Conflict { + kind: "disk_group with children or assignment".into(), + id: dg_id.to_string(), + }); + } + if !node.disk_group_ids.contains(&dg_id) { + return Err(Error::Conflict { + kind: "disk_group membership".into(), + id: dg_id.to_string(), + }); + } + node.disk_group_ids.retain(|id| *id != dg_id); + if fenced( + ctx, + &rack_path, + &rack, + revision, + vec![put(&node_path, &node)?, delete(&group_path)], + ) + .await? + { + return Ok(()); + } + uncertain = true; + pause(attempt).await; + } + Err(KvError::OutcomeUnknown.into()) +} + +/// Add a validated disk and update its group membership in one confirmed write. +/// +/// # Errors +/// Returns a validation, authority, missing group, or conflicting disk error. +pub async fn add_disk_to_group0( + ctx: &OpContext, + node_id: u64, + dg_id: u64, + input: &AddDiskInput, +) -> Result { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let (entry, disk_id, value) = validate_disk_input(input, dg_id, rack_id, node_id)?; + let group_path = DiskGroupKey { + rack_id, + node_id, + disk_group_id: dg_id, + } + .to_path(); + let disk_path = DiskKey { + rack_id, + node_id, + disk_group_id: dg_id, + disk_id, + } + .to_path(); + for attempt in 0..RETRIES { + let (rack_path, rack, revision) = rack_revision(ctx, rack_id).await?; + let (mut group, _) = + required::(ctx, &group_path, "disk_group", dg_id.to_string()).await?; + if let Some(actual) = optional::(ctx, &disk_path).await? { + if actual == value && group.disk_ids.contains(&disk_id) { + return Ok(entry); + } + return Err(Error::Conflict { + kind: "disk".into(), + id: input.disk_id.clone(), + }); + } + if group.disk_ids.contains(&disk_id) { + return Err(Error::Conflict { + kind: "disk membership".into(), + id: input.disk_id.clone(), + }); + } + group.disk_ids.push(disk_id); + group.disk_ids.sort_unstable_by_key(|id| (id.high, id.low)); + if fenced( + ctx, + &rack_path, + &rack, + revision, + vec![put(&group_path, &group)?, put(&disk_path, &value)?], + ) + .await? + { + return Ok(entry); + } + pause(attempt).await; + } + Err(KvError::OutcomeUnknown.into()) +} + +/// Return confirmed disks on one node and disk group. +/// +/// # Errors +/// Returns an authority error or a missing node. +pub async fn list_disks_from_group0(ctx: &OpContext, node_id: u64, dg_id: u64) -> Result> { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let mut disks = Vec::new(); + for (disk_id, value) in ctx.sysmd().list_disks_in_group(rack_id, node_id, dg_id).await? { + disks.push(disk_entry(rack_id, node_id, dg_id, disk_id, &value)); + } + disks.sort_unstable_by(|a, b| a.disk_id.cmp(&b.disk_id)); + Ok(disks) +} + +/// Remove a disk and its group membership in one confirmed write. +/// +/// # Errors +/// Returns an authority error, missing disk, or membership conflict. +pub async fn remove_disk_from_group0( + ctx: &OpContext, + node_id: u64, + dg_id: u64, + disk_id: &str, +) -> Result { + authority::ready(ctx).await?; + let rack_id = node_location(&ctx.sysmd().list_nodes().await?, node_id)?; + let id = DiskId::from_display_string(disk_id).map_err(|message| Error::Validation { + field: "disk_id".into(), + message, + })?; + let group_path = DiskGroupKey { + rack_id, + node_id, + disk_group_id: dg_id, + } + .to_path(); + let disk_path = DiskKey { + rack_id, + node_id, + disk_group_id: dg_id, + disk_id: id, + } + .to_path(); + let mut removed = None; + for attempt in 0..RETRIES { + let (rack_path, rack, revision) = rack_revision(ctx, rack_id).await?; + let (mut group, _) = + required::(ctx, &group_path, "disk_group", dg_id.to_string()).await?; + let current = optional::(ctx, &disk_path).await?; + if current.is_none() && !group.disk_ids.contains(&id) { + return removed.ok_or_else(|| Error::NotFound { + kind: "disk".into(), + id: disk_id.into(), + }); + } + let value = current.ok_or_else(|| Error::Conflict { + kind: "disk membership".into(), + id: disk_id.into(), + })?; + if !group.disk_ids.contains(&id) { + return Err(Error::Conflict { + kind: "disk membership".into(), + id: disk_id.into(), + }); + } + let entry = disk_entry(rack_id, node_id, dg_id, id, &value); + group.disk_ids.retain(|candidate| *candidate != id); + if fenced( + ctx, + &rack_path, + &rack, + revision, + vec![put(&group_path, &group)?, delete(&disk_path)], + ) + .await? + { + return Ok(entry); + } + removed = Some(entry); + pause(attempt).await; + } + Err(KvError::OutcomeUnknown.into()) +} + +fn disk_entry(rack_id: u64, node_id: u64, dg_id: u64, disk_id: DiskId, value: &DiskValue) -> DiskEntry { + DiskEntry { + disk_id: disk_id.to_display_string(), + rack_id, + node_id, + disk_group_id: dg_id, + disk_type: match crowdb_protocol::diskdb::rpc::DiskType::try_from(value.disk_type) { + Ok(kind) => format!("{kind:?}"), + Err(()) => value.disk_type.to_string(), + }, + capacity_bytes: value + .capacity_units + .saturating_mul(u64::from(value.unit_size_bytes)), + zone_size_bytes: value + .zone_size_units + .saturating_mul(u64::from(value.unit_size_bytes)), + unit_size_bytes: value.unit_size_bytes, + device_path: value.device_path.clone(), + } +} diff --git a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs index fa763535..f8d8d80b 100644 --- a/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs +++ b/lib/crowdb-console-shared/tests/ops_hardware_authority_test.rs @@ -176,3 +176,122 @@ async fn committed_rack_survives_a_lost_conditional_write_response() { assert_eq!(proxy.dropped.load(Ordering::SeqCst), 1); assert_eq!(ctx.sysmd().get_rack(9).await.unwrap().unwrap().name, "rack-nine"); } + +#[tokio::test] +async fn two_consoles_share_disk_groups_and_disks_without_local_topology() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + cluster_ops::init(&bootstrap, &[1]).await.unwrap(); + let first = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + let second = OpContext::new( + cluster.group0_leader_endpoint.clone(), + cluster.mgmt_endpoints.clone(), + ConsoleConfig::default(), + ); + hardware::add_rack_to_group0(&first, 12, "storage").await.unwrap(); + hardware::add_node_to_group0( + &first, + NodeEntry { + id: 12, + rack_id: 12, + host: "127.0.0.1".into(), + ssh_port: 22, + ssh_user: String::new(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: None, + }, + ) + .await + .unwrap(); + hardware::add_disk_group_to_group0(&first, 12, 5, "hot") + .await + .unwrap(); + hardware::add_disk_group_to_group0(&second, 12, 5, "hot") + .await + .unwrap(); + let conflict = hardware::add_disk_group_to_group0(&second, 12, 5, "cold") + .await + .unwrap_err(); + assert!(matches!(conflict, Error::Conflict { .. })); + assert_eq!( + hardware::list_disk_groups_from_group0(&second, 12).await.unwrap()[0].name, + "hot" + ); + let disk = hardware::AddDiskInput { + disk_id: "0000000000000000-000000000000000c".into(), + disk_type: "Ssd".into(), + capacity_bytes: 4096, + zone_size_bytes: 4096, + unit_size_bytes: 4096, + device_path: "/dev/test".into(), + }; + hardware::add_disk_to_group0(&first, 12, 5, &disk).await.unwrap(); + hardware::add_disk_to_group0(&second, 12, 5, &disk).await.unwrap(); + assert_eq!( + hardware::list_disks_from_group0(&second, 12, 5) + .await + .unwrap() + .len(), + 1 + ); + assert!(matches!( + hardware::remove_disk_group_from_group0(&second, 12, 5).await, + Err(Error::Conflict { .. }) + )); + assert!(matches!( + hardware::remove_node_from_group0(&second, 12).await, + Err(Error::Conflict { .. }) + )); + hardware::remove_disk_from_group0(&second, 12, 5, &disk.disk_id) + .await + .unwrap(); + hardware::remove_disk_group_from_group0(&first, 12, 5) + .await + .unwrap(); + hardware::remove_node_from_group0(&second, 12).await.unwrap(); + assert!(first.config().disk_groups.is_empty()); + assert!(second.config().disks.is_empty()); +} + +#[tokio::test] +async fn committed_disk_group_survives_a_lost_response() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + cluster_ops::init(&bootstrap, &[1]).await.unwrap(); + let proxy = rpc_response_proxy::TestResponseProxy::start(cluster.group0_leader_endpoint.clone()).await; + let ctx = OpContext::new( + proxy.endpoint.clone(), + vec![proxy.management_endpoint.clone()], + ConsoleConfig::default(), + ); + hardware::add_rack_to_group0(&ctx, 13, "storage").await.unwrap(); + hardware::add_node_to_group0( + &ctx, + NodeEntry { + id: 13, + rack_id: 13, + host: "127.0.0.1".into(), + ssh_port: 22, + ssh_user: String::new(), + ssh_key: None, + ssh_password: None, + ssh_credential_ref: None, + }, + ) + .await + .unwrap(); + proxy.armed.store(true, Ordering::SeqCst); + hardware::add_disk_group_to_group0(&ctx, 13, 1, "recovered") + .await + .unwrap(); + assert_eq!(proxy.dropped.load(Ordering::SeqCst), 1); + assert_eq!( + hardware::list_disk_groups_from_group0(&ctx, 13).await.unwrap()[0].name, + "recovered" + ); +} diff --git a/lib/crowdb-diskio-client/tests/disk_io_group0_sync_test.rs b/lib/crowdb-diskio-client/tests/disk_io_group0_sync_test.rs index f7c7a348..4a10d7c8 100644 --- a/lib/crowdb-diskio-client/tests/disk_io_group0_sync_test.rs +++ b/lib/crowdb-diskio-client/tests/disk_io_group0_sync_test.rs @@ -149,6 +149,7 @@ async fn disk_io_e2e_group0_sync() { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: all_disk_ids, + name: String::new(), }, ) .await diff --git a/lib/crowdb-protocol/src/types/diskdb.rs b/lib/crowdb-protocol/src/types/diskdb.rs index fa8aa3c5..6f8d4844 100644 --- a/lib/crowdb-protocol/src/types/diskdb.rs +++ b/lib/crowdb-protocol/src/types/diskdb.rs @@ -150,6 +150,8 @@ pub struct DiskValue { pub struct DiskGroupValue { pub status: i32, pub disk_ids: Vec, + #[serde(default, skip_serializing_if = "String::is_empty")] + pub name: String, } #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Default, Serialize, Deserialize)] diff --git a/lib/crowdb-test-harness/src/hardware.rs b/lib/crowdb-test-harness/src/hardware.rs index edad8a50..cb635526 100644 --- a/lib/crowdb-test-harness/src/hardware.rs +++ b/lib/crowdb-test-harness/src/hardware.rs @@ -61,6 +61,7 @@ pub async fn seed_hardware(hw: &HardwareClient, disk_ids: &[DiskId]) { &DiskGroupValue { status: HwStatus::Up as i32, disk_ids: disk_ids.to_vec(), + name: String::new(), }, ) .await From ff9ee0091decd529be160779ead44c96683f453c Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 14:18:31 +0800 Subject: [PATCH 55/74] Resume interrupted S3 bootstrap from confirmed authority --- app/crowdb-cli/tests/s3_cli_test.rs | 41 ++++++ doc/working/plan-console-authority.md | 9 +- lib/crowdb-console-shared/src/ops/cluster.rs | 110 +++++++++++++--- .../src/ops/cluster/bootstrap/publication.rs | 34 ++++- lib/crowdb-console-shared/src/ops/s3.rs | 121 +++++++++++++++++- .../tests/ops_bootstrap_publication_test.rs | 38 +++++- 6 files changed, 327 insertions(+), 26 deletions(-) diff --git a/app/crowdb-cli/tests/s3_cli_test.rs b/app/crowdb-cli/tests/s3_cli_test.rs index c46a200b..29bfe2cb 100644 --- a/app/crowdb-cli/tests/s3_cli_test.rs +++ b/app/crowdb-cli/tests/s3_cli_test.rs @@ -1,6 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. +use crowdb_test_harness::test_dirs::TestDir; use std::io::{Read, Write}; use std::net::TcpListener; use std::path::{Path, PathBuf}; @@ -11,6 +12,46 @@ fn cli() -> Command { Command::new(env!("CARGO_BIN_EXE_crowdb-cli")) } +#[test] +#[ignore = "starts the complete local storage stack twice"] +fn interrupted_s3_launch_resumes_confirmed_group_zero_without_topology_file() { + let directory = TestDir::new("s3-bootstrap-replay-cli").expect("test directory"); + let root = directory.path(); + let failed = cli() + .args(["s3", "cluster", "start", "--root"]) + .arg(root) + .env("CROWDB_CHUNK_KV_SERVER_BIN", "/bin/false") + .output() + .expect("run interrupted launch"); + assert!(!failed.status.success(), "failure injection must stop launch"); + assert!(root.join("s3-mini-cluster.initializing.json").exists()); + assert!(root.join("s3-local-state.toml").exists()); + assert!(!root.join("console.toml").exists()); + let restarted = cli() + .args(["s3", "cluster", "start", "--root"]) + .arg(root) + .env_remove("CROWDB_CHUNK_KV_SERVER_BIN") + .output() + .expect("resume interrupted launch"); + assert!( + restarted.status.success(), + "{}", + String::from_utf8_lossy(&restarted.stderr) + ); + assert!(!root.join("bootstrap-intent.toml").exists()); + assert!(!root.join("s3-mini-cluster.initializing.json").exists()); + let deleted = cli() + .args(["s3", "cluster", "delete", "--root"]) + .arg(root) + .output() + .expect("delete test cluster"); + assert!( + deleted.status.success(), + "{}", + String::from_utf8_lossy(&deleted.stderr) + ); +} + fn tempdir(tag: &str) -> PathBuf { let nonce = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 7def9ce8..0570ba36 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -176,7 +176,14 @@ configuration, documentation and crash-diagnostics tasks below are pending. bootstrap-interruption replay before the old format can be removed. Its completed cluster now restarts from launch-only local state and Group 0 seeds; no local topology is loaded after publication. A full persistent S3 - stop/restart and range-read E2E passes. + stop/restart and range-read E2E passes. S3 now saves its local KV launch + state before sealing bootstrap intent and publishing Group 0. An interrupted + launch retains that state for identity-checked retry instead of archiving the + committed cluster. A real failure injected at Chunk KV startup recovers on + the next CLI invocation, with all 15 services ready and no topology file. + The CLI integration test reproduces this interruption and recovery. Partial + storage-service launch sets still fail closed and need completion or an + explicit operator recovery path; mixed CLI/Web config remains to remove. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with diff --git a/lib/crowdb-console-shared/src/ops/cluster.rs b/lib/crowdb-console-shared/src/ops/cluster.rs index b00e11b2..8f5d6e51 100644 --- a/lib/crowdb-console-shared/src/ops/cluster.rs +++ b/lib/crowdb-console-shared/src/ops/cluster.rs @@ -477,8 +477,22 @@ pub async fn local_deploy_combined( diskio_dummy_disk_type: &str, ) -> Result { local_deploy(ctx, 3, Some(workspace), tunables).await?; + local_deploy_combined_after_kv(ctx, workspace, disk, chunk, diskio_dummy_disk_type).await +} + +/// Complete the local storage stack after a verified KV bootstrap. +/// +/// # Errors +/// Returns a provisioning or readiness error. +pub async fn local_deploy_combined_after_kv( + ctx: &OpContext, + workspace: &std::path::Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, + diskio_dummy_disk_type: &str, +) -> Result { for group_id in &disk.data_groups { - crate::ops::kv_logical::add_group(ctx, 0, *group_id, 100 + *group_id, &[1, 2, 3]).await?; + ensure_local_data_group(ctx, *group_id).await?; } let diskdb = local_deploy_diskdb(ctx, workspace, disk).await?; let diskio = local_deploy_diskio( @@ -514,8 +528,21 @@ pub async fn local_deploy_combined_file_backed( chunk: &LocalChunkdbDeployConfig, ) -> Result { local_deploy(ctx, 3, Some(workspace), tunables).await?; + local_deploy_combined_file_backed_after_kv(ctx, workspace, disk, chunk).await +} + +/// Complete the file-backed storage stack after a verified KV bootstrap. +/// +/// # Errors +/// Returns a provisioning or readiness error. +pub async fn local_deploy_combined_file_backed_after_kv( + ctx: &OpContext, + workspace: &std::path::Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, +) -> Result { for group_id in &disk.data_groups { - crate::ops::kv_logical::add_group(ctx, 0, *group_id, 100 + *group_id, &[1, 2, 3]).await?; + ensure_local_data_group(ctx, *group_id).await?; } let diskdb = local_deploy_diskdb(ctx, workspace, disk).await?; let diskio = local_deploy_diskio( @@ -537,6 +564,28 @@ pub async fn local_deploy_combined_file_backed( }) } +async fn ensure_local_data_group(ctx: &OpContext, group_id: u64) -> Result<()> { + let group = ctx.sysmd().get_group(0, group_id).await?; + let replicas = ctx.sysmd().list_replicas_in_group(0, group_id).await?; + if group.is_none() && replicas.is_empty() { + return crate::ops::kv_logical::add_group(ctx, 0, group_id, 100 + group_id, &[1, 2, 3]).await; + } + let mut actual: Vec<_> = replicas + .iter() + .map(|replica| (replica.replica_id, replica.node_id)) + .collect(); + actual.sort_unstable(); + let expected = vec![(100 + group_id, 1), (101 + group_id, 2), (102 + group_id, 3)]; + if group.is_some() && actual == expected { + Ok(()) + } else { + Err(Error::Conflict { + kind: "local data group".into(), + id: format!("0/{group_id}"), + }) + } +} + async fn local_deploy_diskio( ctx: &OpContext, workspace: &std::path::Path, @@ -975,8 +1024,13 @@ async fn provision_diskdb_topology( for node in nodes { for local_group in 0..cfg.disk_groups_per_node { let disk_group_id = node.id * 100 + u64::try_from(local_group).unwrap_or(u64::MAX) + 1; - hardware::add_disk_group(ctx, node.id, disk_group_id, &format!("bench-dg-{disk_group_id}")) - .await?; + hardware::add_disk_group_to_group0( + ctx, + node.id, + disk_group_id, + &format!("bench-dg-{disk_group_id}"), + ) + .await?; let disks = (0..cfg.disks_per_group) .map(|disk| AddDiskInput { disk_id: format!("{:016x}{:016x}", disk_group_id, disk + 1), @@ -987,15 +1041,26 @@ async fn provision_diskdb_topology( device_path: String::new(), }) .collect::>(); - hardware::add_disks_batch(ctx, node.id, disk_group_id, &disks).await?; + for disk in &disks { + hardware::add_disk_to_group0(ctx, node.id, disk_group_id, disk).await?; + } let instance_id = 10_000 + node.id; ctx.sysmd() .set_owner(node.rack_id, node.id, disk_group_id, instance_id, lease_expiry_ms) .await?; let data_group = cfg.data_groups[disk_group_count % cfg.data_groups.len()]; - ctx.sysmd() - .set_bind(node.rack_id, node.id, disk_group_id, 0, data_group) - .await?; + if let Some(binding) = ctx.sysmd().get_bind(node.rack_id, node.id, disk_group_id).await? { + if binding.store_id != 0 || binding.group_id != data_group { + return Err(Error::Conflict { + kind: "disk group binding".into(), + id: disk_group_id.to_string(), + }); + } + } else { + ctx.sysmd() + .set_bind(node.rack_id, node.id, disk_group_id, 0, data_group) + .await?; + } disk_group_count += 1; disk_count += disks.len(); } @@ -1162,6 +1227,26 @@ pub async fn local_deploy( workspace_dir: Option<&std::path::Path>, tunables: Option<&KvDeployTunables>, ) -> Result { + let (rack_id, node_ids) = prepare_local_deploy(ctx, node_count, workspace_dir, tunables).await?; + let init_summary = init(ctx, &node_ids).await?; + Ok(LocalDeploySummary { + node_count, + rack_id, + node_ids, + init_summary, + }) +} + +/// Start local KV processes and retain their bootstrap inputs without writing Group 0. +/// +/// # Errors +/// Returns a validation, binary, spawn, or readiness error. +pub async fn prepare_local_deploy( + ctx: &OpContext, + node_count: usize, + workspace_dir: Option<&std::path::Path>, + tunables: Option<&KvDeployTunables>, +) -> Result<(u64, Vec)> { if node_count == 0 { return Err(Error::Validation { field: "node_count".into(), @@ -1202,14 +1287,7 @@ pub async fn local_deploy( } } - let init_summary = init(ctx, &node_ids).await?; - - Ok(LocalDeploySummary { - node_count, - rack_id, - node_ids, - init_summary, - }) + Ok((rack_id, node_ids)) } /// Default workspace path for `local_deploy` when no explicit diff --git a/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs b/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs index d81b7d59..48ea5224 100644 --- a/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs +++ b/lib/crowdb-console-shared/src/ops/cluster/bootstrap/publication.rs @@ -32,7 +32,7 @@ impl Record { GetOutcome::Found { value, .. } => { let actual: serde_json::Value = serde_json::from_slice(&value).map_err(|error| Error::Config(error.to_string()))?; - if actual == self.value { + if self.same_identity(ctx, &actual)? { Ok(true) } else { Err(Error::Conflict { @@ -44,6 +44,38 @@ impl Record { } } + fn same_identity(&self, ctx: &OpContext, actual: &serde_json::Value) -> Result { + if self.key.starts_with("/hw/rack/") { + let intended: RackValue = serde_json::from_value(self.value.clone()) + .map_err(|error| Error::Config(error.to_string()))?; + let actual: RackValue = + serde_json::from_value(actual.clone()).map_err(|error| Error::Config(error.to_string()))?; + let rack_id = RackKey::from_path(&self.key) + .map_err(|error| Error::Config(error.to_string()))? + .rack_id; + let expected_nodes: Vec<_> = ctx + .config() + .nodes + .iter() + .filter(|node| node.rack_id == rack_id) + .map(|node| node.id) + .collect(); + return Ok(actual.name == intended.name + && actual.node_ids.iter().all(|node| expected_nodes.contains(node))); + } + if self.key.starts_with("/hw/node/") { + let intended: NodeValue = serde_json::from_value(self.value.clone()) + .map_err(|error| Error::Config(error.to_string()))?; + let actual: NodeValue = + serde_json::from_value(actual.clone()).map_err(|error| Error::Config(error.to_string()))?; + return Ok(actual.management_host == intended.management_host + && actual.ssh_port == intended.ssh_port + && actual.ssh_user == intended.ssh_user + && actual.ssh_credential_ref == intended.ssh_credential_ref); + } + Ok(actual == &self.value) + } + async fn create(&self, ctx: &OpContext) -> Result<()> { let payload = serde_json::to_vec(&self.value).map_err(|error| Error::Config(error.to_string()))?; match ctx.kv().put_cas(0, 0, self.key.as_bytes(), &payload, 0).await { diff --git a/lib/crowdb-console-shared/src/ops/s3.rs b/lib/crowdb-console-shared/src/ops/s3.rs index 8a38a1cd..1160bc1e 100644 --- a/lib/crowdb-console-shared/src/ops/s3.rs +++ b/lib/crowdb-console-shared/src/ops/s3.rs @@ -22,6 +22,7 @@ mod local_state; const MARKER_FILE: &str = "s3-mini-cluster.json"; const INITIALIZING_FILE: &str = "s3-mini-cluster.initializing.json"; +const BOOTSTRAP_INTENT_FILE: &str = "bootstrap-intent.toml"; const MASTER_KEY: &str = "1111111111111111111111111111111111111111111111111111111111111111"; const NAMESPACE_ID: &str = "s3-mini-cluster"; const BODY_PREVIEW_LIMIT: usize = 64 * 1024; @@ -175,6 +176,9 @@ async fn start_with_profile( diskdb_client_rpc_workers: None, metrics_interval: None, }; + if let Some(status) = resume_if_interrupted(data_dir, &disk, &chunk, storage_profile).await? { + return Ok(status); + } let mut record = MiniClusterRecord { version: 1, endpoint: String::new(), @@ -202,10 +206,33 @@ async fn start_with_profile( Ok(status) } +async fn resume_if_interrupted( + data_dir: &Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, + storage_profile: StorageProfile, +) -> Result> { + if !data_dir.join(INITIALIZING_FILE).exists() || !local_state::path(data_dir).exists() { + return Ok(None); + } + let record: MiniClusterRecord = serde_json::from_slice(&std::fs::read(data_dir.join(INITIALIZING_FILE))?) + .map_err(|error| Error::Config(error.to_string()))?; + if record.version != 1 || record.storage_profile != storage_profile { + return Err(Error::Conflict { + kind: "S3 bootstrap profile".into(), + id: data_dir.display().to_string(), + }); + } + resume_incomplete(data_dir, disk, chunk, record).await.map(Some) +} + fn archive_incomplete_attempt(data_dir: &Path) -> Result<()> { if !data_dir.join(INITIALIZING_FILE).exists() || data_dir.join(MARKER_FILE).exists() { return Ok(()); } + if local_state::path(data_dir).exists() || data_dir.join(BOOTSTRAP_INTENT_FILE).exists() { + return Ok(()); + } let name = data_dir .file_name() .and_then(|value| value.to_str()) @@ -240,13 +267,54 @@ async fn initialize_new( no_fsync: (storage_profile == StorageProfile::Memory).then_some(true), ..KvDeployTunables::default() }; - match storage_profile { - StorageProfile::Persistent => { - cluster::local_deploy_combined_file_backed(ctx, data_dir, Some(&tunables), disk, chunk).await?; + let (_, nodes) = cluster::prepare_local_deploy(ctx, 3, Some(data_dir), Some(&tunables)).await?; + local_state::save(data_dir, &ctx.config())?; + cluster::init_with_intent(ctx, &nodes, &data_dir.join(BOOTSTRAP_INTENT_FILE)).await?; + initialize_after_kv(ctx, data_dir, disk, chunk, storage_profile).await +} + +async fn initialize_after_kv( + ctx: &OpContext, + data_dir: &Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, + storage_profile: StorageProfile, +) -> Result { + let storage_services = ctx + .config() + .servers + .iter() + .filter(|server| { + matches!( + server.service_type, + ServiceType::Diskdb | ServiceType::Diskio | ServiceType::Chunkdb + ) + }) + .count(); + if storage_services == 9 { + for group in &disk.data_groups { + if ctx.sysmd().get_group(0, *group).await?.is_none() { + return Err(Error::NotFound { + kind: "S3 data group".into(), + id: group.to_string(), + }); + } } - StorageProfile::Memory => { - cluster::local_deploy_combined(ctx, data_dir, Some(&tunables), disk, chunk, "mem").await?; + cluster::restart_storage_services(ctx).await?; + } else if storage_services == 0 { + match storage_profile { + StorageProfile::Persistent => { + cluster::local_deploy_combined_file_backed_after_kv(ctx, data_dir, disk, chunk).await?; + } + StorageProfile::Memory => { + cluster::local_deploy_combined_after_kv(ctx, data_dir, disk, chunk, "mem").await?; + } } + } else { + return Err(Error::Conflict { + kind: "partial S3 storage launch state".into(), + id: format!("{storage_services} of 9 services"), + }); } let seeds = management_seeds(&ctx.config()); local_state::save(data_dir, &ctx.config())?; @@ -265,6 +333,42 @@ async fn initialize_new( }) } +async fn resume_incomplete( + data_dir: &Path, + disk: &LocalDiskdbDeployConfig, + chunk: &LocalChunkdbDeployConfig, + mut record: MiniClusterRecord, +) -> Result { + let (mut config, seeds) = local_state::load(data_dir)?; + restore_launch_nodes(&mut config)?; + let group0 = config + .servers + .iter() + .find(|server| server.service_type == ServiceType::Kv) + .and_then(|server| server.rpc_url.as_deref()) + .ok_or_else(|| Error::Config("S3 bootstrap has no KV RPC seed".into()))? + .trim_start_matches("http://") + .to_owned(); + let ctx = OpContext::new(group0, seeds.clone(), config); + for node_id in 1..=3 { + let server_dir = data_dir + .join("rack1") + .join(format!("node{node_id}")) + .join(format!("kv-server-{node_id}")); + crate::ops::kv_server::restart(&ctx, node_id, Some(&server_dir), None, &seeds).await?; + } + local_state::save(data_dir, &ctx.config())?; + cluster::init_with_intent(&ctx, &[1, 2, 3], &data_dir.join(BOOTSTRAP_INTENT_FILE)).await?; + let endpoints = initialize_after_kv(&ctx, data_dir, disk, chunk, record.storage_profile).await?; + record.endpoint = endpoints.s3; + record.web_endpoint = endpoints.web; + record.web_pid = Some(endpoints.web_pid); + save_record(&data_dir.join(MARKER_FILE), &record)?; + std::fs::remove_file(data_dir.join(INITIALIZING_FILE))?; + let status = status_from(data_dir, false, &ctx.config(), &record); + Ok(status) +} + async fn restart(data_dir: &Path) -> Result { let (mut config, mut record) = load(data_dir)?; restore_launch_nodes(&mut config)?; @@ -395,7 +499,10 @@ fn validate_location(data_dir: &Path) -> Result<()> { return Ok(()); } let mut entries = std::fs::read_dir(data_dir)?; - if entries.next().transpose()?.is_none() || data_dir.join(MARKER_FILE).exists() { + if entries.next().transpose()?.is_none() + || data_dir.join(MARKER_FILE).exists() + || (data_dir.join(INITIALIZING_FILE).exists() && local_state::path(data_dir).exists()) + { return Ok(()); } Err(Error::Validation { @@ -427,7 +534,7 @@ fn management_seeds(config: &ConsoleConfig) -> Vec { fn restore_launch_nodes(config: &mut ConsoleConfig) -> Result<()> { config.add_rack(RackEntry { id: 1, - name: "local-launch".into(), + name: "rack-1".into(), })?; let node_ids: Vec<_> = config .servers diff --git a/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs b/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs index 6404a3d2..f3802124 100644 --- a/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs +++ b/lib/crowdb-console-shared/tests/ops_bootstrap_publication_test.rs @@ -3,7 +3,7 @@ use crowdb_console_shared::error::Error; use crowdb_console_shared::ops::cluster; -use crowdb_protocol::common::{HwStatus, RackValue}; +use crowdb_protocol::common::{HwStatus, NodeValue, RackValue}; use crowdb_protocol::TextKey; use crowdb_test_harness::cluster::KvCluster; #[path = "common/bootstrap_authority.rs"] @@ -27,6 +27,42 @@ async fn bootstrap_rejects_conflicting_hardware_without_overwriting_authority() assert!(ctx.config().stores.is_empty()); } +#[tokio::test] +async fn bootstrap_retry_preserves_known_rack_membership_and_node_runtime_fields() { + let cluster = KvCluster::start().await; + let ctx = context(&cluster).await; + ctx.sysmd() + .add_rack( + 1, + &RackValue { + status: HwStatus::Maintenance as i32, + node_ids: vec![1], + name: String::new(), + }, + ) + .await + .unwrap(); + ctx.sysmd() + .add_node( + 1, + 1, + &NodeValue { + status: HwStatus::Maintenance as i32, + management_host: "127.0.0.1".into(), + ssh_port: 22, + disk_group_ids: vec![42], + ..Default::default() + }, + ) + .await + .unwrap(); + cluster::init(&ctx, &[1]).await.unwrap(); + assert_eq!(ctx.sysmd().get_rack(1).await.unwrap().unwrap().node_ids, vec![1]); + let node = ctx.sysmd().get_node(1, 1).await.unwrap().unwrap(); + assert_eq!(node.status, HwStatus::Maintenance as i32); + assert_eq!(node.disk_group_ids, vec![42]); +} + #[tokio::test] async fn bootstrap_preflights_logical_conflicts_before_publishing_missing_hardware() { let cluster = KvCluster::start().await; From 6e9367ec021097a82df2364bb9b3b12272c7b8e7 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 14:26:08 +0800 Subject: [PATCH 56/74] Require versioned Web startup configuration --- app/crowdb-web/src/lib.rs | 17 +++- app/crowdb-web/src/main.rs | 22 ++--- app/crowdb-web/src/managed_hardware.rs | 88 +++++++++++++++++++ .../tests/bare_metal_authority_test.rs | 46 ++++++++++ app/crowdb-web/tests/managed_mode_test.rs | 16 ++-- doc/working/plan-console-authority.md | 6 +- 6 files changed, 168 insertions(+), 27 deletions(-) diff --git a/app/crowdb-web/src/lib.rs b/app/crowdb-web/src/lib.rs index 24757e6a..c0e40bab 100644 --- a/app/crowdb-web/src/lib.rs +++ b/app/crowdb-web/src/lib.rs @@ -88,7 +88,12 @@ pub fn router(state: AppState) -> axum::Router { ) .route( "/api/racks/:rack_id", - delete(managed_hardware::remove_rack).route_layer(authorization.clone()), + get(managed_hardware::get_rack) + .merge(delete(managed_hardware::remove_rack).route_layer(authorization.clone())), + ) + .route( + "/api/racks/:rack_id/nodes", + get(managed_hardware::list_rack_nodes), ) .route( "/api/nodes", @@ -97,7 +102,8 @@ pub fn router(state: AppState) -> axum::Router { ) .route( "/api/nodes/:id", - delete(managed_hardware::remove_node).route_layer(authorization.clone()), + get(managed_hardware::get_node) + .merge(delete(managed_hardware::remove_node).route_layer(authorization.clone())), ) .route( "/api/nodes/:id/disk-groups", @@ -106,7 +112,9 @@ pub fn router(state: AppState) -> axum::Router { ) .route( "/api/nodes/:id/disk-groups/:dg_id", - delete(managed_hardware::remove_disk_group).route_layer(authorization.clone()), + get(managed_hardware::get_disk_group).merge( + delete(managed_hardware::remove_disk_group).route_layer(authorization.clone()), + ), ) .route( "/api/nodes/:id/disk-groups/:dg_id/disks", @@ -115,7 +123,8 @@ pub fn router(state: AppState) -> axum::Router { ) .route( "/api/nodes/:id/disk-groups/:dg_id/disks/:disk_id", - delete(managed_hardware::remove_disk).route_layer(authorization.clone()), + get(managed_hardware::get_disk) + .merge(delete(managed_hardware::remove_disk).route_layer(authorization.clone())), ); managed .merge(hardware) diff --git a/app/crowdb-web/src/main.rs b/app/crowdb-web/src/main.rs index 56780929..880f3617 100644 --- a/app/crowdb-web/src/main.rs +++ b/app/crowdb-web/src/main.rs @@ -68,6 +68,9 @@ struct Args { #[tokio::main] async fn main() -> Result<(), Box> { let args = Args::parse(); + if args.config.is_none() && !args.test_mode { + return Err("crowdb-web requires a versioned --config outside test mode".into()); + } let process_config = args.config.as_deref().map(WebProcessConfig::load).transpose()?; if args.registry.is_some() && process_config @@ -89,22 +92,7 @@ async fn main() -> Result<(), Box> { let addr: SocketAddr = format!("{bind}:{port}").parse()?; info!(%addr, "crowdb-web starting"); - // Load the persisted registry; absence yields an empty default. - // Mutating handlers (rack/node/server CRUD) write back to this path. - let path = if args.test_mode || process_config.is_some() { - None - } else { - crowdb_console_shared::TomlFileEngine::default_path() - }; - let cfg = match path.as_ref() { - Some(p) => { - let engine = crowdb_console_shared::TomlFileEngine::new(p.clone()); - crowdb_console_shared::ConsoleConfig::load_with_engine(&engine)? - } - None => crowdb_console_shared::ConsoleConfig::default(), - }; - let server_count = cfg.servers.len(); - let mut state = crowdb_web::AppState::with_config(cfg, path).with_test_mode(args.test_mode); + let mut state = crowdb_web::AppState::default().with_test_mode(args.test_mode); if let Some(config) = process_config { state = state.with_process_config(&config); state = state.with_management_token(std::env::var("CROWDB_ICEBERG_MANAGE_TOKEN")?)?; @@ -115,7 +103,7 @@ async fn main() -> Result<(), Box> { info!(started, "reconciled configured service launches"); } tracing::info!( - servers = server_count, + servers = 0, launches = launch_registry .as_ref() .map_or(0, |registry| registry.launches.len()), diff --git a/app/crowdb-web/src/managed_hardware.rs b/app/crowdb-web/src/managed_hardware.rs index ca88d926..6fbe6c2b 100644 --- a/app/crowdb-web/src/managed_hardware.rs +++ b/app/crowdb-web/src/managed_hardware.rs @@ -13,6 +13,7 @@ use serde::Deserialize; use crate::error::ErrorBody; use crate::managed_logical::api_error; use crate::state::AppState; +use crowdb_console_shared::error::Error as ConsoleError; type ApiError = (StatusCode, Json); @@ -43,6 +44,25 @@ pub(crate) async fn list_racks(State(state): State) -> Result, + Path(id): Path, +) -> Result, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_racks_from_group0(&ctx) + .await + .map_err(api_error)? + .into_iter() + .find(|rack| rack.id == id) + .map(Json) + .ok_or_else(|| { + api_error(ConsoleError::NotFound { + kind: "rack".into(), + id: id.to_string(), + }) + }) +} + pub(crate) async fn add_rack( State(state): State, Json(body): Json, @@ -76,6 +96,36 @@ pub(crate) async fn list_nodes( .map_err(api_error) } +pub(crate) async fn list_rack_nodes( + State(state): State, + Path(rack_id): Path, +) -> Result>, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_nodes_from_group0(&ctx, Some(rack_id)) + .await + .map(Json) + .map_err(api_error) +} + +pub(crate) async fn get_node( + State(state): State, + Path(id): Path, +) -> Result, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_nodes_from_group0(&ctx, None) + .await + .map_err(api_error)? + .into_iter() + .find(|node| node.id == id) + .map(Json) + .ok_or_else(|| { + api_error(ConsoleError::NotFound { + kind: "node".into(), + id: id.to_string(), + }) + }) +} + pub(crate) async fn add_node( State(state): State, Json(node): Json, @@ -117,6 +167,25 @@ pub(crate) async fn list_disk_groups( .map_err(api_error) } +pub(crate) async fn get_disk_group( + State(state): State, + Path((node_id, dg_id)): Path<(u64, u64)>, +) -> Result, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_disk_groups_from_group0(&ctx, node_id) + .await + .map_err(api_error)? + .into_iter() + .find(|group| group.id == dg_id) + .map(Json) + .ok_or_else(|| { + api_error(ConsoleError::NotFound { + kind: "disk_group".into(), + id: dg_id.to_string(), + }) + }) +} + pub(crate) async fn add_disk_group( State(state): State, Path(node_id): Path, @@ -151,6 +220,25 @@ pub(crate) async fn list_disks( .map_err(api_error) } +pub(crate) async fn get_disk( + State(state): State, + Path((node_id, dg_id, disk_id)): Path<(u64, u64, String)>, +) -> Result, ApiError> { + let ctx = state.op_context().await.map_err(api_error)?; + hardware::list_disks_from_group0(&ctx, node_id, dg_id) + .await + .map_err(api_error)? + .into_iter() + .find(|disk| disk.disk_id == disk_id) + .map(Json) + .ok_or_else(|| { + api_error(ConsoleError::NotFound { + kind: "disk".into(), + id: disk_id, + }) + }) +} + pub(crate) async fn add_disk( State(state): State, Path((node_id, dg_id)): Path<(u64, u64)>, diff --git a/app/crowdb-web/tests/bare_metal_authority_test.rs b/app/crowdb-web/tests/bare_metal_authority_test.rs index 70110102..8f1d7b71 100644 --- a/app/crowdb-web/tests/bare_metal_authority_test.rs +++ b/app/crowdb-web/tests/bare_metal_authority_test.rs @@ -198,6 +198,24 @@ async fn bare_metal_hardware_routes_share_confirmed_group_zero_state() { assert_eq!(nodes[0]["host"], "node-nine.example"); assert_eq!(nodes[0]["ssh_credential_ref"], "ops-key"); assert!(nodes[0].get("ssh_key").is_none()); + assert_eq!( + hardware_request(&first, axum::http::Method::GET, "/api/racks/8", None, false) + .await + .1["name"], + "rack-eight" + ); + assert_eq!( + hardware_request(&first, axum::http::Method::GET, "/api/racks/8/nodes", None, false) + .await + .1[0]["id"], + 9 + ); + assert_eq!( + hardware_request(&first, axum::http::Method::GET, "/api/nodes/9", None, false) + .await + .1["host"], + "node-nine.example" + ); assert_eq!( hardware_request(&second, axum::http::Method::DELETE, "/api/racks/8", None, true) .await @@ -248,6 +266,18 @@ async fn verify_storage_hardware(first: &axum::Router, second: &axum::Router) { "hot" ); let disks = "/api/nodes/9/disk-groups/4/disks"; + assert_eq!( + hardware_request( + second, + axum::http::Method::GET, + "/api/nodes/9/disk-groups/4", + None, + false + ) + .await + .1["name"], + "hot" + ); let disk = serde_json::json!({ "disk_id": "0000000000000000-0000000000000009", "disk_type": "Ssd", "capacity_bytes": 4096, "zone_size_bytes": 4096, "unit_size_bytes": 4096, @@ -268,6 +298,22 @@ async fn verify_storage_hardware(first: &axum::Router, second: &axum::Router) { .len(), 1 ); + assert_eq!( + hardware_request( + first, + axum::http::Method::GET, + "/api/nodes/9/disk-groups/4/disks/0000000000000000-0000000000000009", + None, + false + ) + .await + .1["device_path"], + "/dev/test" + ); + verify_storage_removal(first, second).await; +} + +async fn verify_storage_removal(first: &axum::Router, second: &axum::Router) { assert_eq!( hardware_request( first, diff --git a/app/crowdb-web/tests/managed_mode_test.rs b/app/crowdb-web/tests/managed_mode_test.rs index 74d5799d..3395f045 100644 --- a/app/crowdb-web/tests/managed_mode_test.rs +++ b/app/crowdb-web/tests/managed_mode_test.rs @@ -370,7 +370,16 @@ fn old_mixed_config_is_rejected_before_startup() { } #[test] -fn malformed_bare_metal_registry_fails_before_listener_bind() { +fn production_web_requires_versioned_process_config() { + let output = std::process::Command::new(env!("CARGO_BIN_EXE_crowdb-web")) + .output() + .unwrap(); + assert!(!output.status.success()); + assert!(String::from_utf8_lossy(&output.stderr).contains("requires a versioned --config")); +} + +#[test] +fn legacy_mixed_registry_is_not_loaded_without_versioned_config() { let root = std::env::temp_dir().join(format!( "crowdb-web-malformed-registry-{}-{}", std::process::id(), @@ -390,10 +399,7 @@ fn malformed_bare_metal_registry_fails_before_listener_bind() { std::fs::remove_dir_all(root).unwrap(); assert!(!output.status.success()); let stderr = String::from_utf8_lossy(&output.stderr); - assert!( - stderr.contains("Error: Config(") && stderr.contains("invalid table header"), - "{stderr}" - ); + assert!(stderr.contains("requires a versioned --config"), "{stderr}"); } #[test] diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 0570ba36..15377ee7 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -147,7 +147,11 @@ configuration, documentation and crash-diagnostics tasks below are pending. cover missing, duplicate and expired registrations, recovery, and outage without stale topology. Docker and launch-route regressions, fmt and clippy pass. Logs: `/tmp/crowdb-bare-authority-*.log`. Legacy physical routes and - monitor refresh still remain for the mixed-config removal. + monitor refresh still remain for the mixed-config removal. Production Web + startup now requires a versioned process config, so it never loads the old + mixed file; bare-metal rack/node/disk-group/disk detail and collection + routes read Group 0 directly. The old in-process router and CLI no-registry + paths remain to migrate or remove. - [ ] **Replay-safe bootstrap cutover**: persist bootstrap identity, verify committed records, write only safely missing content, reject conflicts and delete topology intent after verified transfer. Clean/destroy use confirmed From bc06ed8bb7d185e7e2f4b5260cfa99a1c6e88956 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 14:32:55 +0800 Subject: [PATCH 57/74] Resolve relative S3 roots before launching services --- app/crowdb-cli/tests/s3_cli_test.rs | 12 +++++++++++- doc/working/plan-console-authority.md | 4 ++++ lib/crowdb-console-shared/src/ops/s3.rs | 4 +++- 3 files changed, 18 insertions(+), 2 deletions(-) diff --git a/app/crowdb-cli/tests/s3_cli_test.rs b/app/crowdb-cli/tests/s3_cli_test.rs index 29bfe2cb..446405d3 100644 --- a/app/crowdb-cli/tests/s3_cli_test.rs +++ b/app/crowdb-cli/tests/s3_cli_test.rs @@ -17,9 +17,14 @@ fn cli() -> Command { fn interrupted_s3_launch_resumes_confirmed_group_zero_without_topology_file() { let directory = TestDir::new("s3-bootstrap-replay-cli").expect("test directory"); let root = directory.path(); + let workspace = crowdb_test_harness::test_dirs::workspace_root(); + let relative_root = root + .strip_prefix(&workspace) + .expect("test root is below workspace"); let failed = cli() + .current_dir(&workspace) .args(["s3", "cluster", "start", "--root"]) - .arg(root) + .arg(relative_root) .env("CROWDB_CHUNK_KV_SERVER_BIN", "/bin/false") .output() .expect("run interrupted launch"); @@ -75,6 +80,11 @@ fn write_cluster_record(root: &Path, endpoint: &str) { serde_json::to_vec_pretty(&record).expect("record json"), ) .expect("write record"); + std::fs::write( + root.join("s3-local-state.toml"), + "version = 1\ngroup0_seeds = ['http://127.0.0.1:10000']\n[[service]]\nid = 'kv-1'\nurl = 'http://127.0.0.1:10000'\nnode_id = 1\n", + ) + .expect("write launch-only state"); } fn mock_http_once(response_content_type: &str, response_body: &[u8]) -> (String, mpsc::Receiver>) { diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 15377ee7..ca62af0b 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -188,6 +188,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. The CLI integration test reproduces this interruption and recovery. Partial storage-service launch sets still fail closed and need completion or an explicit operator recovery path; mixed CLI/Web config remains to remove. + S3 now canonicalizes a relative root before creating child launch paths; + the interrupted CLI test covers a relative root. The S3 CLI mock fixture + supplies the required launch-only state; its three previously failing cases + now pass. The complete Console gate needs rerun. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with diff --git a/lib/crowdb-console-shared/src/ops/s3.rs b/lib/crowdb-console-shared/src/ops/s3.rs index 1160bc1e..9df30e02 100644 --- a/lib/crowdb-console-shared/src/ops/s3.rs +++ b/lib/crowdb-console-shared/src/ops/s3.rs @@ -108,6 +108,9 @@ async fn start_with_profile( ) -> Result { archive_incomplete_attempt(data_dir)?; validate_location(data_dir)?; + std::fs::create_dir_all(data_dir)?; + let canonical_root = std::fs::canonicalize(data_dir)?; + let data_dir = canonical_root.as_path(); let marker_path = data_dir.join(MARKER_FILE); if marker_path.exists() { let (_, record) = load(data_dir)?; @@ -129,7 +132,6 @@ async fn start_with_profile( return restart(data_dir).await; } - std::fs::create_dir_all(data_dir)?; if storage_profile == StorageProfile::Persistent { RuntimeNamespace::persistent(data_dir, NAMESPACE_ID).map_err(namespace_error)?; } From 67a20888db30dc8b3559360c36a24f8480338219 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 14:40:18 +0800 Subject: [PATCH 58/74] Resolve clean targets from confirmed replicas --- doc/working/plan-console-authority.md | 6 +++++- lib/crowdb-console-shared/src/ops/cluster.rs | 21 ++++++++++++------- .../tests/ops_cluster_test.rs | 20 +++++++++++++++--- 3 files changed, 36 insertions(+), 11 deletions(-) diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index ca62af0b..d0e3cc1c 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -191,7 +191,11 @@ configuration, documentation and crash-diagnostics tasks below are pending. S3 now canonicalizes a relative root before creating child launch paths; the interrupted CLI test covers a relative root. The S3 CLI mock fixture supplies the required launch-only state; its three previously failing cases - now pass. The complete Console gate needs rerun. + now pass. The complete Console gate passes after the S3 fixture update. + `cluster clean` now derives its target nodes from confirmed Group 0 replica + membership and resolves each live management registration; local launch + entries cannot justify a wipe. A real Group 0 regression rejects a group + absent from authority even when the console has a local server entry. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with diff --git a/lib/crowdb-console-shared/src/ops/cluster.rs b/lib/crowdb-console-shared/src/ops/cluster.rs index 8f5d6e51..107035d0 100644 --- a/lib/crowdb-console-shared/src/ops/cluster.rs +++ b/lib/crowdb-console-shared/src/ops/cluster.rs @@ -179,13 +179,20 @@ pub struct CleanResult { /// # Errors /// Returns an error if no servers are configured. pub async fn clean(ctx: &OpContext, store_id: u64, group_id: u64) -> Result { - let cfg = ctx.config().clone(); - let mut mgmt_urls: Vec = cfg - .servers - .iter() - .filter(|server| server.service_type == ServiceType::Kv) - .map(|server| server.url.clone()) - .collect(); + // The confirmed replica set determines which nodes must be wiped. A local + // launch registry may contain stopped or unrelated processes, and cannot + // substitute for Group 0 membership after bootstrap. + let replicas = ctx.sysmd().list_replicas_in_group(store_id, group_id).await?; + if replicas.is_empty() { + return Err(Error::NotFound { + kind: "group replicas".into(), + id: format!("{store_id}/{group_id}"), + }); + } + let mut mgmt_urls = Vec::with_capacity(replicas.len()); + for replica in replicas { + mgmt_urls.push(ctx.live_node_mgmt_url(replica.node_id).await?); + } mgmt_urls.sort(); mgmt_urls.dedup(); if mgmt_urls.is_empty() { diff --git a/lib/crowdb-console-shared/tests/ops_cluster_test.rs b/lib/crowdb-console-shared/tests/ops_cluster_test.rs index fd0238cc..774d5538 100644 --- a/lib/crowdb-console-shared/tests/ops_cluster_test.rs +++ b/lib/crowdb-console-shared/tests/ops_cluster_test.rs @@ -1,13 +1,15 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Tests for [`ops::cluster`] validation paths. The `init` / `reset` / -//! `clean` logic requires a running cluster and is covered by E2E tests -//! in Phase 4; here we verify the guard clauses. +//! Tests for [`ops::cluster`] validation and authority boundaries. use crowdb_console_shared::config::ConsoleConfig; use crowdb_console_shared::error::Error; use crowdb_console_shared::ops::{self, OpContext}; +use crowdb_test_harness::cluster::KvCluster; + +#[path = "common/bootstrap_authority.rs"] +mod bootstrap_authority; fn ctx() -> OpContext { OpContext::new("127.0.0.1:1".into(), vec![], ConsoleConfig::default()) @@ -33,3 +35,15 @@ async fn init_dedup_nodes() { Error::NodeUnreachable { .. } | Error::NotFound { .. } )); } + +#[tokio::test] +async fn clean_requires_confirmed_group_replicas() { + let cluster = KvCluster::start().await; + let bootstrap = bootstrap_authority::context(&cluster).await; + ops::cluster::init(&bootstrap, &[1]).await.unwrap(); + + // The bootstrap context still has a local server entry, but no local + // entry may justify wiping a group absent from Group 0. + let err = ops::cluster::clean(&bootstrap, 17, 2).await.unwrap_err(); + assert!(matches!(err, Error::NotFound { .. }), "{err:?}"); +} From 0a9cea899e2e24cfebb8a08854cb5227cb6d0e68 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 14:45:26 +0800 Subject: [PATCH 59/74] Replay partial S3 storage launches --- app/crowdb-cli/tests/s3_cli_test.rs | 39 ++++++++++++++++ doc/working/plan-console-authority.md | 7 ++- lib/crowdb-console-shared/src/ops/s3.rs | 59 +++++++++++++++++++++++-- 3 files changed, 99 insertions(+), 6 deletions(-) diff --git a/app/crowdb-cli/tests/s3_cli_test.rs b/app/crowdb-cli/tests/s3_cli_test.rs index 446405d3..1ee9da09 100644 --- a/app/crowdb-cli/tests/s3_cli_test.rs +++ b/app/crowdb-cli/tests/s3_cli_test.rs @@ -57,6 +57,45 @@ fn interrupted_s3_launch_resumes_confirmed_group_zero_without_topology_file() { ); } +#[test] +#[ignore = "starts the complete local storage stack twice"] +fn interrupted_s3_storage_launch_replays_without_local_topology() { + let directory = TestDir::new("s3-storage-replay-cli").expect("test directory"); + let root = directory.path(); + let failed = cli() + .args(["s3", "cluster", "start", "--root"]) + .arg(root) + .env("CROWDB_DISKIO_BIN", "/bin/false") + .output() + .expect("run interrupted storage launch"); + assert!(!failed.status.success(), "failure injection must stop launch"); + let local = std::fs::read_to_string(root.join("s3-local-state.toml")).unwrap(); + assert!(local.contains("diskdb-1"), "{local}"); + assert!(!local.contains("[[rack]]")); + let restarted = cli() + .args(["s3", "cluster", "start", "--root"]) + .arg(root) + .env_remove("CROWDB_DISKIO_BIN") + .output() + .expect("resume interrupted storage launch"); + assert!( + restarted.status.success(), + "{}", + String::from_utf8_lossy(&restarted.stderr) + ); + assert!(!root.join("s3-mini-cluster.initializing.json").exists()); + let deleted = cli() + .args(["s3", "cluster", "delete", "--root"]) + .arg(root) + .output() + .expect("delete test cluster"); + assert!( + deleted.status.success(), + "{}", + String::from_utf8_lossy(&deleted.stderr) + ); +} + fn tempdir(tag: &str) -> PathBuf { let nonce = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index d0e3cc1c..776c58b8 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -186,8 +186,11 @@ configuration, documentation and crash-diagnostics tasks below are pending. committed cluster. A real failure injected at Chunk KV startup recovers on the next CLI invocation, with all 15 services ready and no topology file. The CLI integration test reproduces this interruption and recovery. Partial - storage-service launch sets still fail closed and need completion or an - explicit operator recovery path; mixed CLI/Web config remains to remove. + storage-service launch sets now clear only the incomplete local process + entries after checking a still-live process's work directory, then replay + provisioning from confirmed Group 0 metadata. A DiskIO startup failure + after DiskDB launch recovers on the next CLI invocation with no local + topology copy. Mixed CLI/Web config remains to remove. S3 now canonicalizes a relative root before creating child launch paths; the interrupted CLI test covers a relative root. The S3 CLI mock fixture supplies the required launch-only state; its three previously failing cases diff --git a/lib/crowdb-console-shared/src/ops/s3.rs b/lib/crowdb-console-shared/src/ops/s3.rs index 9df30e02..424727dc 100644 --- a/lib/crowdb-console-shared/src/ops/s3.rs +++ b/lib/crowdb-console-shared/src/ops/s3.rs @@ -313,10 +313,15 @@ async fn initialize_after_kv( } } } else { - return Err(Error::Conflict { - kind: "partial S3 storage launch state".into(), - id: format!("{storage_services} of 9 services"), - }); + clear_partial_storage_launches(ctx, data_dir)?; + match storage_profile { + StorageProfile::Persistent => { + cluster::local_deploy_combined_file_backed_after_kv(ctx, data_dir, disk, chunk).await?; + } + StorageProfile::Memory => { + cluster::local_deploy_combined_after_kv(ctx, data_dir, disk, chunk, "mem").await?; + } + } } let seeds = management_seeds(&ctx.config()); local_state::save(data_dir, &ctx.config())?; @@ -335,6 +340,52 @@ async fn initialize_after_kv( }) } +fn clear_partial_storage_launches(ctx: &OpContext, data_dir: &Path) -> Result<()> { + let partial = ctx + .config() + .servers + .iter() + .filter(|server| { + matches!( + server.service_type, + ServiceType::Diskdb | ServiceType::Diskio | ServiceType::Chunkdb + ) + }) + .cloned() + .collect::>(); + for server in &partial { + if let Some(pid) = server.pid.filter(|pid| lifecycle::process_is_alive(*pid)) { + let launch = ctx + .config() + .local_launches + .get(&server.id) + .cloned() + .ok_or_else(|| Error::Config(format!("{} has no launch specification", server.id)))?; + let actual_cwd = std::fs::read_link(format!("/proc/{pid}/cwd"))?; + if actual_cwd != Path::new(&launch.workdir) { + return Err(Error::Conflict { + kind: "S3 process identity".into(), + id: server.id.clone(), + }); + } + lifecycle::stop_pid_with_timeout(pid, Duration::from_secs(5))?; + if lifecycle::process_is_alive(pid) { + return Err(Error::Conflict { + kind: "S3 process still running".into(), + id: server.id.clone(), + }); + } + } + } + let ids = partial.into_iter().map(|server| server.id).collect::>(); + let mut config = ctx.config_mut(); + config.servers.retain(|server| !ids.contains(&server.id)); + for id in ids { + config.local_launches.remove(&id); + } + local_state::save(data_dir, &config) +} + async fn resume_incomplete( data_dir: &Path, disk: &LocalDiskdbDeployConfig, From b9926884de99f67a8e6d98cb8d4050052d38b8e0 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 14:49:00 +0800 Subject: [PATCH 60/74] Read CLI hardware from Group 0 in every mode --- app/crowdb-cli/src/commands.rs | 19 ++++ .../src/commands/cluster/hardware.rs | 103 +++++------------- app/crowdb-cli/tests/common/direct.rs | 6 + app/crowdb-cli/tests/direct_cli_test.rs | 10 +- doc/working/plan-console-authority.md | 4 + 5 files changed, 63 insertions(+), 79 deletions(-) diff --git a/app/crowdb-cli/src/commands.rs b/app/crowdb-cli/src/commands.rs index adc41892..129e806b 100644 --- a/app/crowdb-cli/src/commands.rs +++ b/app/crowdb-cli/src/commands.rs @@ -72,6 +72,25 @@ pub(crate) fn op_context(cli: &Cli) -> Result { Ok(OpContext::new(effective_g0, seeds, config)) } +/// Build a Group 0 context for hardware operations without reading the old +/// local console state. The CLI endpoint is a discovery seed, while the +/// optional launch registry is validated only as local process policy. +pub(crate) fn authority_context(cli: &Cli) -> Result { + if let Some(path) = &cli.registry { + crowdb_console_shared::config::web::LaunchRegistry::load(path).map_err(|error| { + eprintln!("error: load launch registry: {error}"); + ExitCode::from(2) + })?; + } + let mgmt_url = format!("http://{}:{}", cli.system_ip, cli.system_port); + let group0_endpoint = format!("{}:{}", cli.system_ip, cli.system_port); + Ok(OpContext::new( + group0_endpoint, + vec![mgmt_url], + crowdb_console_shared::ConsoleConfig::default(), + )) +} + /// Load the CLI's internal persisted state from its fixed runtime location. pub(crate) fn load_config(cli: &Cli) -> Result { if let Some(path) = &cli.registry { diff --git a/app/crowdb-cli/src/commands/cluster/hardware.rs b/app/crowdb-cli/src/commands/cluster/hardware.rs index 644f17ba..ef31420b 100644 --- a/app/crowdb-cli/src/commands/cluster/hardware.rs +++ b/app/crowdb-cli/src/commands/cluster/hardware.rs @@ -10,7 +10,7 @@ use clap::Subcommand; use crowdb_console_shared::config::NodeEntry; use crowdb_protocol::{NodeId, RackId}; -use crate::commands::{commit_config, op_context}; +use crate::commands::authority_context; use crate::Cli; // ── rack ───────────────────────────────────────────────────────── @@ -40,22 +40,13 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli) { Ok(c) => c, Err(c) => return c, }; - let result = if cli.registry.is_some() { - crowdb_console_shared::ops::hardware::add_rack_to_group0(&ctx, rack_id, &name).await - } else { - crowdb_console_shared::ops::hardware::add_rack(&ctx, rack_id, &name).await - }; + let result = crowdb_console_shared::ops::hardware::add_rack_to_group0(&ctx, rack_id, &name).await; match result { Ok(entry) => { - if cli.registry.is_none() { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - } println!("added rack {}", entry.id); ExitCode::SUCCESS } @@ -73,22 +64,13 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli) { Ok(c) => c, Err(c) => return c, }; - let result = if cli.registry.is_some() { - crowdb_console_shared::ops::hardware::remove_rack_from_group0(&ctx, rack_id).await - } else { - crowdb_console_shared::ops::hardware::remove_rack(&ctx, rack_id).await - }; + let result = crowdb_console_shared::ops::hardware::remove_rack_from_group0(&ctx, rack_id).await; match result { Ok(()) => { - if cli.registry.is_none() { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - } println!("removed rack {id}"); ExitCode::SUCCESS } @@ -99,20 +81,16 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { } } RackVerb::List => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli) { Ok(c) => c, Err(c) => return c, }; - let racks = if cli.registry.is_some() { - match crowdb_console_shared::ops::hardware::list_racks_from_group0(&ctx).await { - Ok(racks) => racks, - Err(error) => { - eprintln!("error: list racks: {error}"); - return ExitCode::from(2); - } + let racks = match crowdb_console_shared::ops::hardware::list_racks_from_group0(&ctx).await { + Ok(racks) => racks, + Err(error) => { + eprintln!("error: list racks: {error}"); + return ExitCode::from(2); } - } else { - crowdb_console_shared::ops::hardware::list_racks(&ctx) }; if racks.is_empty() { println!("(no racks)"); @@ -171,8 +149,8 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { ssh_key, ssh_credential_ref, } => { - if cli.registry.is_some() && ssh_key.is_some() { - eprintln!("error: --ssh-key is local secret material; use --ssh-credential-ref with a launch registry"); + if ssh_key.is_some() { + eprintln!("error: --ssh-key is local secret material; use --ssh-credential-ref"); return ExitCode::from(1); } let node_id: NodeId = match id.parse() { @@ -199,22 +177,13 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { ssh_password: None, ssh_credential_ref, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli) { Ok(c) => c, Err(c) => return c, }; - let result = if cli.registry.is_some() { - crowdb_console_shared::ops::hardware::add_node_to_group0(&ctx, entry.clone()).await - } else { - crowdb_console_shared::ops::hardware::add_node(&ctx, entry.clone()).await - }; + let result = crowdb_console_shared::ops::hardware::add_node_to_group0(&ctx, entry.clone()).await; match result { Ok(e) => { - if cli.registry.is_none() { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - } println!("added node {} (rack {})", e.id, e.rack_id); ExitCode::SUCCESS } @@ -232,22 +201,13 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli) { Ok(c) => c, Err(c) => return c, }; - let result = if cli.registry.is_some() { - crowdb_console_shared::ops::hardware::remove_node_from_group0(&ctx, node_id).await - } else { - crowdb_console_shared::ops::hardware::remove_node(&ctx, node_id).await - }; + let result = crowdb_console_shared::ops::hardware::remove_node_from_group0(&ctx, node_id).await; match result { Ok(()) => { - if cli.registry.is_none() { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - } println!("removed node {id}"); ExitCode::SUCCESS } @@ -258,20 +218,16 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { } } NodeVerb::List => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli) { Ok(c) => c, Err(c) => return c, }; - let nodes = if cli.registry.is_some() { - match crowdb_console_shared::ops::hardware::list_nodes_from_group0(&ctx, None).await { - Ok(nodes) => nodes, - Err(error) => { - eprintln!("error: list nodes: {error}"); - return ExitCode::from(2); - } + let nodes = match crowdb_console_shared::ops::hardware::list_nodes_from_group0(&ctx, None).await { + Ok(nodes) => nodes, + Err(error) => { + eprintln!("error: list nodes: {error}"); + return ExitCode::from(2); } - } else { - crowdb_console_shared::ops::hardware::list_nodes(&ctx, None) }; print_node_table(&nodes) } @@ -283,11 +239,11 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli) { Ok(c) => c, Err(c) => return c, }; - let nodes = if cli.registry.is_some() { + let nodes = match crowdb_console_shared::ops::hardware::list_nodes_from_group0(&ctx, Some(rack_id)).await { Ok(nodes) => nodes, @@ -295,10 +251,7 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { eprintln!("error: list nodes: {error}"); return ExitCode::from(2); } - } - } else { - crowdb_console_shared::ops::hardware::list_nodes(&ctx, Some(rack_id)) - }; + }; print_node_table(&nodes) } } @@ -342,7 +295,7 @@ pub enum DiskGroupVerb { pub async fn run_disk_group_verb(cli: &Cli, verb: DiskGroupVerb) -> ExitCode { use crowdb_console_shared::ops::hardware; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli) { Ok(ctx) => ctx, Err(code) => return code, }; @@ -455,7 +408,7 @@ pub enum DiskVerb { pub async fn run_disk_verb(cli: &Cli, verb: DiskVerb) -> ExitCode { use crowdb_console_shared::ops::hardware::{self, AddDiskInput}; use crowdb_protocol::DiskIdExt; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli) { Ok(ctx) => ctx, Err(code) => return code, }; diff --git a/app/crowdb-cli/tests/common/direct.rs b/app/crowdb-cli/tests/common/direct.rs index bb7e41d1..b4a69f88 100644 --- a/app/crowdb-cli/tests/common/direct.rs +++ b/app/crowdb-cli/tests/common/direct.rs @@ -179,6 +179,12 @@ pub async fn spawn_group0() -> Option { ); tokio::time::sleep(Duration::from_millis(100)).await; } + crowdb_console_shared::ops::hardware::add_rack_to_group0(&context, 1, "rack-1") + .await + .expect("publish rack to Group 0"); + crowdb_console_shared::ops::hardware::add_node_to_group0(&context, local_node(1, 1)) + .await + .expect("publish node to Group 0"); Some(Group0 { pid: deployed.pid, diff --git a/app/crowdb-cli/tests/direct_cli_test.rs b/app/crowdb-cli/tests/direct_cli_test.rs index 443942cf..1feb88f7 100644 --- a/app/crowdb-cli/tests/direct_cli_test.rs +++ b/app/crowdb-cli/tests/direct_cli_test.rs @@ -33,7 +33,7 @@ async fn cluster_status_via_direct_group0() { } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn cluster_rack_list_via_direct_config() { +async fn cluster_rack_list_ignores_legacy_local_state() { let Some(g0) = spawn_group0().await else { eprintln!("skipping: crowdb-kv-server binary not built"); return; @@ -44,14 +44,15 @@ async fn cluster_rack_list_via_direct_config() { return; } - // `cluster rack list` should list rack 1 (from the config). + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); + // `cluster rack list` reads rack 1 from Group 0 despite the old file. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "rack", "list"]); assert_eq!(code, 0, "cluster rack list stderr={stderr}"); assert!(stdout.contains('1'), "cluster rack list stdout={stdout}"); } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn cluster_node_list_via_direct_config() { +async fn cluster_node_list_ignores_legacy_local_state() { let Some(g0) = spawn_group0().await else { eprintln!("skipping: crowdb-kv-server binary not built"); return; @@ -62,7 +63,8 @@ async fn cluster_node_list_via_direct_config() { return; } - // `cluster node list` should list node 1 (from the config). + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); + // `cluster node list` reads node 1 from Group 0 despite the old file. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "node", "list"]); assert_eq!(code, 0, "cluster node list stderr={stderr}"); assert!(stdout.contains('1'), "cluster node list stdout={stdout}"); diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 776c58b8..729982dd 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -138,6 +138,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. Web; tests cover two consoles, conflicts, occupied deletion, private Web writes and a lost committed write response. Legacy Web routes and CLI mixed-config paths still need removal. + Every CLI rack/node/disk-group/disk command now builds a Group 0-only + context, including when `--registry` is omitted. A malformed legacy file + cannot alter or block rack/node reads. The complete CLI suite passes; + remaining mixed CLI operations are outside hardware. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. From 12487af65ac009c129e6e675a02475d25155ce08 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 14:53:24 +0800 Subject: [PATCH 61/74] Remove local topology from CLI logical reads --- app/crowdb-cli/src/commands.rs | 13 +++++++++++-- app/crowdb-cli/src/commands/cluster.rs | 6 +++--- app/crowdb-cli/src/commands/kv/logical.rs | 18 +++++++++--------- app/crowdb-cli/tests/cluster_cli_test.rs | 1 + app/crowdb-cli/tests/mgmt_cli_test.rs | 1 + doc/working/plan-console-authority.md | 5 +++++ lib/crowdb-console-shared/src/ops/cluster.rs | 3 ++- 7 files changed, 32 insertions(+), 15 deletions(-) diff --git a/app/crowdb-cli/src/commands.rs b/app/crowdb-cli/src/commands.rs index 129e806b..b578eca1 100644 --- a/app/crowdb-cli/src/commands.rs +++ b/app/crowdb-cli/src/commands.rs @@ -75,7 +75,7 @@ pub(crate) fn op_context(cli: &Cli) -> Result { /// Build a Group 0 context for hardware operations without reading the old /// local console state. The CLI endpoint is a discovery seed, while the /// optional launch registry is validated only as local process policy. -pub(crate) fn authority_context(cli: &Cli) -> Result { +pub(crate) async fn authority_context(cli: &Cli) -> Result { if let Some(path) = &cli.registry { crowdb_console_shared::config::web::LaunchRegistry::load(path).map_err(|error| { eprintln!("error: load launch registry: {error}"); @@ -83,7 +83,16 @@ pub(crate) fn authority_context(cli: &Cli) -> Result { })?; } let mgmt_url = format!("http://{}:{}", cli.system_ip, cli.system_port); - let group0_endpoint = format!("{}:{}", cli.system_ip, cli.system_port); + let group0_endpoint = match crowdb_console_shared::clients::http::ServerClient::new(&mgmt_url) { + Ok(client) => client + .topology() + .await + .ok() + .and_then(|stores| stores.into_iter().find(|store| store.store_id == 0)) + .and_then(|store| store.listen_addr), + Err(_) => None, + } + .unwrap_or_else(|| format!("{}:{}", cli.system_ip, cli.system_port)); Ok(OpContext::new( group0_endpoint, vec![mgmt_url], diff --git a/app/crowdb-cli/src/commands/cluster.rs b/app/crowdb-cli/src/commands/cluster.rs index d81c4909..39b137d3 100644 --- a/app/crowdb-cli/src/commands/cluster.rs +++ b/app/crowdb-cli/src/commands/cluster.rs @@ -15,7 +15,7 @@ use std::process::ExitCode; use clap::Subcommand; -use crate::commands::{commit_config, config_path, op_context}; +use crate::commands::{authority_context, commit_config, config_path, op_context}; use crate::Cli; #[derive(Subcommand, Debug)] @@ -609,7 +609,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } ClusterVerb::Status => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -641,7 +641,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } ClusterVerb::Topology { node } => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/kv/logical.rs b/app/crowdb-cli/src/commands/kv/logical.rs index e5a83a4c..98b60a42 100644 --- a/app/crowdb-cli/src/commands/kv/logical.rs +++ b/app/crowdb-cli/src/commands/kv/logical.rs @@ -7,7 +7,7 @@ use std::process::ExitCode; use clap::Subcommand; -use crate::commands::op_context; +use crate::commands::authority_context; use crate::Cli; // ── store ──────────────────────────────────────────────────────── @@ -45,7 +45,7 @@ pub async fn run_store_verb(cli: &Cli, verb: StoreVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -75,7 +75,7 @@ pub async fn run_store_verb(cli: &Cli, verb: StoreVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -91,7 +91,7 @@ pub async fn run_store_verb(cli: &Cli, verb: StoreVerb) -> ExitCode { } } StoreVerb::List => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -187,7 +187,7 @@ pub async fn run_group_verb(cli: &Cli, verb: GroupVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -221,7 +221,7 @@ pub async fn run_group_verb(cli: &Cli, verb: GroupVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -244,7 +244,7 @@ pub async fn run_group_verb(cli: &Cli, verb: GroupVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -333,7 +333,7 @@ pub async fn run_replica_verb(cli: &Cli, verb: ReplicaVerb) -> ExitCode { }, None => None, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -376,7 +376,7 @@ pub async fn run_replica_verb(cli: &Cli, verb: ReplicaVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; diff --git a/app/crowdb-cli/tests/cluster_cli_test.rs b/app/crowdb-cli/tests/cluster_cli_test.rs index 317be07e..d9aadb5d 100644 --- a/app/crowdb-cli/tests/cluster_cli_test.rs +++ b/app/crowdb-cli/tests/cluster_cli_test.rs @@ -82,6 +82,7 @@ async fn cluster_status_topology_via_direct_group0() { ); assert_eq!(code, 0, "cluster init stderr={stderr}"); assert!(!g0.config_path.with_extension("bootstrap.toml").exists()); + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); // status — lists stores from group-0 sysdata. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "status"]); diff --git a/app/crowdb-cli/tests/mgmt_cli_test.rs b/app/crowdb-cli/tests/mgmt_cli_test.rs index e4f3eeec..b75a15f7 100644 --- a/app/crowdb-cli/tests/mgmt_cli_test.rs +++ b/app/crowdb-cli/tests/mgmt_cli_test.rs @@ -22,6 +22,7 @@ async fn store_group_replica_round_trip() { eprintln!("skipping: crowdb-cli binary not built ({})", cli.display()); return; } + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); let store_id = "9"; let group_id = "90"; diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 729982dd..cd8716b1 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -142,6 +142,11 @@ configuration, documentation and crash-diagnostics tasks below are pending. context, including when `--registry` is omitted. A malformed legacy file cannot alter or block rack/node reads. The complete CLI suite passes; remaining mixed CLI operations are outside hardware. + CLI logical store/group/replica mutations and reads, plus cluster status + and node topology, now also ignore the old file. Live node registration + locates the topology endpoint; a management read supplies only the initial + RPC connection hint. The logical round-trip and status/topology tests pass + with a deliberately invalid legacy file, and the complete CLI suite passes. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. diff --git a/lib/crowdb-console-shared/src/ops/cluster.rs b/lib/crowdb-console-shared/src/ops/cluster.rs index 107035d0..97b24f61 100644 --- a/lib/crowdb-console-shared/src/ops/cluster.rs +++ b/lib/crowdb-console-shared/src/ops/cluster.rs @@ -50,7 +50,8 @@ pub async fn status(ctx: &OpContext) -> Result Result> { - let client = server_client(ctx, node_id)?; + let url = ctx.live_node_mgmt_url(node_id).await?; + let client = ServerClient::new(&url)?; client.topology().await } From 6b3bd1a8d5d4cb368f69c25ddb81de5cd64a0ae8 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 14:55:14 +0800 Subject: [PATCH 62/74] Remove local state from CLI KV data commands --- app/crowdb-cli/src/commands/kv/data.rs | 16 ++++++++-------- app/crowdb-cli/tests/kv_cli_test.rs | 1 + doc/working/plan-console-authority.md | 2 ++ 3 files changed, 11 insertions(+), 8 deletions(-) diff --git a/app/crowdb-cli/src/commands/kv/data.rs b/app/crowdb-cli/src/commands/kv/data.rs index 876b74bc..4e22f9b1 100644 --- a/app/crowdb-cli/src/commands/kv/data.rs +++ b/app/crowdb-cli/src/commands/kv/data.rs @@ -8,7 +8,7 @@ use std::process::ExitCode; use clap::Subcommand; use crowdb_kv_client::GetOutcome; -use crate::commands::op_context; +use crate::commands::authority_context; use crate::Cli; #[derive(Subcommand, Debug)] @@ -90,7 +90,7 @@ pub async fn run_kv_data_verb(cli: &Cli, verb: KvDataVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -119,7 +119,7 @@ pub async fn run_kv_data_verb(cli: &Cli, verb: KvDataVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -147,7 +147,7 @@ pub async fn run_kv_data_verb(cli: &Cli, verb: KvDataVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -174,7 +174,7 @@ pub async fn run_kv_data_verb(cli: &Cli, verb: KvDataVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -211,7 +211,7 @@ async fn run_snapshot_verb(cli: &Cli, verb: SnapshotVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -231,7 +231,7 @@ async fn run_snapshot_verb(cli: &Cli, verb: SnapshotVerb) -> ExitCode { Ok(ids) => ids, Err(c) => return c, }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -260,7 +260,7 @@ async fn run_snapshot_verb(cli: &Cli, verb: SnapshotVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; diff --git a/app/crowdb-cli/tests/kv_cli_test.rs b/app/crowdb-cli/tests/kv_cli_test.rs index 23730a2e..84ea4bd4 100644 --- a/app/crowdb-cli/tests/kv_cli_test.rs +++ b/app/crowdb-cli/tests/kv_cli_test.rs @@ -22,6 +22,7 @@ async fn kv_put_get_delete_round_trip() { eprintln!("skipping: crowdb-cli binary not built ({})", cli.display()); return; } + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); // put on group 0 (system store). let (code, stdout, stderr) = run( diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index cd8716b1..6aae12fa 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -147,6 +147,8 @@ configuration, documentation and crash-diagnostics tasks below are pending. locates the topology endpoint; a management read supplies only the initial RPC connection hint. The logical round-trip and status/topology tests pass with a deliberately invalid legacy file, and the complete CLI suite passes. + CLI KV data commands also use this context; the put/get/delete/scan + round-trip passes with an invalid legacy file. - [ ] **Authority-only reads**: replace local monitor/config topology and endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or expired registrations remain unavailable. From fa4c0480df186721cfc043c6879dfc6d2444d997 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 15:02:08 +0800 Subject: [PATCH 63/74] Remove Web mixed topology persistence --- .../src/commands/cluster/hardware.rs | 18 +++---- app/crowdb-web/src/mgmt/cluster_init.rs | 2 +- app/crowdb-web/src/state.rs | 49 +++++-------------- app/crowdb-web/tests/cluster_deployer_test.rs | 2 +- .../tests/cluster_restart_incremental_test.rs | 2 +- app/crowdb-web/tests/lifecycle_routes_test.rs | 10 +--- app/crowdb-web/tests/mgmt_routes_test.rs | 6 +-- doc/working/plan-console-authority.md | 6 ++- 8 files changed, 34 insertions(+), 61 deletions(-) diff --git a/app/crowdb-cli/src/commands/cluster/hardware.rs b/app/crowdb-cli/src/commands/cluster/hardware.rs index ef31420b..e9aba57a 100644 --- a/app/crowdb-cli/src/commands/cluster/hardware.rs +++ b/app/crowdb-cli/src/commands/cluster/hardware.rs @@ -40,7 +40,7 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match authority_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -64,7 +64,7 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match authority_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -81,7 +81,7 @@ pub async fn run_rack_verb(cli: &Cli, verb: RackVerb) -> ExitCode { } } RackVerb::List => { - let ctx = match authority_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -177,7 +177,7 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { ssh_password: None, ssh_credential_ref, }; - let ctx = match authority_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -201,7 +201,7 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match authority_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -218,7 +218,7 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { } } NodeVerb::List => { - let ctx = match authority_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -239,7 +239,7 @@ pub async fn run_node_verb(cli: &Cli, verb: NodeVerb) -> ExitCode { return ExitCode::from(1); } }; - let ctx = match authority_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -295,7 +295,7 @@ pub enum DiskGroupVerb { pub async fn run_disk_group_verb(cli: &Cli, verb: DiskGroupVerb) -> ExitCode { use crowdb_console_shared::ops::hardware; - let ctx = match authority_context(cli) { + let ctx = match authority_context(cli).await { Ok(ctx) => ctx, Err(code) => return code, }; @@ -408,7 +408,7 @@ pub enum DiskVerb { pub async fn run_disk_verb(cli: &Cli, verb: DiskVerb) -> ExitCode { use crowdb_console_shared::ops::hardware::{self, AddDiskInput}; use crowdb_protocol::DiskIdExt; - let ctx = match authority_context(cli) { + let ctx = match authority_context(cli).await { Ok(ctx) => ctx, Err(code) => return code, }; diff --git a/app/crowdb-web/src/mgmt/cluster_init.rs b/app/crowdb-web/src/mgmt/cluster_init.rs index 946c2021..353815f7 100644 --- a/app/crowdb-web/src/mgmt/cluster_init.rs +++ b/app/crowdb-web/src/mgmt/cluster_init.rs @@ -41,7 +41,7 @@ pub(crate) async fn http_cluster_init( Json(body): Json, ) -> Result<(StatusCode, Json), (StatusCode, Json)> { let ctx = state.op_context().await.map_err(|e| err_502(format!("{e}")))?; - let summary = if state.config_engine.is_some() || state.web_mode.is_some() { + let summary = if state.web_mode.is_some() { let path = state.runtime_root.join("bootstrap-intent.toml"); if state.web_mode == Some(crowdb_console_shared::config::web::WebMode::BareMetal) { if let Some(source) = &body.bootstrap_file { diff --git a/app/crowdb-web/src/state.rs b/app/crowdb-web/src/state.rs index 16c98abd..e4ff7a43 100644 --- a/app/crowdb-web/src/state.rs +++ b/app/crowdb-web/src/state.rs @@ -10,23 +10,18 @@ use crowdb_console_shared::error::{Error, Result}; use crowdb_console_shared::launch::LaunchRuntime; use crowdb_console_shared::monitor::MonitorCache; use crowdb_console_shared::ops::OpContext; -use crowdb_console_shared::{ - config::{ConsoleConfigEngine, ServerEntry, TomlFileEngine}, - ConsoleConfig, -}; +use crowdb_console_shared::{config::ServerEntry, ConsoleConfig}; /// Shared, mutable console state. /// -/// `config` carries the full `ConsoleConfig` (racks, nodes, servers) -/// behind a `RwLock`; mutations are persisted via `ConsoleConfig::save` -/// to `config_path` when present. +/// `config` is an in-memory context for bootstrap and test-only routes. +/// Production authority reads use Group 0 and live registrations. /// /// `diskdb_client` is lazily initialized on the first `/api/diskdb/*` /// request (the service registry may not be ready at console startup). #[derive(Clone)] pub struct AppState { pub config: Arc>, - pub config_engine: Option>, pub runtime_root: Arc, pub monitor_cache: Arc, pub runtime_pids: Arc>>, @@ -79,32 +74,24 @@ impl AppState { Self::with_config(cfg, None) } - /// Build state from an already-loaded `ConsoleConfig`. `path` is the - /// on-disk location used by mutating handlers to persist changes; - /// pass `None` for in-memory-only state (tests). + /// Build test state from an in-memory `ConsoleConfig`. `path` contributes + /// only the runtime workspace directory; it is never a topology file. #[must_use] pub fn with_config(config: ConsoleConfig, path: Option) -> Self { let runtime_root = path - .as_ref() .and_then(|path| path.parent().map(std::path::Path::to_path_buf)) .unwrap_or_else(|| { crowdb_protocol::port::namespace::runtime_root() .join("persistent") .join("console") }); - let engine = path.map(|path| Arc::new(TomlFileEngine::new(path)) as Arc); - Self::with_config_engine(config, engine, runtime_root) + Self::with_runtime_root(config, runtime_root) } #[must_use] - pub fn with_config_engine( - config: ConsoleConfig, - engine: Option>, - runtime_root: PathBuf, - ) -> Self { + pub fn with_runtime_root(config: ConsoleConfig, runtime_root: PathBuf) -> Self { Self { config: Arc::new(RwLock::new(config)), - config_engine: engine, runtime_root: Arc::new(runtime_root), monitor_cache: Arc::new(MonitorCache::new()), runtime_pids: Arc::new(std::sync::Mutex::new(HashMap::new())), @@ -189,19 +176,12 @@ impl AppState { self } - /// Persist the current config to `config_path`, if one was provided. - /// No-op for in-memory state. - /// - /// # Panics - /// Panics if the `RwLock` is poisoned. + /// The legacy in-memory handlers share state only within this Web process. + /// No topology is written to a local file. /// /// # Errors - /// Returns an error if config saving fails. + /// Reserved for callers that propagate operation errors. pub fn persist(&self) -> crowdb_console_shared::error::Result<()> { - if let Some(engine) = self.config_engine.as_ref() { - let cfg = self.config.read().unwrap(); - cfg.save_with_engine(engine.as_ref())?; - } Ok(()) } @@ -590,12 +570,10 @@ impl AppState { /// /// The write-back is a short critical section with no `await` /// inside the lock — the `OpContext`'s config is cloned in, the - /// old config is replaced, and the lock is released before - /// persistence (which may do file I/O). + /// old config is replaced, and the lock is released. /// /// # Errors - /// Returns an error if the config lock is poisoned or persistence - /// fails. + /// Returns an error if the config lock is poisoned. pub fn commit_op_context(&self, ctx: &OpContext) -> Result<()> { let new_config = ctx.config().clone(); { @@ -680,8 +658,7 @@ mod tests { let root = tempdir("relative-runtime-root"); std::env::set_current_dir(&root).unwrap(); - let state = - AppState::with_config_engine(ConsoleConfig::default(), None, PathBuf::from("example-runtime")); + let state = AppState::with_runtime_root(ConsoleConfig::default(), PathBuf::from("example-runtime")); let workspace = state.prepare_node_workspace("n1").unwrap(); assert!(workspace.is_absolute()); diff --git a/app/crowdb-web/tests/cluster_deployer_test.rs b/app/crowdb-web/tests/cluster_deployer_test.rs index 09cf806f..bcec457a 100644 --- a/app/crowdb-web/tests/cluster_deployer_test.rs +++ b/app/crowdb-web/tests/cluster_deployer_test.rs @@ -26,7 +26,7 @@ fn tempdir(tag: &str) -> PathBuf { async fn spawn_web(cfg_path: PathBuf) -> String { let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.expect("bind"); let addr = listener.local_addr().expect("local_addr"); - let cfg = ConsoleConfig::load(&cfg_path).unwrap_or_default(); + let cfg = ConsoleConfig::default(); let state = AppState::with_config(cfg, Some(cfg_path)); tokio::spawn(async move { let _ = axum::serve(listener, router(state)).await; diff --git a/app/crowdb-web/tests/cluster_restart_incremental_test.rs b/app/crowdb-web/tests/cluster_restart_incremental_test.rs index e16897b5..8a1f1181 100644 --- a/app/crowdb-web/tests/cluster_restart_incremental_test.rs +++ b/app/crowdb-web/tests/cluster_restart_incremental_test.rs @@ -97,7 +97,7 @@ async fn spawn_web_with_path(path: PathBuf) -> SocketAddr { .await .expect("bind"); let addr = listener.local_addr().expect("local_addr"); - let cfg = ConsoleConfig::load(&path).unwrap_or_default(); + let cfg = ConsoleConfig::default(); let state = AppState::with_config(cfg, Some(path.clone())); tokio::spawn(async move { axum::serve(listener, router(state)).await.unwrap(); diff --git a/app/crowdb-web/tests/lifecycle_routes_test.rs b/app/crowdb-web/tests/lifecycle_routes_test.rs index 60aa7266..2e728c5e 100644 --- a/app/crowdb-web/tests/lifecycle_routes_test.rs +++ b/app/crowdb-web/tests/lifecycle_routes_test.rs @@ -37,7 +37,7 @@ async fn spawn_web_with_path(path: std::path::PathBuf) -> SocketAddr { .await .expect("bind"); let addr = listener.local_addr().expect("local_addr"); - let cfg = ConsoleConfig::load(&path).unwrap_or_default(); + let cfg = ConsoleConfig::default(); let state = AppState::with_config(cfg, Some(path)).with_test_mode(true); tokio::spawn(async move { axum::serve(listener, router(state)).await.unwrap(); @@ -220,13 +220,7 @@ async fn rack_node_crud_through_web_routes() { let s = delete_status(&client, &format!("{base}/api/racks/1")).await; assert_eq!(s.as_u16(), 204); - // Persisted file reflects the empty state. - let on_disk = std::fs::read_to_string(&cfg_path).unwrap_or_default(); - assert!( - !on_disk.contains("[[rack]]"), - "rack should be gone from {cfg_path:?}: {on_disk}" - ); - assert!(!on_disk.contains("[[node]]")); + assert!(!cfg_path.exists(), "Web must not write local topology"); } #[tokio::test] diff --git a/app/crowdb-web/tests/mgmt_routes_test.rs b/app/crowdb-web/tests/mgmt_routes_test.rs index 8d65b5b7..483d71c4 100644 --- a/app/crowdb-web/tests/mgmt_routes_test.rs +++ b/app/crowdb-web/tests/mgmt_routes_test.rs @@ -145,7 +145,7 @@ fn config_for_upstream(upstream: &Upstream) -> ConsoleConfig { } #[tokio::test] -async fn persistent_web_bootstrap_clears_verified_intent() { +async fn legacy_in_memory_web_bootstrap_does_not_persist_topology() { let Some(upstream) = spawn_upstream().await else { eprintln!("skipping: crowdb-kv-server binary not built"); return; @@ -159,7 +159,7 @@ async fn persistent_web_bootstrap_clears_verified_intent() { .await .unwrap(); assert_eq!(response.status(), 201, "{}", response.text().await.unwrap()); - assert!(config_path.exists()); + assert!(!config_path.exists()); assert!(!upstream.workspace.join("bootstrap-intent.toml").exists()); } @@ -187,7 +187,7 @@ async fn bare_metal_web_bootstrap_consumes_sealed_topology_input() { log_max_files: 5, request_timeout_ms: Some(500), }; - let state = AppState::with_config_engine(ConsoleConfig::default(), None, upstream.workspace.clone()) + let state = AppState::with_runtime_root(ConsoleConfig::default(), upstream.workspace.clone()) .with_process_config(&config) .with_management_token("bare-metal-bootstrap-test-token-12345".into()) .unwrap(); diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 6aae12fa..ae670dec 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -99,8 +99,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. S3 mini-clusters now persist versioned local process/seed state rather than `console.toml`; restored KV launch nodes are ephemeral process inputs and the bundled Web uses `WebProcessConfig`. The full persistent S3 stop/restart and - range-read E2E passes. Remaining CLI/Web legacy config paths and the shared - parser/writer still need removal. + range-read E2E passes. Web no longer constructs a `TomlFileEngine` or writes + mixed local topology, including its in-process legacy test router; focused + bootstrap, lifecycle and deployer tests pass. The CLI legacy config path and + shared parser/writer still need removal. - [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through shared Group 0 hardware operations; preserve conflicts and uncertain writes without local-first commits. Docker keeps its hardware restrictions. Hardware From 78ddbb96f41f37dae055ed1f0f5bd51e43e1004c Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 15:05:37 +0800 Subject: [PATCH 64/74] Require launch registry for KV server controls --- app/crowdb-cli/src/commands/kv/server.rs | 175 +----------------- .../src/commands/kv/server/registry.rs | 20 +- app/crowdb-cli/tests/direct_cli_test.rs | 8 +- app/crowdb-cli/tests/lifecycle_cli_test.rs | 51 +---- doc/working/plan-console-authority.md | 3 + 5 files changed, 18 insertions(+), 239 deletions(-) diff --git a/app/crowdb-cli/src/commands/kv/server.rs b/app/crowdb-cli/src/commands/kv/server.rs index 260fff1e..652528c1 100644 --- a/app/crowdb-cli/src/commands/kv/server.rs +++ b/app/crowdb-cli/src/commands/kv/server.rs @@ -1,16 +1,12 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! `kv server` command handlers — deploy/restart/stop/delete/list. +//! `kv server` process controls backed by a launch-only registry. -use std::path::PathBuf; use std::process::ExitCode; use clap::Subcommand; -use crowdb_console_shared::lifecycle::DeployRequest; -use crowdb_protocol::NodeId; -use crate::commands::{commit_config, op_context}; use crate::Cli; mod registry; @@ -20,12 +16,6 @@ pub enum KvServerVerb { Deploy { #[arg(short = 'n', long)] node: String, - #[arg(short = 'r', long)] - rest_port: Option, - #[arg(short = 'R', long)] - rpc_port: Option, - #[arg(short = 'b', long)] - binary: Option, }, Start { #[arg(short = 'n', long)] @@ -46,163 +36,10 @@ pub enum KvServerVerb { List, } -#[allow(clippy::too_many_lines)] pub async fn run_kv_server_verb(cli: &Cli, verb: KvServerVerb) -> ExitCode { - if let Some(path) = &cli.registry { - return registry::run(cli, path, verb).await; - } - match verb { - KvServerVerb::Deploy { - node, - rest_port, - rpc_port, - binary, - } => { - let (Some(rest_port), Some(rpc_port)) = (rest_port, rpc_port) else { - eprintln!("error: deploy requires management and RPC ports without --registry"); - return ExitCode::from(2); - }; - let node_id: NodeId = match node.parse() { - Ok(n) => n, - Err(e) => { - eprintln!("error: invalid node id: {e}"); - return ExitCode::from(1); - } - }; - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - let req = DeployRequest { - server_id: node_id.to_string(), - rest_port, - rpc_port, - binary: binary.map(PathBuf::from), - ..Default::default() - }; - match crowdb_console_shared::ops::kv_server::deploy(&ctx, &req, None).await { - Ok(d) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - println!( - "deployed server on node {} -> {} (pid {}, rpc {})", - node_id, d.mgmt_url, d.pid, d.rpc_url - ); - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: deploy on node {node_id}: {e}"); - ExitCode::from(2) - } - } - } - KvServerVerb::Restart { node } | KvServerVerb::Start { node } => { - let node_id: NodeId = match node.parse() { - Ok(n) => n, - Err(e) => { - eprintln!("error: invalid node id: {e}"); - return ExitCode::from(1); - } - }; - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - match crowdb_console_shared::ops::kv_server::restart(&ctx, node_id, None, None, &[]).await { - Ok(d) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - println!( - "restarted server on node {} -> {} (pid {}, rpc {})", - node_id, d.mgmt_url, d.pid, d.rpc_url - ); - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: restart on node {node_id}: {e}"); - ExitCode::from(2) - } - } - } - KvServerVerb::Stop { node } => { - let node_id: NodeId = match node.parse() { - Ok(n) => n, - Err(e) => { - eprintln!("error: invalid node id: {e}"); - return ExitCode::from(1); - } - }; - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - match crowdb_console_shared::ops::kv_server::stop(&ctx, node_id, None).await { - Ok(sent) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - if sent { - println!("sent SIGTERM to server on node {node_id}"); - } else { - println!("server on node {node_id} was already gone"); - } - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: stop on node {node_id}: {e}"); - ExitCode::from(2) - } - } - } - KvServerVerb::Delete { node } => { - let node_id: NodeId = match node.parse() { - Ok(n) => n, - Err(e) => { - eprintln!("error: invalid node id: {e}"); - return ExitCode::from(1); - } - }; - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - match crowdb_console_shared::ops::kv_server::delete(&ctx, node_id).await { - Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - println!("deleted server on node {node_id}"); - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: delete on node {node_id}: {e}"); - ExitCode::from(2) - } - } - } - KvServerVerb::List => { - let ctx = match op_context(cli) { - Ok(c) => c, - Err(c) => return c, - }; - let servers = crowdb_console_shared::ops::kv_server::list(&ctx); - if servers.is_empty() { - println!("(no servers deployed)"); - return ExitCode::SUCCESS; - } - println!("{:<12} {:<26} {:<26} {:<8}", "NODE", "MGMT", "RPC", "PID"); - for s in &servers { - println!( - "{:<12} {:<26} {:<26} {:<8}", - s.node_id.map_or_else(|| "-".into(), |n| n.to_string()), - s.url, - s.rpc_url.as_deref().unwrap_or("-"), - s.pid.map_or_else(|| "-".into(), |p| p.to_string()), - ); - } - ExitCode::SUCCESS - } - } + let Some(path) = &cli.registry else { + eprintln!("error: kv server controls require --registry"); + return ExitCode::from(2); + }; + registry::run(cli, path, verb).await } diff --git a/app/crowdb-cli/src/commands/kv/server/registry.rs b/app/crowdb-cli/src/commands/kv/server/registry.rs index 16e6ed99..30b993a0 100644 --- a/app/crowdb-cli/src/commands/kv/server/registry.rs +++ b/app/crowdb-cli/src/commands/kv/server/registry.rs @@ -64,22 +64,7 @@ async fn execute(cli: &Cli, path: &Path, verb: KvServerVerb) -> Result<()> { id: node.to_string(), })?; match verb { - KvServerVerb::Deploy { - rest_port, - rpc_port, - binary, - .. - } => { - if rest_port.is_some() || rpc_port.is_some() || binary.is_some() { - return Err(Error::Validation { - field: "registry".into(), - message: "binary and listener arguments must come from the launch registry".into(), - }); - } - let identity = runtime.start(&launch).await?; - println!("started kv on node {node} (pid {})", identity.pid); - } - KvServerVerb::Start { .. } => { + KvServerVerb::Deploy { .. } | KvServerVerb::Start { .. } => { let identity = runtime.start(&launch).await?; println!("started kv on node {node} (pid {})", identity.pid); } @@ -92,7 +77,8 @@ async fn execute(cli: &Cli, path: &Path, verb: KvServerVerb) -> Result<()> { println!("stopped kv on node {node}"); } KvServerVerb::Delete { .. } => { - let ctx = crate::commands::op_context(cli) + let ctx = crate::commands::authority_context(cli) + .await .map_err(|_| Error::Config("cannot initialize authority client".into()))?; if ctx .sysmd() diff --git a/app/crowdb-cli/tests/direct_cli_test.rs b/app/crowdb-cli/tests/direct_cli_test.rs index 1feb88f7..23195841 100644 --- a/app/crowdb-cli/tests/direct_cli_test.rs +++ b/app/crowdb-cli/tests/direct_cli_test.rs @@ -71,7 +71,7 @@ async fn cluster_node_list_ignores_legacy_local_state() { } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn kv_server_list_via_direct_config() { +async fn kv_server_controls_require_launch_registry() { let Some(g0) = spawn_group0().await else { eprintln!("skipping: crowdb-kv-server binary not built"); return; @@ -82,8 +82,8 @@ async fn kv_server_list_via_direct_config() { return; } - // `kv server list` should list the server on node 1. + // A legacy topology file cannot supply process launch policy. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["kv", "server", "list"]); - assert_eq!(code, 0, "kv server list stderr={stderr}"); - assert!(stdout.contains('1'), "kv server list stdout={stdout}"); + assert_eq!(code, 2, "stdout={stdout} stderr={stderr}"); + assert!(stderr.contains("--registry"), "stderr={stderr}"); } diff --git a/app/crowdb-cli/tests/lifecycle_cli_test.rs b/app/crowdb-cli/tests/lifecycle_cli_test.rs index 1fdd7559..04b11dd4 100644 --- a/app/crowdb-cli/tests/lifecycle_cli_test.rs +++ b/app/crowdb-cli/tests/lifecycle_cli_test.rs @@ -1,19 +1,16 @@ // Copyright 2026-present Gian -//! CLI e2e for the physical lifecycle verbs: `cluster rack/node` and -//! `kv server` round-trips through a system-group endpoint against a real +//! CLI e2e for `cluster rack/node` round-trips through a system-group endpoint against a real //! `crowdb-kv-server` with the system group //! initialized — no `crowdb-web` intermediary. mod common; -use std::time::Duration; - use common::direct::{crowdb_cli_bin, run, spawn_group0}; #[tokio::test(flavor = "multi_thread", worker_threads = 2)] #[allow(clippy::too_many_lines)] -async fn rack_node_server_lifecycle() { +async fn rack_node_lifecycle() { let Some(g0) = spawn_group0().await else { eprintln!("skipping: crowdb-kv-server binary not built"); return; @@ -67,48 +64,4 @@ async fn rack_node_server_lifecycle() { !stdout.contains("2 2"), "node 2 should be gone: stdout={stdout}" ); - - // kv server list — server on node 1 already exists. - let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["kv", "server", "list"]); - assert_eq!(code, 0, "server list stderr={stderr}"); - assert!(stdout.contains('1'), "stdout={stdout}"); - - // kv server restart — recover node 1 after an out-of-band process exit. - let pid = g0.pid; - tokio::task::spawn_blocking(move || { - let _ = crowdb_console_shared::lifecycle::stop_pid_with_timeout(pid, Duration::from_millis(100)); - }) - .await - .unwrap(); - let (code, stdout, stderr) = run( - &cli, - g0.mgmt_port, - &g0.config_path, - &["kv", "server", "restart", "--node", "1"], - ); - assert_eq!(code, 0, "server restart stdout={stdout}\nstderr={stderr}"); - - let restarted = crowdb_console_shared::ConsoleConfig::load(&g0.config_path).unwrap(); - let restarted_pid = restarted.server_for_node(1).unwrap().pid.unwrap(); - tokio::task::spawn_blocking(move || { - let _ = crowdb_console_shared::lifecycle::stop_pid_with_timeout( - restarted_pid, - Duration::from_millis(100), - ); - }) - .await - .unwrap(); - - // kv server stop — clear the deployment state after an out-of-band exit. - let (code, _, stderr) = run( - &cli, - g0.mgmt_port, - &g0.config_path, - &["kv", "server", "stop", "--node", "1"], - ); - assert_eq!(code, 0, "server stop stderr={stderr}"); - let stopped = crowdb_console_shared::ConsoleConfig::load(&g0.config_path).unwrap(); - assert!(stopped.server_for_node(1).unwrap().pid.is_none()); - - tokio::time::sleep(Duration::from_millis(100)).await; } diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index ae670dec..7926b285 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -103,6 +103,9 @@ configuration, documentation and crash-diagnostics tasks below are pending. mixed local topology, including its in-process legacy test router; focused bootstrap, lifecycle and deployer tests pass. The CLI legacy config path and shared parser/writer still need removal. + `kv server` process commands now require the versioned launch registry; + their former no-registry branch, including persisted PID/topology updates, + is removed. The registry lifecycle regressions and complete CLI suite pass. - [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through shared Group 0 hardware operations; preserve conflicts and uncertain writes without local-first commits. Docker keeps its hardware restrictions. Hardware From 90f7eef3aa2f516a958e0eb35798fc3981450eb1 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 15:08:47 +0800 Subject: [PATCH 65/74] Require sealed bootstrap input for CLI init --- app/crowdb-cli/src/commands/cluster.rs | 58 ++++++++++-------------- app/crowdb-cli/tests/cluster_cli_test.rs | 14 +++--- doc/working/plan-console-authority.md | 3 ++ 3 files changed, 33 insertions(+), 42 deletions(-) diff --git a/app/crowdb-cli/src/commands/cluster.rs b/app/crowdb-cli/src/commands/cluster.rs index 39b137d3..faa22b1f 100644 --- a/app/crowdb-cli/src/commands/cluster.rs +++ b/app/crowdb-cli/src/commands/cluster.rs @@ -15,7 +15,7 @@ use std::process::ExitCode; use clap::Subcommand; -use crate::commands::{authority_context, commit_config, config_path, op_context}; +use crate::commands::{authority_context, commit_config, op_context}; use crate::Cli; #[derive(Subcommand, Debug)] @@ -187,7 +187,11 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { nodes, bootstrap_file, } => { - let ctx = match op_context(cli) { + let Some(registry) = &cli.registry else { + eprintln!("error: cluster init requires --registry and a versioned bootstrap file"); + return ExitCode::from(2); + }; + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; @@ -198,45 +202,31 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { return ExitCode::from(1); } }; - let result = if let Some(registry) = &cli.registry { - let sealed_path = registry.with_extension("bootstrap-intent.toml"); - if let Some(source) = bootstrap_file { - let intent = match crowdb_console_shared::bootstrap_intent::BootstrapIntent::load(&source) - { - Ok(intent) => intent, - Err(error) => { - eprintln!("error: load bootstrap file: {error}"); - return ExitCode::from(2); - } - }; - if intent.members() != node_ids.as_slice() { - eprintln!("error: bootstrap file members differ from --nodes"); - return ExitCode::from(1); - } - if let Err(error) = intent.seal(&sealed_path) { - eprintln!("error: seal bootstrap intent: {error}"); + let sealed_path = registry.with_extension("bootstrap-intent.toml"); + if let Some(source) = bootstrap_file { + let intent = match crowdb_console_shared::bootstrap_intent::BootstrapIntent::load(&source) { + Ok(intent) => intent, + Err(error) => { + eprintln!("error: load bootstrap file: {error}"); return ExitCode::from(2); } - } else if !sealed_path.exists() { - eprintln!("error: --bootstrap-file is required for the first registry-mode init"); + }; + if intent.members() != node_ids.as_slice() { + eprintln!("error: bootstrap file members differ from --nodes"); return ExitCode::from(1); } - crowdb_console_shared::ops::cluster::init_with_intent(&ctx, &node_ids, &sealed_path).await - } else { - if bootstrap_file.is_some() { - eprintln!("error: --bootstrap-file requires --registry"); - return ExitCode::from(1); + if let Err(error) = intent.seal(&sealed_path) { + eprintln!("error: seal bootstrap intent: {error}"); + return ExitCode::from(2); } - let intent_path = config_path().with_extension("bootstrap.toml"); - crowdb_console_shared::ops::cluster::init_with_intent(&ctx, &node_ids, &intent_path).await - }; + } else if !sealed_path.exists() { + eprintln!("error: --bootstrap-file is required for the first init"); + return ExitCode::from(1); + } + let result = + crowdb_console_shared::ops::cluster::init_with_intent(&ctx, &node_ids, &sealed_path).await; match result { Ok(summary) => { - if cli.registry.is_none() { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } - } println!( "cluster initialized: store {}, group {}, {} nodes", summary.store_id, diff --git a/app/crowdb-cli/tests/cluster_cli_test.rs b/app/crowdb-cli/tests/cluster_cli_test.rs index d9aadb5d..10e5f411 100644 --- a/app/crowdb-cli/tests/cluster_cli_test.rs +++ b/app/crowdb-cli/tests/cluster_cli_test.rs @@ -71,23 +71,21 @@ async fn cluster_status_topology_via_direct_group0() { return; } - // cluster init — writes store/group/replica topology into group-0 - // sysdata (idempotent: group 0 already exists from spawn_group0, - // init handles the 409 conflict and still writes topology). + // Bootstrap requires an independent versioned intent and launch registry. let (code, _, stderr) = run( &cli, g0.mgmt_port, &g0.config_path, &["cluster", "init", "-n", "1"], ); - assert_eq!(code, 0, "cluster init stderr={stderr}"); - assert!(!g0.config_path.with_extension("bootstrap.toml").exists()); + assert_eq!(code, 2, "cluster init stderr={stderr}"); + assert!(stderr.contains("--registry"), "stderr={stderr}"); std::fs::write(&g0.config_path, "invalid local topology").unwrap(); // status — lists stores from group-0 sysdata. let (code, stdout, stderr) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "status"]); assert_eq!(code, 0, "status stderr={stderr}"); - assert!(stdout.contains('0'), "stdout={stdout}"); + assert!(stdout.contains("(no stores)"), "stdout={stdout}"); // topology — from a node's /topology endpoint. let (code, stdout, stderr) = run( @@ -99,10 +97,10 @@ async fn cluster_status_topology_via_direct_group0() { assert_eq!(code, 0, "topology stderr={stderr}"); assert!(stdout.contains("store"), "stdout={stdout}"); - // Status always uses the human-readable console table. + // Status reflects the uninitialized authority without a local fallback. let (code, stdout, _) = run(&cli, g0.mgmt_port, &g0.config_path, &["cluster", "status"]); assert_eq!(code, 0); - assert!(stdout.contains("STORE"), "stdout={stdout}"); + assert!(stdout.contains("(no stores)"), "stdout={stdout}"); tokio::time::sleep(Duration::from_millis(50)).await; } diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 7926b285..138f3532 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -106,6 +106,9 @@ configuration, documentation and crash-diagnostics tasks below are pending. `kv server` process commands now require the versioned launch registry; their former no-registry branch, including persisted PID/topology updates, is removed. The registry lifecycle regressions and complete CLI suite pass. + `cluster init` now requires the registry and independent versioned bootstrap + input; it cannot seal or publish from the old mixed file. The registry + bootstrap test passes, and a no-registry CLI test verifies the rejection. - [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through shared Group 0 hardware operations; preserve conflicts and uncertain writes without local-first commits. Docker keeps its hardware restrictions. Hardware From bddf37feddd59da271bfd6eaa075430df73e3556 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 15:12:10 +0800 Subject: [PATCH 66/74] Use authority and launch registry for CLI clean --- app/crowdb-cli/src/commands/cluster.rs | 47 +++++++++++++++++++------ app/crowdb-cli/tests/direct_cli_test.rs | 10 ++++++ doc/working/plan-console-authority.md | 4 +++ 3 files changed, 50 insertions(+), 11 deletions(-) diff --git a/app/crowdb-cli/src/commands/cluster.rs b/app/crowdb-cli/src/commands/cluster.rs index faa22b1f..45fe8107 100644 --- a/app/crowdb-cli/src/commands/cluster.rs +++ b/app/crowdb-cli/src/commands/cluster.rs @@ -567,24 +567,49 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { group, restart_services, } => { - let ctx = match op_context(cli) { + let restart = if restart_services { + let Some(path) = &cli.registry else { + eprintln!("error: --restart-services requires --registry"); + return ExitCode::from(2); + }; + let registry = match crowdb_console_shared::config::web::LaunchRegistry::load(path) { + Ok(registry) => registry, + Err(error) => { + eprintln!("error: load launch registry: {error}"); + return ExitCode::from(2); + } + }; + let runtime = match crowdb_console_shared::launch::LaunchRuntime::for_registry(path) { + Ok(runtime) => runtime, + Err(error) => { + eprintln!("error: launch runtime: {error}"); + return ExitCode::from(2); + } + }; + Some((registry, runtime)) + } else { + None + }; + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; match crowdb_console_shared::ops::cluster::clean(&ctx, store, group).await { Ok(mut result) => { - if restart_services { - match crowdb_console_shared::ops::cluster::restart_storage_services(&ctx).await { - Ok(count) => result.restarted_services = count, - Err(error) => { - let _ = commit_config(cli, &ctx); - eprintln!("error: cluster clean service restart: {error}"); - return ExitCode::from(2); + if let Some((registry, runtime)) = restart { + for kind in ["diskio", "diskdb", "chunkdb"] { + for launch in registry + .launches + .iter() + .filter(|launch| launch.service_id == kind) + { + if let Err(error) = runtime.restart(launch).await { + eprintln!("error: restart {kind} on node {}: {error}", launch.node_id); + return ExitCode::from(2); + } + result.restarted_services += 1; } } - if let Err(code) = commit_config(cli, &ctx) { - return code; - } } println!( "cluster clean: wiped {} nodes, restarted {} services, leader = {}", diff --git a/app/crowdb-cli/tests/direct_cli_test.rs b/app/crowdb-cli/tests/direct_cli_test.rs index 23195841..65b52c71 100644 --- a/app/crowdb-cli/tests/direct_cli_test.rs +++ b/app/crowdb-cli/tests/direct_cli_test.rs @@ -10,6 +10,16 @@ mod common; use common::direct::{crowdb_cli_bin, run, spawn_group0}; +#[test] +fn clean_rejects_service_restart_without_launch_registry() { + let output = std::process::Command::new(crowdb_cli_bin()) + .args(["cluster", "clean", "--restart-services"]) + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(2)); + assert!(String::from_utf8_lossy(&output.stderr).contains("--registry")); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn cluster_status_via_direct_group0() { let Some(g0) = spawn_group0().await else { diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 138f3532..5f95480c 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -109,6 +109,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. `cluster init` now requires the registry and independent versioned bootstrap input; it cannot seal or publish from the old mixed file. The registry bootstrap test passes, and a no-registry CLI test verifies the rejection. + CLI `cluster clean` now obtains replica membership and live endpoints from + Group 0. Optional storage process restarts use the validated launch registry + in DiskIO, DiskDB, ChunkDB order; without one the command rejects the + restart request before wiping data. - [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through shared Group 0 hardware operations; preserve conflicts and uncertain writes without local-first commits. Docker keeps its hardware restrictions. Hardware From 9238fcca094dd2b2f7a08f7a4da5de84519627d6 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 15:20:11 +0800 Subject: [PATCH 67/74] Remove mixed state from CLI local deploy --- app/crowdb-cli/src/commands/cluster.rs | 73 +++++++++++++++++-------- app/crowdb-cli/tests/direct_cli_test.rs | 17 ++++++ doc/working/plan-console-authority.md | 5 ++ 3 files changed, 72 insertions(+), 23 deletions(-) diff --git a/app/crowdb-cli/src/commands/cluster.rs b/app/crowdb-cli/src/commands/cluster.rs index 45fe8107..4bad76ca 100644 --- a/app/crowdb-cli/src/commands/cluster.rs +++ b/app/crowdb-cli/src/commands/cluster.rs @@ -18,6 +18,51 @@ use clap::Subcommand; use crate::commands::{authority_context, commit_config, op_context}; use crate::Cli; +async fn local_deploy_context( + cli: &Cli, + existing_cluster: bool, +) -> Result { + if !existing_cluster { + let mgmt = format!("http://{}:{}", cli.system_ip, cli.system_port); + let rpc_hint = format!("{}:{}", cli.system_ip, cli.system_port); + return Ok(crowdb_console_shared::ops::OpContext::new( + rpc_hint, + vec![mgmt], + crowdb_console_shared::ConsoleConfig::default(), + )); + } + let ctx = authority_context(cli).await?; + { + let racks = crowdb_console_shared::ops::hardware::list_racks_from_group0(&ctx) + .await + .map_err(|error| { + eprintln!("error: read Group 0 racks: {error}"); + ExitCode::from(2) + })?; + let nodes = crowdb_console_shared::ops::hardware::list_nodes_from_group0(&ctx, None) + .await + .map_err(|error| { + eprintln!("error: read Group 0 nodes: {error}"); + ExitCode::from(2) + })?; + let mut servers = Vec::with_capacity(nodes.len()); + for node in &nodes { + let url = ctx.live_node_mgmt_url(node.id).await.map_err(|error| { + eprintln!("error: resolve live KV node {}: {error}", node.id); + ExitCode::from(2) + })?; + let mut server = crowdb_console_shared::config::ServerEntry::new(node.id.to_string(), url); + server.node_id = Some(node.id); + servers.push(server); + } + let mut config = ctx.config_mut(); + config.racks = racks; + config.nodes = nodes; + config.servers = servers; + } + Ok(ctx) +} + #[derive(Subcommand, Debug)] pub enum ClusterVerb { /// Initialize the cluster by bootstrapping group 0 on the listed nodes. @@ -276,7 +321,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { allow_unsafe_ec, } => match service_type.as_str() { "combined" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, false).await { Ok(context) => context, Err(code) => return code, }; @@ -329,9 +374,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(code) = commit_config(cli, &ctx) { - return code; - } println!( "local-deploy combined: {} KV nodes, {} racks, {} DiskDB, {} ChunkDB, {} DiskIO", summary.kv_nodes, @@ -343,16 +385,13 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { ExitCode::SUCCESS } Err(error) => { - if let Err(code) = commit_config(cli, &ctx) { - return code; - } eprintln!("error: local-deploy combined: {error}"); ExitCode::from(2) } } } "kv" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, false).await { Ok(c) => c, Err(c) => return c, }; @@ -386,9 +425,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!( "local-deploy complete: {} nodes (rack {}, nodes [{}]), group 0 bootstrapped", summary.node_count, @@ -409,7 +445,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } "rpc" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, false).await { Ok(c) => c, Err(c) => return c, }; @@ -429,9 +465,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; - } println!( "local-deploy rpc: port={}, pid={}, io_engines={}, io_workers={}, nagle={}", summary.port, summary.pid, summary.io_engines, summary.io_workers, summary.nagle @@ -445,7 +478,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } "diskdb" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, true).await { Ok(c) => c, Err(c) => return c, }; @@ -468,9 +501,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(code) = commit_config(cli, &ctx) { - return code; - } println!( "local-deploy diskdb: {} instances, {} disk-groups, {} disks, data-groups {:?}", summary.instance_count, @@ -487,7 +517,7 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } "chunkdb" => { - let ctx = match op_context(cli) { + let ctx = match local_deploy_context(cli, true).await { Ok(context) => context, Err(code) => return code, }; @@ -508,9 +538,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .await { Ok(summary) => { - if let Err(code) = commit_config(cli, &ctx) { - return code; - } println!("local-deploy chunkdb: {} instances", summary.instance_count); ExitCode::SUCCESS } diff --git a/app/crowdb-cli/tests/direct_cli_test.rs b/app/crowdb-cli/tests/direct_cli_test.rs index 65b52c71..7b4dd8d0 100644 --- a/app/crowdb-cli/tests/direct_cli_test.rs +++ b/app/crowdb-cli/tests/direct_cli_test.rs @@ -97,3 +97,20 @@ async fn kv_server_controls_require_launch_registry() { assert_eq!(code, 2, "stdout={stdout} stderr={stderr}"); assert!(stderr.contains("--registry"), "stderr={stderr}"); } + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn incremental_local_deploy_reads_group_zero_instead_of_legacy_file() { + let Some(g0) = spawn_group0().await else { + return; + }; + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); + let (code, _, stderr) = run( + &crowdb_cli_bin(), + g0.mgmt_port, + &g0.config_path, + &["cluster", "local-deploy", "-t", "diskdb", "--data-groups", "99"], + ); + assert_eq!(code, 2, "stderr={stderr}"); + assert!(stderr.contains("99"), "stderr={stderr}"); + assert!(!stderr.contains("load config"), "stderr={stderr}"); +} diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 5f95480c..2a7ec8f8 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -109,6 +109,11 @@ configuration, documentation and crash-diagnostics tasks below are pending. `cluster init` now requires the registry and independent versioned bootstrap input; it cannot seal or publish from the old mixed file. The registry bootstrap test passes, and a no-registry CLI test verifies the rejection. + `cluster local-deploy` uses a fresh in-memory context for new loopback + clusters and reconstructs incremental DiskDB/ChunkDB inputs from confirmed + Group 0 hardware and live KV registrations. It no longer reads or writes + the old topology file. A real CLI regression reaches Group 0 validation + with a deliberately invalid legacy file. CLI `cluster clean` now obtains replica membership and live endpoints from Group 0. Optional storage process restarts use the validated launch registry in DiskIO, DiskDB, ChunkDB order; without one the command rejects the From e0c6c0932c65912217a2da8926621f929e53e53d Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 15:24:31 +0800 Subject: [PATCH 68/74] Remove legacy topology from CLI discovery --- app/crowdb-cli/src/commands/bench/chunk.rs | 2 +- app/crowdb-cli/src/commands/bench/disk/db.rs | 2 +- app/crowdb-cli/src/commands/bench/io.rs | 12 +---- .../src/commands/bench/kv/client.rs | 49 ++++++++----------- .../src/commands/bench/kv/prepare.rs | 4 +- app/crowdb-cli/src/commands/bench/kv/read.rs | 4 +- app/crowdb-cli/src/commands/bench/kv/scan.rs | 4 +- app/crowdb-cli/src/commands/bench/kv/write.rs | 20 ++++++-- app/crowdb-cli/src/commands/chunk/diskdb.rs | 4 +- app/crowdb-cli/src/commands/chunk/stub.rs | 2 +- app/crowdb-cli/tests/common/direct.rs | 12 ++--- .../tests/launch_registry_cli_test.rs | 3 +- doc/working/plan-console-authority.md | 4 ++ 13 files changed, 62 insertions(+), 60 deletions(-) diff --git a/app/crowdb-cli/src/commands/bench/chunk.rs b/app/crowdb-cli/src/commands/bench/chunk.rs index 1f074635..2dd7445c 100644 --- a/app/crowdb-cli/src/commands/bench/chunk.rs +++ b/app/crowdb-cli/src/commands/bench/chunk.rs @@ -40,7 +40,7 @@ pub async fn run(cli: &Cli, verb: ChunkdbBenchVerb) -> ExitCode { if !valid_args(&args) { return ExitCode::from(2); } - let kv = match build_kv_client(cli, ReadEndpointPolicy::Leader, &KvClientTunables::default()) { + let kv = match build_kv_client(cli, ReadEndpointPolicy::Leader, &KvClientTunables::default()).await { Ok(client) => Arc::new(client), Err(code) => return code, }; diff --git a/app/crowdb-cli/src/commands/bench/disk/db.rs b/app/crowdb-cli/src/commands/bench/disk/db.rs index 63875aca..21b35946 100644 --- a/app/crowdb-cli/src/commands/bench/disk/db.rs +++ b/app/crowdb-cli/src/commands/bench/disk/db.rs @@ -63,7 +63,7 @@ pub async fn run(cli: &Cli, verb: DiskdbBenchVerb) -> ExitCode { if !valid_args(&args) { return ExitCode::from(2); } - let kv = match build_kv_client(cli, ReadEndpointPolicy::Leader, &KvClientTunables::default()) { + let kv = match build_kv_client(cli, ReadEndpointPolicy::Leader, &KvClientTunables::default()).await { Ok(kv) => Arc::new(kv), Err(code) => return code, }; diff --git a/app/crowdb-cli/src/commands/bench/io.rs b/app/crowdb-cli/src/commands/bench/io.rs index 57b20f1f..2a8fd4a0 100644 --- a/app/crowdb-cli/src/commands/bench/io.rs +++ b/app/crowdb-cli/src/commands/bench/io.rs @@ -35,17 +35,7 @@ async fn connect( diskio_connections_per_endpoint: usize, diskio_rpc_workers: u32, ) -> Result { - let config = crate::commands::load_config(cli)?; - let mut seeds = vec![format!("http://{}:{}", cli.system_ip, cli.system_port)]; - for server in config - .servers - .iter() - .filter(|server| server.service_type == crowdb_console_shared::config::ServiceType::Kv) - { - if !seeds.contains(&server.url) { - seeds.push(server.url.clone()); - } - } + let seeds = vec![format!("http://{}:{}", cli.system_ip, cli.system_port)]; ChunkIoClient::connect(ChunkIoClientConfig { management_seeds: seeds, diskio_connections_per_endpoint, diff --git a/app/crowdb-cli/src/commands/bench/kv/client.rs b/app/crowdb-cli/src/commands/bench/kv/client.rs index 54222504..ce6fe8df 100644 --- a/app/crowdb-cli/src/commands/bench/kv/client.rs +++ b/app/crowdb-cli/src/commands/bench/kv/client.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Shared helper for building a [`CrowdbKvClient`] from the CLI's config +//! Shared helper for building a [`CrowdbKvClient`] from the CLI endpoint //! with a bench-specific [`ReadEndpointPolicy`]. The standard `op_context` //! helper always uses the default `Leader` policy; bench read/scan //! commands need `AnyReplica` for distributed `MinSlot` reads. @@ -38,39 +38,32 @@ impl Default for KvClientTunables { } } -/// Build a `CrowdbKvClient` from private CLI state plus `--system-*` -/// with the given `read_endpoint_policy`. Seeds the system-group hint -/// from the first config server's RPC URL (same logic as `op_context`). +/// Build a `CrowdbKvClient` from `--system-*` with the given +/// `read_endpoint_policy`. /// /// # Errors -/// Returns `ExitCode::from(2)` if the config cannot be loaded. -pub(crate) fn build_kv_client( +/// Returns `ExitCode::from(2)` if the management endpoint is unavailable. +pub(crate) async fn build_kv_client( cli: &Cli, read_endpoint_policy: ReadEndpointPolicy, tunables: &KvClientTunables, ) -> Result { - let config = crate::commands::load_config(cli)?; - let mgmt_url = format!("http://{}:{}", cli.system_ip, cli.system_port); - let group0_endpoint = format!("{}:{}", cli.system_ip, cli.system_port); - - let mut seeds = vec![mgmt_url]; - for server in &config.servers { - if !seeds.contains(&server.url) { - seeds.push(server.url.clone()); - } - } - - let effective_g0 = - config - .servers - .first() - .and_then(|s| s.rpc_url.as_ref()) - .map_or(group0_endpoint, |url| { - url.strip_prefix("http://") - .or_else(|| url.strip_prefix("https://")) - .unwrap_or(url) - .to_string() - }); + let seed = format!("http://{}:{}", cli.system_ip, cli.system_port); + let server = crowdb_console_shared::clients::http::ServerClient::new(&seed).map_err(|error| { + eprintln!("bench kv client: {error}"); + ExitCode::from(2) + })?; + let effective_g0 = server + .topology() + .await + .ok() + .and_then(|stores| stores.into_iter().find(|store| store.store_id == 0)) + .and_then(|store| store.listen_addr) + .ok_or_else(|| { + eprintln!("bench kv client: Group 0 endpoint unavailable at {seed}"); + ExitCode::from(2) + })?; + let seeds = vec![seed]; let mut client_config = ClientConfig::new(seeds); client_config.read_endpoint_policy = read_endpoint_policy; diff --git a/app/crowdb-cli/src/commands/bench/kv/prepare.rs b/app/crowdb-cli/src/commands/bench/kv/prepare.rs index 46432255..f7fbbe4c 100644 --- a/app/crowdb-cli/src/commands/bench/kv/prepare.rs +++ b/app/crowdb-cli/src/commands/bench/kv/prepare.rs @@ -23,7 +23,9 @@ pub async fn run(cli: &Cli, args: PrepareArgs) -> ExitCode { cli, crowdb_kv_client::ReadEndpointPolicy::Leader, &KvClientTunables::default(), - ) { + ) + .await + { Ok(c) => Arc::new(c), Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/bench/kv/read.rs b/app/crowdb-cli/src/commands/bench/kv/read.rs index 3f668c7c..4d707330 100644 --- a/app/crowdb-cli/src/commands/bench/kv/read.rs +++ b/app/crowdb-cli/src/commands/bench/kv/read.rs @@ -42,7 +42,9 @@ pub async fn run(cli: &Cli, args: ReadArgs) -> ExitCode { pool_size: args.connections, ..Default::default() }, - ) { + ) + .await + { Ok(c) => Arc::new(c), Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/bench/kv/scan.rs b/app/crowdb-cli/src/commands/bench/kv/scan.rs index 0f5945d9..cd84dc08 100644 --- a/app/crowdb-cli/src/commands/bench/kv/scan.rs +++ b/app/crowdb-cli/src/commands/bench/kv/scan.rs @@ -43,7 +43,9 @@ pub async fn run(cli: &Cli, args: ScanArgs) -> ExitCode { pool_size: args.connections, ..Default::default() }, - ) { + ) + .await + { Ok(c) => Arc::new(c), Err(c) => return c, }; diff --git a/app/crowdb-cli/src/commands/bench/kv/write.rs b/app/crowdb-cli/src/commands/bench/kv/write.rs index 49f1a29f..0c38e51e 100644 --- a/app/crowdb-cli/src/commands/bench/kv/write.rs +++ b/app/crowdb-cli/src/commands/bench/kv/write.rs @@ -18,13 +18,13 @@ use rand::rngs::SmallRng; use rand::{Rng, SeedableRng}; use super::client::{build_kv_client, KvClientTunables}; +use crate::commands::authority_context; use crate::commands::bench::loader::{run_workload, BenchRecorder}; use crate::commands::bench::metrics::BenchMetrics; use crate::commands::bench::result::{ BenchOps, BenchResult, ReplicaStats, ServerMetrics, ServerRpcLatency, SnapshotStats, TransportStats, }; use crate::commands::bench::verb::WriteArgs; -use crate::commands::load_config; use crate::Cli; #[allow(clippy::too_many_lines)] @@ -45,7 +45,9 @@ pub async fn run(cli: &Cli, args: WriteArgs) -> ExitCode { pool_size: args.connections, ..Default::default() }, - ) { + ) + .await + { Ok(c) => Arc::new(c), Err(c) => return c, }; @@ -173,16 +175,24 @@ fn build_value(id: u64, size: usize) -> Vec { .collect() } -/// Fetch `/metrics` from every server in the config and aggregate into +/// Fetch `/metrics` from confirmed replica hosts and aggregate into /// `ServerMetrics`. Metrics are summed across nodes except for averages /// (which are averaged across nodes that report them). #[allow(clippy::too_many_lines)] async fn fetch_server_metrics(cli: &Cli, store_id: u64, group_id: u64) -> Option { - let config = load_config(cli).ok()?; + let ctx = authority_context(cli).await.ok()?; + let replicas = ctx + .sysmd() + .list_replicas_in_group(store_id, group_id) + .await + .ok()?; let group_prefix = format!("s.{store_id}.g.{group_id}."); let store_prefix = format!("s.{store_id}.rpc."); - let mut mgmt_urls: Vec = config.servers.iter().map(|s| s.url.clone()).collect(); + let mut mgmt_urls = Vec::with_capacity(replicas.len()); + for replica in replicas { + mgmt_urls.push(ctx.live_node_mgmt_url(replica.node_id).await.ok()?); + } mgmt_urls.sort(); mgmt_urls.dedup(); if mgmt_urls.is_empty() { diff --git a/app/crowdb-cli/src/commands/chunk/diskdb.rs b/app/crowdb-cli/src/commands/chunk/diskdb.rs index fbea0760..737697fe 100644 --- a/app/crowdb-cli/src/commands/chunk/diskdb.rs +++ b/app/crowdb-cli/src/commands/chunk/diskdb.rs @@ -9,8 +9,8 @@ use clap::Subcommand; use crowdb_console_shared::ops::chunk; +use crate::commands::authority_context; use crate::commands::launch::{self, LaunchVerb}; -use crate::commands::op_context; use crate::Cli; #[derive(Subcommand, Debug)] @@ -93,7 +93,7 @@ pub async fn run_chunk_diskdb_verb(cli: &Cli, verb: ChunkDiskdbVerb) -> ExitCode } async fn run_list(cli: &Cli, explicit_endpoint: Option<&str>) -> ExitCode { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(ctx) => ctx, Err(code) => return code, }; diff --git a/app/crowdb-cli/src/commands/chunk/stub.rs b/app/crowdb-cli/src/commands/chunk/stub.rs index 825dde45..a7900561 100644 --- a/app/crowdb-cli/src/commands/chunk/stub.rs +++ b/app/crowdb-cli/src/commands/chunk/stub.rs @@ -70,7 +70,7 @@ async fn run_service(cli: &Cli, service: &str, verb: ServiceVerb) -> ExitCode { } async fn run_list(cli: &Cli, service: &str) -> ExitCode { - let ctx = match crate::commands::op_context(cli) { + let ctx = match crate::commands::authority_context(cli).await { Ok(ctx) => ctx, Err(code) => return code, }; diff --git a/app/crowdb-cli/tests/common/direct.rs b/app/crowdb-cli/tests/common/direct.rs index b4a69f88..0516de9a 100644 --- a/app/crowdb-cli/tests/common/direct.rs +++ b/app/crowdb-cli/tests/common/direct.rs @@ -17,7 +17,7 @@ use std::time::Duration; use crowdb_console_shared::clients::http::ServerClient; use crowdb_console_shared::config::{NodeEntry, RackEntry, ServerEntry, ServiceType}; use crowdb_console_shared::lifecycle::{self, crowdb_kv_server_bin, DeployRequest}; -use crowdb_console_shared::{ConsoleConfig, ConsoleConfigEngine}; +use crowdb_console_shared::ConsoleConfig; use crowdb_test_harness::test_dirs; /// Allocate a free mgmt port for a kv-server. @@ -61,6 +61,7 @@ pub struct Group0 { pub mgmt_port: u16, pub rpc_port: u16, pub config_path: PathBuf, + pub bootstrap_config: ConsoleConfig, workspace: std::path::PathBuf, } @@ -90,7 +91,7 @@ pub fn local_node(id: u64, rack: u64) -> NodeEntry { } /// Fork a real `crowdb-kv-server` for node 1, initialize group 0 on it, -/// and write a console config with the rack/node/server entries. Returns +/// and prepare in-memory bootstrap entries. Returns /// `None` when the server binary has not been built. pub async fn spawn_group0() -> Option { let bin = crowdb_kv_server_bin()?; @@ -123,7 +124,7 @@ pub async fn spawn_group0() -> Option { .await .expect("deploy_local_in_dir"); - // Write console config with rack/node/server entries. + // Prepare bootstrap input without persisting a topology copy. let mut cfg = ConsoleConfig::default(); cfg.racks.push(RackEntry { id: 1, @@ -148,8 +149,6 @@ pub async fn spawn_group0() -> Option { .unwrap(); let config_path = workspace.join("console.toml"); - let engine = crowdb_console_shared::TomlFileEngine::new(config_path.clone()); - engine.save(&cfg).expect("save config"); // Initialize group 0 on the server (single-node, self-elect). let client = ServerClient::new(deployed.mgmt_url.clone()).unwrap(); @@ -166,7 +165,7 @@ pub async fn spawn_group0() -> Option { let context = crowdb_console_shared::ops::OpContext::new( deployed.rpc_url.trim_start_matches("http://").to_string(), vec![deployed.mgmt_url.clone()], - cfg, + cfg.clone(), ); let deadline = std::time::Instant::now() + Duration::from_secs(5); loop { @@ -193,6 +192,7 @@ pub async fn spawn_group0() -> Option { mgmt_port: rest_port, rpc_port, config_path, + bootstrap_config: cfg, workspace, }) } diff --git a/app/crowdb-cli/tests/launch_registry_cli_test.rs b/app/crowdb-cli/tests/launch_registry_cli_test.rs index a48e362f..821f2452 100644 --- a/app/crowdb-cli/tests/launch_registry_cli_test.rs +++ b/app/crowdb-cli/tests/launch_registry_cli_test.rs @@ -236,8 +236,7 @@ async fn registry_bootstrap_uses_sealed_intent_without_legacy_topology_file() { .save(&path) .unwrap(); let source = dir.path().join("bootstrap-source.toml"); - let config = crowdb_console_shared::ConsoleConfig::load(&g0.config_path).unwrap(); - let intent = BootstrapIntent::capture(&config, &[1]).unwrap(); + let intent = BootstrapIntent::capture(&g0.bootstrap_config, &[1]).unwrap(); intent.seal(&source).unwrap(); std::fs::write(dir.path().join("invalid-legacy.toml"), "invalid legacy config").unwrap(); diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 2a7ec8f8..f47f9139 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -118,6 +118,10 @@ configuration, documentation and crash-diagnostics tasks below are pending. Group 0. Optional storage process restarts use the validated launch registry in DiskIO, DiskDB, ChunkDB order; without one the command rejects the restart request before wiping data. + CLI chunk service lists and benchmark discovery no longer load the old + topology file. KV benchmark metrics resolve confirmed replica hosts and + their live management registrations. The direct CLI fixture keeps bootstrap + input only in memory. The complete CLI suite and workspace clippy pass. - [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through shared Group 0 hardware operations; preserve conflicts and uncertain writes without local-first commits. Docker keeps its hardware restrictions. Hardware From 32dffe93f07af145401c216cc24efde40cde3396 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 15:31:56 +0800 Subject: [PATCH 69/74] Remove mixed console configuration persistence --- app/crowdb-cli/src/commands.rs | 97 --- app/crowdb-cli/src/commands/cluster.rs | 38 +- app/crowdb-cli/tests/direct_cli_test.rs | 62 ++ doc/working/plan-console-authority.md | 6 + lib/crowdb-console-shared/src/config.rs | 702 +----------------- lib/crowdb-console-shared/src/lib.rs | 5 +- lib/crowdb-console-shared/src/ops.rs | 4 +- lib/crowdb-console-shared/src/ops/cluster.rs | 135 ++-- lib/crowdb-console-shared/src/ops/context.rs | 5 +- .../tests/common/bootstrap_node.rs | 6 +- 10 files changed, 157 insertions(+), 903 deletions(-) diff --git a/app/crowdb-cli/src/commands.rs b/app/crowdb-cli/src/commands.rs index b578eca1..b33f611a 100644 --- a/app/crowdb-cli/src/commands.rs +++ b/app/crowdb-cli/src/commands.rs @@ -28,50 +28,9 @@ pub(crate) use s3::{run_s3_verb, S3Verb}; use std::process::ExitCode; use crowdb_console_shared::ops::OpContext; -use crowdb_console_shared::ConsoleConfigEngine; use crate::Cli; -/// Build an [`OpContext`] from the CLI global flags. The system endpoint -/// is `http://{system_ip}:{system_port}` and the -/// CLI state is loaded from the fixed runtime location. -/// -/// When the config has server entries, their mgmt URLs are added as -/// additional seeds so the client can discover the group-0 leader even -/// if `--system-port` doesn't point at a running server (e.g. after -/// `local-deploy` which allocates dynamic ports). -pub(crate) fn op_context(cli: &Cli) -> Result { - let config = load_config(cli)?; - let mgmt_url = format!("http://{}:{}", cli.system_ip, cli.system_port); - let group0_endpoint = format!("{}:{}", cli.system_ip, cli.system_port); - - // Collect mgmt seeds: the explicit --system-port endpoint plus all - // server URLs from the config (so local-deploy'd servers are found). - let mut seeds = vec![mgmt_url]; - for server in &config.servers { - if !seeds.contains(&server.url) { - seeds.push(server.url.clone()); - } - } - - // Use the first config server's RPC URL as the group0 endpoint hint - // if available (more accurate than the default port). Strip the - // `http://` prefix since the crowdb-rpc endpoint format is `ip:port`. - let effective_g0 = - config - .servers - .first() - .and_then(|s| s.rpc_url.as_ref()) - .map_or(group0_endpoint, |url| { - url.strip_prefix("http://") - .or_else(|| url.strip_prefix("https://")) - .unwrap_or(url) - .to_string() - }); - - Ok(OpContext::new(effective_g0, seeds, config)) -} - /// Build a Group 0 context for hardware operations without reading the old /// local console state. The CLI endpoint is a discovery seed, while the /// optional launch registry is validated only as local process policy. @@ -99,59 +58,3 @@ pub(crate) async fn authority_context(cli: &Cli) -> Result crowdb_console_shared::ConsoleConfig::default(), )) } - -/// Load the CLI's internal persisted state from its fixed runtime location. -pub(crate) fn load_config(cli: &Cli) -> Result { - if let Some(path) = &cli.registry { - crowdb_console_shared::config::web::LaunchRegistry::load(path).map_err(|error| { - eprintln!("error: load launch registry: {error}"); - ExitCode::from(2) - })?; - return Ok(crowdb_console_shared::ConsoleConfig::default()); - } - let path = config_path(); - if !path.exists() { - return Ok(crowdb_console_shared::ConsoleConfig::default()); - } - let engine = crowdb_console_shared::TomlFileEngine::new(path); - engine.load().map_err(|e| { - eprintln!("error: load config: {e}"); - ExitCode::from(2) - }) -} - -/// Resolve the private CLI state file. The environment override is reserved -/// for isolated test and benchmark harnesses and is intentionally not a CLI -/// option. -pub(crate) fn config_path() -> std::path::PathBuf { - std::env::var_os("CROWDB_CLI_STATE").map_or_else( - || { - crowdb_protocol::port::namespace::runtime_root() - .join("persistent") - .join("console") - .join("crowdb-kv.db.toml") - }, - std::path::PathBuf::from, - ) -} - -/// Persist the config from an [`OpContext`] back to the config file. -pub(crate) fn commit_config(cli: &Cli, ctx: &OpContext) -> Result<(), ExitCode> { - if cli.registry.is_some() { - eprintln!("error: launch registry mode cannot persist local cluster topology"); - return Err(ExitCode::from(2)); - } - let path = config_path(); - if let Some(parent) = path.parent() { - std::fs::create_dir_all(parent).map_err(|e| { - eprintln!("error: create config dir {}: {e}", parent.display()); - ExitCode::from(2) - })?; - } - let engine = crowdb_console_shared::TomlFileEngine::new(path.clone()); - let cfg = ctx.config().clone(); - engine.save(&cfg).map_err(|e| { - eprintln!("error: save config {}: {e}", path.display()); - ExitCode::from(2) - }) -} diff --git a/app/crowdb-cli/src/commands/cluster.rs b/app/crowdb-cli/src/commands/cluster.rs index 4bad76ca..24b3bdc6 100644 --- a/app/crowdb-cli/src/commands/cluster.rs +++ b/app/crowdb-cli/src/commands/cluster.rs @@ -15,7 +15,7 @@ use std::process::ExitCode; use clap::Subcommand; -use crate::commands::{authority_context, commit_config, op_context}; +use crate::commands::authority_context; use crate::Cli; async fn local_deploy_context( @@ -179,7 +179,7 @@ pub enum ClusterVerb { }, /// Tear down the entire cluster (all groups, stores, servers, sysdata). Destroy, - /// Remove orphaned sysdata entries without stopping running servers. + /// Verify confirmed store hosts have live registrations and healthy servers. Reset, /// Wipe user data on every node + wait for re-election. Preserves /// group-0 sysdata + topology — servers stay running. Use --store/--group @@ -555,14 +555,38 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } }, ClusterVerb::Destroy => { - let ctx = match op_context(cli) { + let Some(path) = &cli.registry else { + eprintln!("error: cluster destroy requires --registry to stop local processes"); + return ExitCode::from(2); + }; + let registry = match crowdb_console_shared::config::web::LaunchRegistry::load(path) { + Ok(registry) => registry, + Err(error) => { + eprintln!("error: load launch registry: {error}"); + return ExitCode::from(2); + } + }; + let runtime = match crowdb_console_shared::launch::LaunchRuntime::for_registry(path) { + Ok(runtime) => runtime, + Err(error) => { + eprintln!("error: launch runtime: {error}"); + return ExitCode::from(2); + } + }; + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; match crowdb_console_shared::ops::cluster::destroy(&ctx).await { Ok(()) => { - if let Err(c) = commit_config(cli, &ctx) { - return c; + for launch in ®istry.launches { + if let Err(error) = runtime.stop(launch).await { + eprintln!( + "error: stop {} on node {}: {error}", + launch.service_id, launch.node_id + ); + return ExitCode::from(2); + } } println!("cluster destroy complete"); ExitCode::SUCCESS @@ -574,13 +598,13 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } ClusterVerb::Reset => { - let ctx = match op_context(cli) { + let ctx = match authority_context(cli).await { Ok(c) => c, Err(c) => return c, }; match crowdb_console_shared::ops::cluster::reset(&ctx).await { Ok(()) => { - println!("cluster reset complete"); + println!("cluster membership verified"); ExitCode::SUCCESS } Err(e) => { diff --git a/app/crowdb-cli/tests/direct_cli_test.rs b/app/crowdb-cli/tests/direct_cli_test.rs index 7b4dd8d0..839898ea 100644 --- a/app/crowdb-cli/tests/direct_cli_test.rs +++ b/app/crowdb-cli/tests/direct_cli_test.rs @@ -114,3 +114,65 @@ async fn incremental_local_deploy_reads_group_zero_instead_of_legacy_file() { assert!(stderr.contains("99"), "stderr={stderr}"); assert!(!stderr.contains("load config"), "stderr={stderr}"); } + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reset_verifies_confirmed_hosts_without_legacy_file() { + let Some(g0) = spawn_group0().await else { + return; + }; + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); + let (code, stdout, stderr) = run( + &crowdb_cli_bin(), + g0.mgmt_port, + &g0.config_path, + &["cluster", "reset"], + ); + assert_eq!(code, 0, "stderr={stderr}"); + assert!(stdout.contains("membership verified"), "stdout={stdout}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn destroy_reads_group_zero_and_requires_launch_registry() { + let Some(g0) = spawn_group0().await else { + return; + }; + std::fs::write(&g0.config_path, "invalid local topology").unwrap(); + let (code, _, stderr) = run( + &crowdb_cli_bin(), + g0.mgmt_port, + &g0.config_path, + &["cluster", "destroy"], + ); + assert_eq!(code, 2, "stderr={stderr}"); + assert!(stderr.contains("--registry"), "stderr={stderr}"); + + let context = crowdb_console_shared::ops::OpContext::new( + g0.rpc_url.trim_start_matches("http://").to_string(), + vec![g0.mgmt_url.clone()], + g0.bootstrap_config.clone(), + ); + crowdb_console_shared::ops::cluster::init(&context, &[1]) + .await + .expect("publish confirmed bootstrap metadata"); + + let registry_path = g0.config_path.with_file_name("launches.toml"); + crowdb_console_shared::config::web::LaunchRegistry { + version: 1, + launches: Vec::new(), + } + .save(®istry_path) + .unwrap(); + let output = std::process::Command::new(crowdb_cli_bin()) + .args(["--registry", registry_path.to_str().unwrap(), "--system-port"]) + .arg(g0.mgmt_port.to_string()) + .args(["cluster", "destroy"]) + .env("CROWDB_CLI_STATE", &g0.config_path) + .output() + .unwrap(); + assert!( + output.status.success(), + "stdout={} stderr={}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); +} diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index f47f9139..71b17a2f 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -122,6 +122,12 @@ configuration, documentation and crash-diagnostics tasks below are pending. topology file. KV benchmark metrics resolve confirmed replica hosts and their live management registrations. The direct CLI fixture keeps bootstrap input only in memory. The complete CLI suite and workspace clippy pass. + The `ConsoleConfigEngine`/`TomlFileEngine` parser and writer are removed; + `ConsoleConfig` remains an ephemeral operation/intent input. CLI `destroy` + requires a launch registry, removes confirmed logical metadata before Group + 0, then stops configured local processes. CLI `reset` verifies live confirmed + hosts without deleting stopped nodes as presumed orphans. Real one-node + CLI regressions cover both commands with an invalid legacy file. - [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through shared Group 0 hardware operations; preserve conflicts and uncertain writes without local-first commits. Docker keeps its hardware restrictions. Hardware diff --git a/lib/crowdb-console-shared/src/config.rs b/lib/crowdb-console-shared/src/config.rs index d0c52260..50c913b6 100644 --- a/lib/crowdb-console-shared/src/config.rs +++ b/lib/crowdb-console-shared/src/config.rs @@ -1,167 +1,24 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Console configuration: persisted registry of `crowdb-kv-server` instances. -//! -//! C2 status: file-backed `[[server]]` list; later phases extend with -//! racks, nodes, ssh creds. The struct is the single source of truth so -//! the storage format can evolve without touching call sites. +//! In-memory console operation inputs and topology snapshots. Durable cluster +//! records live in Group 0; process launch policy uses `config::web`. -use std::path::{Path, PathBuf}; -use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::atomic::AtomicU64; use crowdb_protocol::{NodeId, RackId}; use serde::{Deserialize, Serialize}; use std::collections::BTreeMap; -static TMP_FILE_COUNTER: AtomicU64 = AtomicU64::new(0); - use crate::cluster::DiskGroupId; use crate::error::{Error, Result}; -use std::fmt; - pub mod web; -/// Serde helper: serialize a `BTreeMap` with string keys (TOML -/// requires string keys) and deserialize back to `u64` keys. -mod int_key { - use serde::de::{Deserialize, Deserializer, MapAccess, Visitor}; - use serde::ser::{Serialize, Serializer}; - use std::collections::BTreeMap; - use std::fmt; - use std::marker::PhantomData; - use std::str::FromStr; - - pub fn serialize(map: &BTreeMap, serializer: S) -> Result - where - K: ToString + Ord, - V: Serialize, - S: Serializer, - { - let string_map: BTreeMap = map.iter().map(|(k, v)| (k.to_string(), v)).collect(); - string_map.serialize(serializer) - } - - pub fn deserialize<'de, K, V, D>(deserializer: D) -> Result, D::Error> - where - K: FromStr + Ord, - K::Err: fmt::Display, - V: Deserialize<'de>, - D: Deserializer<'de>, - { - struct IntKeyVisitor(PhantomData<(K, V)>); - - impl<'de, K, V> Visitor<'de> for IntKeyVisitor - where - K: FromStr + Ord, - K::Err: fmt::Display, - V: Deserialize<'de>, - { - type Value = BTreeMap; - - fn expecting(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.write_str("a map with string-encoded integer keys") - } - - fn visit_map(self, mut access: A) -> Result - where - A: MapAccess<'de>, - { - let mut map = BTreeMap::new(); - while let Some((key, value)) = access.next_entry::()? { - let k = K::from_str(&key).map_err(serde::de::Error::custom)?; - map.insert(k, value); - } - Ok(map) - } - } - - deserializer.deserialize_map(IntKeyVisitor::(PhantomData)) - } -} - -pub trait ConsoleConfigEngine: Send + Sync { - /// Load the console configuration from the engine's storage. - /// - /// # Errors - /// Returns an error if loading fails (e.g., file not found, parse error). - fn load(&self) -> Result; - - /// Save the console configuration to the engine's storage. - /// - /// # Errors - /// Returns an error if saving fails (e.g., permission denied, write error). - fn save(&self, config: &ConsoleConfig) -> Result<()>; -} - -#[derive(Clone, PartialEq, Eq)] -pub struct TomlFileEngine { - path: PathBuf, -} - -impl fmt::Debug for TomlFileEngine { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.debug_struct("TomlFileEngine") - .field("path", &self.path) - .finish() - } -} - -impl TomlFileEngine { - #[must_use] - pub fn new(path: impl Into) -> Self { - Self { path: path.into() } - } - - #[must_use] - #[allow(dead_code)] - pub(crate) fn path(&self) -> &Path { - &self.path - } - - #[must_use] - pub fn default_path() -> Option { - Some( - crowdb_protocol::port::namespace::runtime_root() - .join("persistent") - .join("console") - .join("crowdb-kv.db.toml"), - ) - } - - #[must_use] - #[allow(dead_code)] - pub(crate) fn from_default_path() -> Option { - Self::default_path().map(Self::new) - } -} - -impl ConsoleConfigEngine for TomlFileEngine { - fn load(&self) -> Result { - match std::fs::read_to_string(&self.path) { - Ok(body) => ConsoleConfig::from_toml_str(&body, &self.path), - Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(ConsoleConfig::default()), - Err(e) => Err(Error::Io(e)), - } - } - - fn save(&self, config: &ConsoleConfig) -> Result<()> { - if let Some(parent) = self.path.parent() { - std::fs::create_dir_all(parent).map_err(Error::Io)?; - } - let body = config.to_toml_string()?; - let seq = TMP_FILE_COUNTER.fetch_add(1, Ordering::Relaxed); - let tmp = self.path.with_extension(format!("toml.tmp.{seq}")); - std::fs::write(&tmp, body).map_err(Error::Io)?; - std::fs::rename(&tmp, &self.path).map_err(Error::Io)?; - Ok(()) - } -} +static TMP_FILE_COUNTER: AtomicU64 = AtomicU64::new(0); -/// On-disk console config. New top-level fields land in later phases -/// (ssh defaults, etc.). Unknown fields are ignored on load and dropped -/// on save (`serde(default)` everywhere) to keep migrations easy. +/// Ephemeral inputs for cluster operations and bootstrap. This struct is not +/// a durable topology store. #[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] pub struct ConsoleConfig { #[serde(default, rename = "rack")] @@ -181,9 +38,6 @@ pub struct ConsoleConfig { /// Reproducible commands for locally deployed benchmark services. #[serde(default)] pub local_launches: BTreeMap, - /// Optional `[bench]` section. Reserved for future use. - #[serde(default, skip_serializing_if = "BenchConfig::is_empty")] - pub(crate) bench: BenchConfig, } /// Retained local process state used to restart a benchmark service without @@ -200,19 +54,6 @@ pub struct LocalLaunchSpec { pub readiness_url: Option, } -/// `[bench]` section. Reserved for future knobs (default reporting -/// dir, max threads, etc.); currently empty. -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -pub(crate) struct BenchConfig {} - -impl BenchConfig { - #[must_use] - #[allow(clippy::unused_self)] - pub(crate) fn is_empty(&self) -> bool { - true - } -} - #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct RackEntry { pub id: RackId, @@ -374,114 +215,6 @@ pub struct ReplicaEntry { pub node_id: NodeId, } -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedConsoleConfig { - #[serde(default, with = "int_key", skip_serializing_if = "BTreeMap::is_empty")] - rack: BTreeMap, - #[serde(default, with = "int_key", skip_serializing_if = "BTreeMap::is_empty")] - node: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - crowdb_kv_server: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - store: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - group: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - disk_group: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - disk: BTreeMap, - #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - local_launch: BTreeMap, - #[serde(default, skip_serializing_if = "BenchConfig::is_empty")] - bench: BenchConfig, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedDiskGroupEntry { - id: DiskGroupId, - rack_id: RackId, - node_id: NodeId, - #[serde(default)] - name: String, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedDiskEntry { - disk_id: String, - disk_group_id: DiskGroupId, - rack_id: RackId, - node_id: NodeId, - disk_type: String, - capacity_bytes: u64, - zone_size_bytes: u64, - unit_size_bytes: u32, - #[serde(default, skip_serializing_if = "String::is_empty")] - device_path: String, -} - -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedRackEntry { - #[serde(default)] - name: String, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedNodeEntry { - rack_id: RackId, - host: String, - #[serde(default = "default_ssh_port")] - ssh_port: u16, - #[serde(default)] - ssh_user: String, - #[serde(default, skip_serializing_if = "Option::is_none")] - ssh_credential_ref: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - ssh_key: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - ssh_password: Option, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedServerEntry { - node_id: Option, - url: String, - #[serde(default, skip_serializing_if = "Option::is_none")] - rpc_url: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - rest_port: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - rpc_port: Option, - #[serde(default)] - auto_start: bool, - #[serde(default, skip_serializing_if = "Option::is_none")] - binary: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - election_profile: Option, - #[serde(default, skip_serializing_if = "is_default_service_type")] - service_type: ServiceType, - #[serde(default, skip_serializing_if = "Option::is_none")] - rpc_workers: Option, - #[serde(default, skip_serializing_if = "std::ops::Not::not")] - no_fsync: bool, - #[serde(default, skip_serializing_if = "Option::is_none")] - pid: Option, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedStoreEntry { - store_id: u64, - #[serde(default)] - nodes: Vec, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -struct PersistedGroupEntry { - store_id: u64, - group_id: u64, - #[serde(default)] - replicas: Vec, -} - impl ServerEntry { /// Convenience constructor for a plain registered server (C2 style). #[must_use] @@ -505,49 +238,6 @@ impl ServerEntry { } impl ConsoleConfig { - /// Default config file path. - /// - /// Config is persisted below the workspace persistent runtime namespace. - /// This file stores registered crowdb-kv-server instances for the console. - #[must_use] - pub(crate) fn default_path() -> Option { - TomlFileEngine::default_path() - } - - /// Load the config from `path`. A missing file yields a default - /// (empty) config so first-run is friendly. - /// - /// # Errors - /// Returns `Error::Io` for non-`NotFound` filesystem errors and - /// `Error::Config` for TOML parse failures. - pub fn load(path: &Path) -> Result { - TomlFileEngine::new(path).load() - } - - /// Save the config atomically (write to a tempfile, then rename). - /// - /// # Errors - /// Filesystem and TOML serialization errors are propagated. - pub(crate) fn save(&self, path: &Path) -> Result<()> { - TomlFileEngine::new(path).save(self) - } - - /// Load configuration using the provided engine. - /// - /// # Errors - /// Returns an error if the engine's load fails. - pub fn load_with_engine(engine: &dyn ConsoleConfigEngine) -> Result { - engine.load() - } - - /// Save configuration using the provided engine. - /// - /// # Errors - /// Returns an error if the engine's save fails. - pub fn save_with_engine(&self, engine: &dyn ConsoleConfigEngine) -> Result<()> { - engine.save(self) - } - /// Add a server entry. Rejects duplicate `id` and duplicate `url`. /// /// # Errors @@ -950,371 +640,11 @@ impl ConsoleConfig { pub fn disks_in_group(&self, dg_id: DiskGroupId) -> Vec<&DiskEntry> { self.disks.iter().filter(|d| d.disk_group_id == dg_id).collect() } - - #[allow(clippy::too_many_lines)] - fn to_persisted(&self) -> PersistedConsoleConfig { - let rack = self - .racks - .iter() - .map(|entry| { - ( - entry.id, - PersistedRackEntry { - name: entry.name.clone(), - }, - ) - }) - .collect(); - let node = self - .nodes - .iter() - .map(|entry| { - ( - entry.id, - PersistedNodeEntry { - rack_id: entry.rack_id, - host: entry.host.clone(), - ssh_port: entry.ssh_port, - ssh_user: entry.ssh_user.clone(), - ssh_credential_ref: entry.ssh_credential_ref.clone(), - ssh_key: entry.ssh_key.clone(), - ssh_password: entry.ssh_password.clone(), - }, - ) - }) - .collect(); - let crowdb_kv_server = self - .servers - .iter() - .map(|entry| { - ( - entry.id.clone(), - PersistedServerEntry { - node_id: entry.node_id, - url: entry.url.clone(), - rpc_url: entry.rpc_url.clone(), - rest_port: entry.rest_port, - rpc_port: entry.rpc_port, - auto_start: entry.auto_start, - binary: entry.binary.clone(), - election_profile: entry.election_profile.clone(), - service_type: entry.service_type, - rpc_workers: entry.rpc_workers, - no_fsync: entry.no_fsync, - pid: entry.pid, - }, - ) - }) - .collect(); - let store = self - .stores - .iter() - .map(|entry| { - ( - entry.store_id.to_string(), - PersistedStoreEntry { - store_id: entry.store_id, - nodes: entry.nodes.clone(), - }, - ) - }) - .collect(); - let group = self - .groups - .iter() - .map(|entry| { - ( - format!("{}:{}", entry.store_id, entry.group_id), - PersistedGroupEntry { - store_id: entry.store_id, - group_id: entry.group_id, - replicas: entry.replicas.clone(), - }, - ) - }) - .collect(); - let disk_group = self - .disk_groups - .iter() - .map(|entry| { - ( - entry.id.to_string(), - PersistedDiskGroupEntry { - id: entry.id, - rack_id: entry.rack_id, - node_id: entry.node_id, - name: entry.name.clone(), - }, - ) - }) - .collect(); - let disk = self - .disks - .iter() - .map(|entry| { - ( - entry.disk_id.clone(), - PersistedDiskEntry { - disk_id: entry.disk_id.clone(), - disk_group_id: entry.disk_group_id, - rack_id: entry.rack_id, - node_id: entry.node_id, - disk_type: entry.disk_type.clone(), - capacity_bytes: entry.capacity_bytes, - zone_size_bytes: entry.zone_size_bytes, - unit_size_bytes: entry.unit_size_bytes, - device_path: entry.device_path.clone(), - }, - ) - }) - .collect(); - PersistedConsoleConfig { - rack, - node, - crowdb_kv_server, - store, - group, - disk_group, - disk, - local_launch: self.local_launches.clone(), - bench: self.bench.clone(), - } - } - - fn from_persisted(persisted: PersistedConsoleConfig) -> Self { - let mut racks: Vec = persisted - .rack - .into_iter() - .map(|(id, entry)| RackEntry { id, name: entry.name }) - .collect(); - racks.sort_by_key(|r| r.id); - let mut nodes: Vec = persisted - .node - .into_iter() - .map(|(id, entry)| NodeEntry { - id, - rack_id: entry.rack_id, - host: entry.host, - ssh_port: entry.ssh_port, - ssh_user: entry.ssh_user, - ssh_credential_ref: entry.ssh_credential_ref, - ssh_key: entry.ssh_key, - ssh_password: entry.ssh_password, - }) - .collect(); - nodes.sort_by_key(|n| n.id); - let mut servers: Vec = persisted - .crowdb_kv_server - .into_iter() - .map(|(id, entry)| ServerEntry { - id, - url: entry.url, - node_id: entry.node_id, - rpc_url: entry.rpc_url, - rest_port: entry.rest_port, - rpc_port: entry.rpc_port, - auto_start: entry.auto_start, - binary: entry.binary, - election_profile: entry.election_profile, - pid: entry.pid, - service_type: entry.service_type, - rpc_workers: entry.rpc_workers, - no_fsync: entry.no_fsync, - }) - .collect(); - servers.sort_by(|a, b| a.id.cmp(&b.id)); - let mut stores: Vec = persisted - .store - .into_values() - .map(|entry| StoreEntry { - store_id: entry.store_id, - nodes: entry.nodes, - }) - .collect(); - stores.sort_by_key(|s| s.store_id); - let mut groups: Vec = persisted - .group - .into_values() - .map(|entry| GroupEntry { - store_id: entry.store_id, - group_id: entry.group_id, - replicas: entry.replicas, - }) - .collect(); - groups.sort_by_key(|g| (g.store_id, g.group_id)); - let mut disk_groups: Vec = persisted - .disk_group - .into_values() - .map(|entry| DiskGroupEntry { - id: entry.id, - rack_id: entry.rack_id, - node_id: entry.node_id, - name: entry.name, - }) - .collect(); - disk_groups.sort_by_key(|dg| dg.id); - let mut disks: Vec = persisted - .disk - .into_values() - .map(|entry| DiskEntry { - disk_id: entry.disk_id, - disk_group_id: entry.disk_group_id, - rack_id: entry.rack_id, - node_id: entry.node_id, - disk_type: entry.disk_type, - capacity_bytes: entry.capacity_bytes, - zone_size_bytes: entry.zone_size_bytes, - unit_size_bytes: entry.unit_size_bytes, - device_path: entry.device_path, - }) - .collect(); - disks.sort_by(|a, b| a.disk_id.cmp(&b.disk_id)); - Self { - racks, - nodes, - servers, - stores, - groups, - disk_groups, - disks, - local_launches: persisted.local_launch, - bench: persisted.bench, - } - } - - fn from_toml_str(body: &str, path: &Path) -> Result { - let persisted: PersistedConsoleConfig = - toml::from_str(body).map_err(|e| Error::Config(format!("{}: {e}", path.display())))?; - Ok(Self::from_persisted(persisted)) - } - - fn to_toml_string(&self) -> Result { - toml::to_string_pretty(&self.to_persisted()).map_err(|e| Error::Config(format!("serialize: {e}"))) - } } #[cfg(test)] mod tests { - use super::{ - ConsoleConfig, GroupEntry, LocalLaunchSpec, ReplicaEntry, ServerEntry, StoreEntry, TomlFileEngine, - }; - use crowdb_test_harness::test_dirs; - - #[test] - fn round_trip_load_save() { - let dir = tempdir(); - let path = dir.join("console.toml"); - - let mut cfg = ConsoleConfig::default(); - let mut a = ServerEntry::new("a", "http://127.0.0.1:10000"); - a.node_id = Some(1); - a.rpc_url = Some("http://127.0.0.1:9921".into()); - a.rest_port = Some(10000); - a.rpc_port = Some(9921); - a.auto_start = true; - a.election_profile = Some("test".into()); - a.pid = Some(12345); - cfg.add_server(a).unwrap(); - cfg.add_server(ServerEntry::new("b", "http://127.0.0.1:10001")) - .unwrap(); - cfg.racks.push(super::RackEntry { - id: 1, - name: "rack-a".into(), - }); - cfg.nodes.push( - serde_json::from_value(serde_json::json!({ - "id": 1, - "rack_id": 1, - "host": "node.example", - "ssh_port": 2222, - "ssh_user": "operator", - "ssh_credential_ref": "node-1" - })) - .unwrap(), - ); - cfg.local_launches.insert( - "b".into(), - LocalLaunchSpec { - program: "/example/deploy/bin/crowdb-diskdb".into(), - args: vec!["--config".into(), "conf/server.toml".into()], - workdir: "/example/deploy".into(), - env: std::collections::BTreeMap::from([("LD_LIBRARY_PATH".into(), "/example/lib".into())]), - readiness_url: Some("http://127.0.0.1:10002".into()), - }, - ); - cfg.stores.push(StoreEntry { - store_id: 7, - nodes: vec![1, 2], - }); - cfg.groups.push(GroupEntry { - store_id: 7, - group_id: 70, - replicas: vec![ - ReplicaEntry { - replica_id: 700, - node_id: 1, - }, - ReplicaEntry { - replica_id: 701, - node_id: 2, - }, - ], - }); - - cfg.save(&path).unwrap(); - let loaded = ConsoleConfig::load(&path).unwrap(); - let expected = cfg.clone(); - assert_eq!(expected, loaded); - } - - #[test] - fn pid_is_persisted_to_disk() { - let dir = tempdir(); - let path = dir.join("console.toml"); - - let mut cfg = ConsoleConfig::default(); - let mut entry = ServerEntry::new("a", "http://127.0.0.1:10000"); - entry.pid = Some(4242); - cfg.add_server(entry).unwrap(); - - cfg.save(&path).unwrap(); - let raw = std::fs::read_to_string(&path).unwrap(); - assert!(raw.contains("pid = 4242"), "runtime pid must be persisted: {raw}"); - } - - #[test] - fn missing_file_yields_default() { - let dir = tempdir(); - let path = dir.join("nope.toml"); - let cfg = ConsoleConfig::load(&path).unwrap(); - assert!(cfg.servers.is_empty()); - } - - #[test] - fn toml_engine_round_trip() { - let dir = tempdir(); - let path = dir.join("engine.toml"); - let engine = TomlFileEngine::new(path.clone()); - - let mut cfg = ConsoleConfig::default(); - cfg.add_server(ServerEntry::new("a", "http://127.0.0.1:10000")) - .unwrap(); - - cfg.save_with_engine(&engine).unwrap(); - let loaded = ConsoleConfig::load_with_engine(&engine).unwrap(); - - assert_eq!(cfg, loaded); - } - - #[test] - fn default_path_points_to_persistent_runtime_namespace() { - let expected = crowdb_protocol::port::namespace::runtime_root() - .join("persistent") - .join("console") - .join("crowdb-kv.db.toml"); - assert_eq!(TomlFileEngine::default_path().unwrap(), expected); - assert_eq!(ConsoleConfig::default_path().unwrap(), expected); - } + use super::{ConsoleConfig, ServerEntry}; #[test] fn duplicate_id_rejected() { @@ -1338,22 +668,4 @@ mod tests { let err = cfg.remove_server("ghost").unwrap_err(); assert!(matches!(err, crate::error::Error::NotFound { .. })); } - - fn tempdir() -> std::path::PathBuf { - use std::sync::atomic::{AtomicU64, Ordering}; - static COUNTER: AtomicU64 = AtomicU64::new(0); - let base = test_dirs::test_data_dir(); - let unique = format!( - "crowdb-console-cfg-{}-{}-{}", - std::process::id(), - COUNTER.fetch_add(1, Ordering::Relaxed), - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_nanos() - ); - let dir = base.join(unique); - std::fs::create_dir_all(&dir).unwrap(); - dir - } } diff --git a/lib/crowdb-console-shared/src/lib.rs b/lib/crowdb-console-shared/src/lib.rs index 83c49681..f911c4a9 100644 --- a/lib/crowdb-console-shared/src/lib.rs +++ b/lib/crowdb-console-shared/src/lib.rs @@ -30,10 +30,7 @@ pub mod snapshot; pub mod ssh; pub mod topology; -pub use config::{ - ConsoleConfig, ConsoleConfigEngine, DiskEntry, DiskGroupEntry, NodeEntry, RackEntry, ServerEntry, - TomlFileEngine, -}; +pub use config::{ConsoleConfig, DiskEntry, DiskGroupEntry, NodeEntry, RackEntry, ServerEntry}; pub use snapshot::{ ClusterSnapshot, CrowdbTreeStatsSnapshot, ElectionStateSnapshot, GroupView, HealthInfo, KvStoreView, LocalReplicaView, MetricFieldView, MetricPointView, MetricsResponse, ReadStateSnapshot, RemoteMetrics, diff --git a/lib/crowdb-console-shared/src/ops.rs b/lib/crowdb-console-shared/src/ops.rs index 73df7c15..66e5d6af 100644 --- a/lib/crowdb-console-shared/src/ops.rs +++ b/lib/crowdb-console-shared/src/ops.rs @@ -5,8 +5,8 @@ //! //! Each submodule wraps a domain area (hardware, KV logical, KV server, //! KV data-plane, cluster, chunk/diskdb, bench) as free functions that -//! take an [`OpContext`] — the shared connection to group-0 sysdata + -//! the local TOML config. The CLI command handlers are thin wrappers +//! take an [`OpContext`] — the shared connection to Group 0 and +//! ephemeral bootstrap inputs. The CLI command handlers are thin wrappers //! that parse args, call these functions, and render the result. pub mod bench; diff --git a/lib/crowdb-console-shared/src/ops/cluster.rs b/lib/crowdb-console-shared/src/ops/cluster.rs index 97b24f61..70454b6c 100644 --- a/lib/crowdb-console-shared/src/ops/cluster.rs +++ b/lib/crowdb-console-shared/src/ops/cluster.rs @@ -29,12 +29,11 @@ use crate::ops::OpContext; mod bootstrap; pub use bootstrap::{init, init_with_intent, InitSummary}; +// Bootstrap runs before Group 0 registration exists, so its sealed intent +// supplies the initial management endpoint for each selected node. fn server_client(ctx: &OpContext, node_id: u64) -> Result { let url = ctx.node_mgmt_url(node_id)?; - ServerClient::new(&url).map_err(|e| Error::UpstreamRpc { - node_id: url, - status: format!("client build: {e}"), - }) + ServerClient::new(&url) } /// Get cluster status: list all stores from group-0 sysdata. @@ -55,111 +54,61 @@ pub async fn topology(ctx: &OpContext, node_id: u64) -> Result Result<()> { - let cfg = ctx.config().clone(); - - // Phase 1: remove all non-system groups while the KV management APIs are - // still reachable. - for server in &cfg.servers { - if server.service_type != crate::config::ServiceType::Kv { - continue; - } - if let Some(node_id) = server.node_id { - if let Ok(client) = server_client(ctx, node_id) { - if let Ok(stores) = client.topology().await { - for s in &stores { - if s.store_id == 0 { - continue; - } - let _ = client.remove_store(s.store_id).await; - } - } - } - } - } - - // Phase 2: clear sysdata, then remove group 0 last (best-effort). - let sysmd = ctx.sysmd(); - let stores = sysmd.list_stores().await.unwrap_or_default(); - for s in &stores { - let _ = sysmd.remove_store(s.store_id).await; + let stores = ctx.sysmd().list_stores().await?; + let system = stores + .iter() + .find(|store| store.store_id == 0) + .ok_or_else(|| Error::NotFound { + kind: "system store".into(), + id: "0".into(), + })?; + let system_nodes = system.node_ids.clone(); + let mut system_clients = Vec::with_capacity(system_nodes.len()); + for node_id in system_nodes { + let url = ctx.live_node_mgmt_url(node_id).await?; + system_clients.push(ServerClient::new(&url)?); } - for server in &cfg.servers { - if server.service_type != crate::config::ServiceType::Kv { - continue; - } - if let Some(node_id) = server.node_id { - if let Ok(client) = server_client(ctx, node_id) { - let _ = client.remove_group(0, 0).await; - } - } + if system_clients.is_empty() { + return Err(Error::Validation { + field: "system store".into(), + message: "Group 0 has no live hosts".into(), + }); } - - // Phase 3: stop all running services concurrently. A graceful stop may - // consume the full per-process timeout, so serial waits can exceed the - // CLI lifecycle bound and leave the persisted config pointing at dead - // processes. - let mut stop_handles = Vec::with_capacity(cfg.servers.len()); - for pid in cfg.servers.iter().filter_map(|server| server.pid) { - stop_handles.push(tokio::task::spawn_blocking(move || { - let _ = crate::lifecycle::stop_pid(pid); - })); + for store in stores.iter().filter(|store| store.store_id != 0) { + super::kv_logical::remove_store(ctx, store.store_id).await?; } - for handle in stop_handles { - let _ = handle.await; + for group in ctx.sysmd().list_groups_in_store(0).await? { + if group.group_id != 0 { + super::kv_logical::remove_group(ctx, 0, group.group_id).await?; + } } - - // Phase 4: clear local config. - { - let mut cfg = ctx.config_mut(); - cfg.stores.clear(); - cfg.groups.clear(); - cfg.servers.clear(); - cfg.local_launches.clear(); - cfg.disks.clear(); - cfg.disk_groups.clear(); - cfg.nodes.clear(); - cfg.racks.clear(); + // Keep the system group available until all other metadata is gone. + // Resolve every endpoint before removing any member. + for client in &system_clients { + client.remove_group(0, 0).await?; + client.remove_store(0).await?; } - Ok(()) } -/// Remove orphaned sysdata entries (stores/groups/replicas that have -/// no corresponding running server). Does not stop any running -/// servers. +/// Verify that every confirmed store host has one live registration. A +/// stopped or unreachable node is not evidence that its metadata is orphaned. /// /// # Errors -/// Returns an error if the sysdata scan fails. +/// Returns an error if any confirmed host cannot be verified. pub async fn reset(ctx: &OpContext) -> Result<()> { - let sysmd = ctx.sysmd(); - let stores = sysmd.list_stores().await?; - - // For each store, check if any hosting node has a running server. - let cfg = ctx.config().clone(); - for store in &stores { - let mut any_alive = false; - for node_id in &store.node_ids { - if cfg.server_for_node(*node_id).is_some() { - if let Ok(client) = server_client(ctx, *node_id) { - if client.health().await.is_ok() { - any_alive = true; - break; - } - } - } - } - if !any_alive { - let _ = sysmd.remove_store(store.store_id).await; + for store in ctx.sysmd().list_stores().await? { + for node_id in store.node_ids { + let url = ctx.live_node_mgmt_url(node_id).await?; + ServerClient::new(&url)?.health().await?; } } - Ok(()) } diff --git a/lib/crowdb-console-shared/src/ops/context.rs b/lib/crowdb-console-shared/src/ops/context.rs index 2757508e..e0b13dd8 100644 --- a/lib/crowdb-console-shared/src/ops/context.rs +++ b/lib/crowdb-console-shared/src/ops/context.rs @@ -17,9 +17,8 @@ use crate::error::{Error, Result}; /// (hardware hierarchy, KV-cluster topology, service registry). /// - **`kv`** — a [`CrowdbKvClient`] for the KV data-plane (put/get/ /// delete/scan on user stores/groups). -/// - **`config`** — the local TOML [`ConsoleConfig`] (rack/node/server -/// entries, bootstrap state). Mutated under an `RwLock` and persisted -/// by the caller via the engine. +/// - **`config`** — ephemeral [`ConsoleConfig`] inputs for bootstrap and +/// local benchmark deployment. Group 0 remains the topology authority. /// - **`discovery`** — an optional [`ServiceDiscoveryClient`] for /// discovering living service instances (diskdb, chunkdb, etc.) via /// the group-0 service registry. `None` when the caller (e.g. a unit diff --git a/lib/crowdb-console-shared/tests/common/bootstrap_node.rs b/lib/crowdb-console-shared/tests/common/bootstrap_node.rs index 1cf61088..82e406aa 100644 --- a/lib/crowdb-console-shared/tests/common/bootstrap_node.rs +++ b/lib/crowdb-console-shared/tests/common/bootstrap_node.rs @@ -72,7 +72,9 @@ pub fn context(first: &TestNode, second: &TestNode) -> OpContext { serde_json::from_value(json!({"id": (i + 1).to_string(), "node_id": i + 1, "url": url})).unwrap() }) .collect(); - let mut config = ConsoleConfig::default(); - config.servers = servers; + let config = ConsoleConfig { + servers, + ..Default::default() + }; OpContext::new("127.0.0.1:9".into(), Vec::new(), config) } From 5631718d5acb74c454c38165bbcb08b0bb80798f Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 16:00:55 +0800 Subject: [PATCH 70/74] Simplify unreleased console lifecycle --- app/crowdb-cli/src/commands/cluster.rs | 26 +- app/crowdb-cli/tests/direct_cli_test.rs | 16 - app/crowdb-web/src/lib.rs | 1 - app/crowdb-web/src/lifecycle.rs | 16 - app/crowdb-web/src/mgmt/group_ops.rs | 4 +- app/crowdb-web/src/mgmt/store_ops.rs | 4 +- app/crowdb-web/tests/rolling_upgrade_test.rs | 496 --------------- doc/design/console/design-crowdb-console.md | 627 +++++-------------- doc/working/plan-console-authority.md | 185 ++---- doc/working/test.md | 1 - lib/crowdb-console-shared/src/config.rs | 14 +- lib/crowdb-console-shared/src/ops/cluster.rs | 22 +- 12 files changed, 225 insertions(+), 1187 deletions(-) delete mode 100644 app/crowdb-web/tests/rolling_upgrade_test.rs diff --git a/app/crowdb-cli/src/commands/cluster.rs b/app/crowdb-cli/src/commands/cluster.rs index 24b3bdc6..c14caa9d 100644 --- a/app/crowdb-cli/src/commands/cluster.rs +++ b/app/crowdb-cli/src/commands/cluster.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! `cluster` domain — cluster-level ops: init, reset, clean, status, +//! `cluster` domain — cluster-level ops: init, destroy, clean, status, //! topology, plus hardware subcommands (rack/node/disk-group/disk). pub mod hardware; @@ -179,8 +179,6 @@ pub enum ClusterVerb { }, /// Tear down the entire cluster (all groups, stores, servers, sysdata). Destroy, - /// Verify confirmed store hosts have live registrations and healthy servers. - Reset, /// Wipe user data on every node + wait for re-election. Preserves /// group-0 sysdata + topology — servers stay running. Use --store/--group /// to target a non-system group (recommended for benchmarks). @@ -382,6 +380,9 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { summary.chunkdb_instances, summary.diskio_instances ); + if let Some(seed) = ctx.config().servers.first() { + println!("Group 0 management seed: {}", seed.url); + } ExitCode::SUCCESS } Err(error) => { @@ -436,6 +437,9 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { .collect::>() .join(", ") ); + if let Some(seed) = ctx.config().servers.first() { + println!("Group 0 management seed: {}", seed.url); + } ExitCode::SUCCESS } Err(e) => { @@ -597,22 +601,6 @@ pub async fn run_cluster_verb(cli: &Cli, verb: ClusterVerb) -> ExitCode { } } } - ClusterVerb::Reset => { - let ctx = match authority_context(cli).await { - Ok(c) => c, - Err(c) => return c, - }; - match crowdb_console_shared::ops::cluster::reset(&ctx).await { - Ok(()) => { - println!("cluster membership verified"); - ExitCode::SUCCESS - } - Err(e) => { - eprintln!("error: cluster reset: {e}"); - ExitCode::from(2) - } - } - } ClusterVerb::Clean { store, group, diff --git a/app/crowdb-cli/tests/direct_cli_test.rs b/app/crowdb-cli/tests/direct_cli_test.rs index 839898ea..5a33bce5 100644 --- a/app/crowdb-cli/tests/direct_cli_test.rs +++ b/app/crowdb-cli/tests/direct_cli_test.rs @@ -115,22 +115,6 @@ async fn incremental_local_deploy_reads_group_zero_instead_of_legacy_file() { assert!(!stderr.contains("load config"), "stderr={stderr}"); } -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn reset_verifies_confirmed_hosts_without_legacy_file() { - let Some(g0) = spawn_group0().await else { - return; - }; - std::fs::write(&g0.config_path, "invalid local topology").unwrap(); - let (code, stdout, stderr) = run( - &crowdb_cli_bin(), - g0.mgmt_port, - &g0.config_path, - &["cluster", "reset"], - ); - assert_eq!(code, 0, "stderr={stderr}"); - assert!(stdout.contains("membership verified"), "stdout={stdout}"); -} - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn destroy_reads_group_zero_and_requires_launch_registry() { let Some(g0) = spawn_group0().await else { diff --git a/app/crowdb-web/src/lib.rs b/app/crowdb-web/src/lib.rs index c0e40bab..3891e428 100644 --- a/app/crowdb-web/src/lib.rs +++ b/app/crowdb-web/src/lib.rs @@ -307,7 +307,6 @@ pub fn router(state: AppState) -> axum::Router { // ── Cluster init (R2): system group bootstrap ──────────────── .route("/api/cluster/init", post(mgmt::http_cluster_init)) .route("/api/cluster/destroy", post(lifecycle::http_internal_reset)) - .route("/api/cluster/reset", post(lifecycle::http_cluster_reset)) .route("/api/cluster/clean", post(lifecycle::http_cluster_clean)) // ── Internal: E2E test reset (alias for destroy) ───────────── .route("/internal/reset", post(lifecycle::http_internal_reset)) diff --git a/app/crowdb-web/src/lifecycle.rs b/app/crowdb-web/src/lifecycle.rs index 6180f635..6eeb1bbd 100644 --- a/app/crowdb-web/src/lifecycle.rs +++ b/app/crowdb-web/src/lifecycle.rs @@ -1047,22 +1047,6 @@ pub async fn http_cluster_clean( .map_err(|e| err_502(format!("{e}"))) } -/// `POST /api/cluster/reset`. Remove orphaned sysdata entries -/// (stores/groups/replicas that have no corresponding running server). -/// Does not stop any running servers. -/// -/// # Errors -/// Returns `502` if the sysdata scan fails. -pub async fn http_cluster_reset( - State(state): State, -) -> Result)> { - let ctx = state.op_context().await.map_err(|e| err_502(format!("{e}")))?; - ops::cluster::reset(&ctx) - .await - .map_err(|e| err_502(format!("{e}")))?; - Ok(StatusCode::NO_CONTENT) -} - /// `POST /api/cluster/destroy` (alias: `/internal/reset`). Tear down /// the entire cluster in dependency order: groups → stores → server /// processes → nodes → racks, then clear workspace dirs and caches. diff --git a/app/crowdb-web/src/mgmt/group_ops.rs b/app/crowdb-web/src/mgmt/group_ops.rs index ee73c4f4..7ac47607 100644 --- a/app/crowdb-web/src/mgmt/group_ops.rs +++ b/app/crowdb-web/src/mgmt/group_ops.rs @@ -243,7 +243,9 @@ pub(crate) async fn http_remove_group( return Err(( StatusCode::CONFLICT, Json(ErrorBody { - error: "group 0 in store 0 is the system group; use POST /api/cluster/reset to tear down the entire cluster".into(), + error: + "group 0 in store 0 is the system group; destroy and recreate the cluster to remove it" + .into(), }), )); } diff --git a/app/crowdb-web/src/mgmt/store_ops.rs b/app/crowdb-web/src/mgmt/store_ops.rs index 804574d9..0188f72a 100644 --- a/app/crowdb-web/src/mgmt/store_ops.rs +++ b/app/crowdb-web/src/mgmt/store_ops.rs @@ -202,9 +202,7 @@ pub(crate) async fn http_remove_store( return Err(( StatusCode::CONFLICT, Json(ErrorBody { - error: - "store 0 is the system store; use POST /api/cluster/reset to tear down the entire cluster" - .into(), + error: "store 0 is the system store; destroy and recreate the cluster to remove it".into(), }), )); } diff --git a/app/crowdb-web/tests/rolling_upgrade_test.rs b/app/crowdb-web/tests/rolling_upgrade_test.rs deleted file mode 100644 index c624d5cd..00000000 --- a/app/crowdb-web/tests/rolling_upgrade_test.rs +++ /dev/null @@ -1,496 +0,0 @@ -// Copyright 2026-present Gian -// Licensed under the Apache License, Version 2.0. - -//! M5 rolling-upgrade version-compat test: exercise a 3-node cluster with -//! two different `crowdb-kv-server` binary builds and verify a KV workload -//! does not diverge. -//! -//! By default the test uses the current `crowdb-kv-server` binary for all -//! nodes, copying it to a distinct path so the harness treats the second -//! node as a separate "version" build. To test a real version boundary, -//! set `CROWDB_KV_SERVER_BIN_V2` to a different binary (e.g. an older build). -//! -//! Uses the same real-subprocess harness as `replica_leader_removal_test.rs`. - -use std::collections::BTreeMap; -use std::net::SocketAddr; -use std::path::{Path, PathBuf}; -use std::time::{Duration, Instant}; - -use crowdb_console_shared::clients::http::ServerClient; -use crowdb_console_shared::cluster::NodeHealth; -use crowdb_console_shared::config::{NodeEntry, RackEntry, ServerEntry, ServiceType}; -use crowdb_console_shared::lifecycle::{self, crowdb_kv_server_bin, process_is_alive, DeployRequest}; -use crowdb_console_shared::monitor::{legacy_topology_to_node_stores, NodeRecord}; -use crowdb_console_shared::ConsoleConfig; -use crowdb_web::{router, AppState}; -use serde_json::json; - -fn pick_free_port() -> u16 { - crowdb_protocol::port::alloc::alloc_test_port(crowdb_protocol::ServicePort::Web) -} - -struct Upstream { - node_id: u64, - pid: u32, - mgmt_url: String, - rpc_url: String, - rest_port: u16, - rpc_port: u16, - binary: PathBuf, -} - -struct Cluster { - nodes: BTreeMap, - web: SocketAddr, - workspace: PathBuf, -} - -impl Cluster { - const fn sid() -> u64 { - 3 - } - - const fn gid() -> u64 { - 3 - } - - fn base_url(&self) -> String { - format!("http://{}", self.web) - } - - fn stop(&mut self) { - for n in self.nodes.values() { - let _ = lifecycle::stop_pid_with_timeout(n.pid, Duration::from_secs(5)); - } - } - - /// Restart a previously-killed node with the same ports, binary, and - /// workspace directory so it recovers from WAL and rejoins the group. - async fn restart_node(&mut self, node_id: u64) { - let u = self.nodes.get(&node_id).expect("node exists"); - let node = NodeEntry { - id: node_id, - rack_id: 1, - host: "127.0.0.1".into(), - ssh_port: 22, - ssh_user: String::new(), - ssh_key: None, - ssh_password: None, - ssh_credential_ref: None, - }; - let replica_id = node_id; - let extra_args = vec![ - "--stores".to_string(), - Cluster::sid().to_string(), - "--groups".to_string(), - Cluster::gid().to_string(), - "--replica".to_string(), - replica_id.to_string(), - ]; - let req = DeployRequest { - server_id: node_id.to_string(), - rest_port: u.rest_port, - rpc_port: u.rpc_port, - election_profile: Some("e2e".into()), - binary: Some(u.binary.clone()), - ..Default::default() - }; - let node_dir = self.workspace.join(node_id.to_string()); - let deployed = lifecycle::deploy_local_in_dir_with_extra_args(&req, &node, &node_dir, &extra_args) - .await - .expect("restart node"); - // Update the stored pid so stop() cleans up the new process. - self.nodes.get_mut(&node_id).unwrap().pid = deployed.pid; - } -} - -impl Drop for Cluster { - fn drop(&mut self) { - for n in self.nodes.values() { - let _ = lifecycle::stop_pid_with_timeout(n.pid, Duration::from_secs(5)); - } - } -} - -fn tempdir(tag: &str) -> PathBuf { - let base = crowdb_test_harness::test_dirs::ephemeral_root().join("web-e2e"); - let millis = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_millis(); - let dir = base.join(format!("{tag}-{millis}")); - std::fs::create_dir_all(&dir).unwrap(); - dir -} - -fn resolve_binary() -> Option { - let bin = crowdb_kv_server_bin()?; - if !bin.exists() { - return None; - } - Some(bin) -} - -fn resolve_second_binary() -> Option { - if let Ok(v2) = std::env::var("CROWDB_KV_SERVER_BIN_V2") { - let p = PathBuf::from(v2); - if p.exists() { - return Some(p); - } - } - None -} - -async fn spawn_upstream(node_id: u64, workspace: &std::path::Path, binary: &Path) -> Option { - let node = NodeEntry { - id: node_id, - rack_id: 1, - host: "127.0.0.1".into(), - ssh_port: 22, - ssh_user: String::new(), - ssh_key: None, - ssh_password: None, - ssh_credential_ref: None, - }; - let req = DeployRequest { - server_id: node_id.to_string(), - rest_port: pick_free_port(), - rpc_port: pick_free_port(), - election_profile: Some("e2e".into()), - binary: Some(binary.to_path_buf()), - ..Default::default() - }; - let node_dir = workspace.join(node_id.to_string()); - std::fs::create_dir_all(node_dir.join("bin")).unwrap(); - std::fs::create_dir_all(node_dir.join("log")).unwrap(); - let deployed = lifecycle::deploy_local_in_dir(&req, &node, &node_dir) - .await - .expect("deploy_local_in_dir"); - Some(Upstream { - node_id, - pid: deployed.pid, - mgmt_url: deployed.mgmt_url, - rpc_url: deployed.rpc_url, - rest_port: req.rest_port, - rpc_port: req.rpc_port, - binary: binary.to_path_buf(), - }) -} - -async fn spawn_web(upstreams: &BTreeMap) -> SocketAddr { - let listener = tokio::net::TcpListener::bind(SocketAddr::from(([127, 0, 0, 1], 0))) - .await - .unwrap(); - let addr = listener.local_addr().unwrap(); - let mut cfg = ConsoleConfig::default(); - cfg.racks.push(RackEntry { - id: 1, - name: "r1".into(), - }); - for u in upstreams.values() { - cfg.nodes.push(NodeEntry { - id: u.node_id, - rack_id: 1, - host: "127.0.0.1".into(), - ssh_port: 22, - ssh_user: String::new(), - ssh_key: None, - ssh_password: None, - ssh_credential_ref: None, - }); - cfg.add_server(ServerEntry { - id: u.node_id.to_string(), - url: u.mgmt_url.clone(), - node_id: Some(u.node_id), - rpc_url: Some(u.rpc_url.clone()), - rest_port: None, - rpc_port: None, - auto_start: true, - binary: None, - election_profile: None, - pid: Some(u.pid), - service_type: ServiceType::Kv, - rpc_workers: None, - no_fsync: false, - }) - .unwrap(); - } - let state = AppState::with_config(cfg, None); - // Register each upstream's pid so `refresh_node_cache` (which - // skips nodes with no tracked runtime pid) refreshes after - // mutations. - for u in upstreams.values() { - state.set_runtime_pid(u.node_id, u.pid); - } - - for u in upstreams.values() { - let client = ServerClient::new(u.mgmt_url.clone()).unwrap(); - if let Ok(stores) = client.topology().await { - let rec = NodeRecord { - health: NodeHealth::Up, - last_seen_ms: 1, - stores: legacy_topology_to_node_stores(u.node_id, &stores), - last_error: None, - recovering: false, - }; - state.monitor_cache.set_node_report(u.node_id, rec).await; - } - } - - tokio::spawn(async move { - axum::serve(listener, router(state)).await.unwrap(); - }); - tokio::time::sleep(Duration::from_millis(50)).await; - addr -} - -async fn spawn_mixed_cluster(workspace: &std::path::Path, v2_binary: &PathBuf) -> Option { - let current = resolve_binary()?; - let mut nodes: BTreeMap = BTreeMap::new(); - for (id, binary) in [(1u64, ¤t), (2u64, ¤t), (3u64, v2_binary)] { - let Some(node) = spawn_upstream(id, workspace, binary).await else { - for n in nodes.into_values() { - let _ = lifecycle::stop_pid_with_timeout(n.pid, Duration::from_secs(5)); - } - eprintln!("skipping: crowdb-kv-server binary not built"); - return None; - }; - nodes.insert(id, node); - } - let web = spawn_web(&nodes).await; - Some(Cluster { - nodes, - web, - workspace: workspace.to_path_buf(), - }) -} - -async fn create_three_node_group(cluster: &Cluster) { - let base = cluster.base_url(); - let http = reqwest::Client::new(); - let sid = Cluster::sid(); - let gid = Cluster::gid(); - - // Initialize the system group so non-zero stores can be created. - let resp = http - .post(format!("{base}/api/cluster/init")) - .json(&json!({"nodes": [1, 2, 3]})) - .send() - .await - .unwrap(); - assert_eq!(resp.status(), 201, "cluster init: {:?}", resp.text().await.ok()); - - let resp = http - .post(format!("{base}/api/stores")) - .json(&json!({"store_id": sid, "nodes": [1]})) - .send() - .await - .unwrap(); - assert_eq!(resp.status(), 201, "create store: {:?}", resp.text().await.ok()); - - let resp = http - .post(format!("{base}/api/stores/{sid}/groups")) - .json(&json!({"group_id": gid, "replica_id": 1, "nodes": [1]})) - .send() - .await - .unwrap(); - assert_eq!(resp.status(), 201, "create group: {:?}", resp.text().await.ok()); - - for node_id in 2u64..=3 { - let resp = http - .post(format!("{base}/api/stores/{sid}/groups/{gid}/replicas")) - .json(&json!({"node_id": node_id, "replica_id": node_id})) - .send() - .await - .unwrap(); - assert_eq!( - resp.status(), - 201, - "add replica {node_id}: {:?}", - resp.text().await.ok() - ); - } -} - -async fn wait_for_leader( - cluster: &Cluster, - timeout: Duration, - exclude_node: Option, -) -> Option<(u64, u64)> { - let base = cluster.base_url(); - let http = reqwest::Client::new(); - let sid = Cluster::sid(); - let gid = Cluster::gid(); - let deadline = Instant::now() + timeout; - while Instant::now() < deadline { - let group: serde_json::Value = http - .get(format!("{base}/api/stores/{sid}/groups/{gid}")) - .send() - .await - .ok()? - .json() - .await - .ok()?; - if let Some(leader) = group["replicas"].as_array().unwrap_or(&vec![]).iter().find(|r| { - r["role"] == "leader" && exclude_node.map_or(true, |ex| r["node_id"].as_u64() != Some(ex)) - }) { - let rid = leader["replica_id"].as_u64()?; - let node_id = leader["node_id"].as_u64()?; - return Some((rid, node_id)); - } - tokio::time::sleep(Duration::from_millis(100)).await; - } - None -} - -async fn put_and_get(base: &str, http: &reqwest::Client, sid: u64, gid: u64, key: &str, value: &str) { - let put_resp = http - .post(format!("{base}/api/stores/{sid}/groups/{gid}/kv/put")) - .json(&json!({"key": key, "value": value})) - .send() - .await - .unwrap(); - assert_eq!( - put_resp.status(), - 200, - "kv put {key}: {:?}", - put_resp.text().await.ok() - ); - let put_body: serde_json::Value = put_resp.json().await.unwrap(); - assert_eq!(put_body["ok"], true, "kv put {key}: {put_body}"); - - let get_resp = http - .get(format!("{base}/api/stores/{sid}/groups/{gid}/kv/get?key={key}")) - .send() - .await - .unwrap(); - assert_eq!( - get_resp.status(), - 200, - "kv get {key}: {:?}", - get_resp.text().await.ok() - ); - let get_body: serde_json::Value = get_resp.json().await.unwrap(); - assert_eq!(get_body["value_utf8"], value, "kv get {key}: {get_body}"); -} - -#[tokio::test] -#[allow(clippy::too_many_lines)] -async fn mixed_version_3_node_cluster_kv_no_divergence() { - let Some(current) = resolve_binary() else { - eprintln!("skipping: crowdb-kv-server binary not built"); - return; - }; - - // If a second binary is not explicitly provided, make a copy of the - // current binary so the test still validates the mixed-build harness. - // For a real version boundary, set CROWDB_KV_SERVER_BIN_V2 to an old build. - let v2_binary = resolve_second_binary().unwrap_or_else(|| { - let copy = current.parent().unwrap().join("crowdb-kv-server-v2"); - let _ = std::fs::remove_file(©); - std::fs::copy(¤t, ©).expect("copy current binary as v2"); - copy - }); - - let workspace = tempdir("rolling_upgrade"); - let Some(mut cluster) = spawn_mixed_cluster(&workspace, &v2_binary).await else { - return; - }; - - create_three_node_group(&cluster).await; - - let _leader = wait_for_leader(&cluster, Duration::from_secs(10), None) - .await - .expect("leader should be elected in 3-node group"); - - let base = cluster.base_url(); - let http = reqwest::Client::new(); - let sid = Cluster::sid(); - let gid = Cluster::gid(); - - // Serve a small KV workload and verify each key round-trips. - for i in 0..20 { - let key = format!("rk-{i:02}"); - let value = format!("rv-{i:02}"); - put_and_get(&base, &http, sid, gid, &key, &value).await; - } - - // Restart each node one at a time (rolling upgrade) and verify the - // workload continues after each restart. - for node_id in 1u64..=3 { - let node = &cluster.nodes[&node_id]; - let pid = node.pid; - let status = std::process::Command::new("kill") - .arg("-TERM") - .arg(pid.to_string()) - .status() - .expect("terminate node"); - assert!(status.success()); - - // Wait for graceful shutdown to complete. The server's per-layer - // shutdown timeout is 10s (ServerConfig::DEFAULT.shutdown_timeout_ms) - // and shutdown performs a real fsync of the WAL + engine snapshot, - // which can take several seconds under CI disk contention. The poll - // returns as soon as the process exits (normally ~200ms); the 15s - // cap is a safety net above the server's own 10s/layer budget. - let dead = Instant::now() + Duration::from_secs(15); - while process_is_alive(pid) && Instant::now() < dead { - tokio::time::sleep(Duration::from_millis(50)).await; - } - assert!(!process_is_alive(pid), "node {node_id} should be stopped"); - - // Wait for a new leader to be elected before attempting reads. - let _new_leader = wait_for_leader(&cluster, Duration::from_secs(10), Some(node_id)) - .await - .expect("survivors should elect a new leader after node stop"); - - // The remaining nodes should still serve reads. - for i in 0..5 { - let key = format!("rk-{i:02}"); - let deadline = Instant::now() + Duration::from_secs(5); - let get_body = loop { - let get_resp = http - .get(format!("{base}/api/stores/{sid}/groups/{gid}/kv/get?key={key}")) - .send() - .await - .unwrap(); - if get_resp.status().is_success() { - break get_resp.json::().await.unwrap(); - } - if Instant::now() >= deadline { - let status = get_resp.status(); - let body = get_resp.text().await.unwrap_or_default(); - panic!("read during rolling restart failed for {key}: status {status}, body: {body}"); - } - tokio::time::sleep(Duration::from_millis(100)).await; - }; - let expected = format!("rv-{i:02}"); - assert_eq!( - get_body["value_utf8"], expected, - "value diverged for {key}: {get_body}" - ); - } - - // Restart the killed node so it recovers from WAL and rejoins the - // group, restoring full quorum before the next rolling step. - cluster.restart_node(node_id).await; - - // Wait for the restarted node to come back up. - let ready = Instant::now() + Duration::from_secs(5); - while Instant::now() < ready { - let client = ServerClient::new(cluster.nodes[&node_id].mgmt_url.clone()).unwrap(); - if client.health().await.is_ok() { - break; - } - tokio::time::sleep(Duration::from_millis(100)).await; - } - - // Wait for leader to stabilize after the node rejoins. - let _leader_after_restart = wait_for_leader(&cluster, Duration::from_secs(10), None) - .await - .expect("leader should stabilize after node restart"); - } - - cluster.stop(); -} diff --git a/doc/design/console/design-crowdb-console.md b/doc/design/console/design-crowdb-console.md index 3f927d88..4aaaf954 100644 --- a/doc/design/console/design-crowdb-console.md +++ b/doc/design/console/design-crowdb-console.md @@ -21,10 +21,10 @@ design is detailed in the sub-design `design-crowdb-console-ui.md`. - [3.2 Logical (usage) view](#32-logical-usage-view) - [3.3 Source of truth and freshness](#33-source-of-truth-and-freshness) - [3.4 Design decisions](#34-design-decisions) -- [4. Console Backend Persistence and Monitor Task](#4-console-backend-persistence-and-monitor-task) - - [4.1 Persisted state (config file)](#41-persisted-state-config-file) - - [4.2 Monitor task](#42-monitor-task) - - [4.3 Persistent Cluster Config](#43-persistent-cluster-config) +- [4. Console Configuration and Authority](#4-console-configuration-and-authority) + - [4.1 Separated local configuration](#41-separated-local-configuration) + - [4.2 Runtime observation](#42-runtime-observation) + - [4.3 Group 0 authority](#43-group-0-authority) - [4.4 Local runtime namespace](#44-local-runtime-namespace) - [5. Node Access Model](#5-node-access-model) - [5.1 Two transports per node](#51-two-transports-per-node) @@ -32,7 +32,7 @@ design is detailed in the sub-design `design-crowdb-console-ui.md`. - [5.3 Process lifecycle (deploy / start / stop)](#53-process-lifecycle-deploy--start--stop) - [6. Web UI Backend (Axum)](#6-web-ui-backend-axum) - [6.1 Design Rules](#61-design-rules) - - [6.2 Recursive reads (?recursive=)](#62-recursive-reads-recursivedepth) + - [6.2 In-process test API](#62-in-process-test-api) - [6.3 Orchestration semantics](#63-orchestration-semantics) - [6.4 Resolution rules](#64-resolution-rules) - [6.5 Frontend contract](#65-frontend-contract) @@ -47,9 +47,8 @@ design is detailed in the sub-design `design-crowdb-console-ui.md`. - [7.8 S3 mini-clusters and benchmarks](#78-s3-mini-clusters-and-benchmarks) - [8. Error Model and Operation Logging](#8-error-model-and-operation-logging) - [9. Observability](#9-observability) -- [10. Open Questions](#10-open-questions) -- [11. Sysdata sync — rack/node/disk-group/disk handlers](#11-sysdata-sync--racknodedisk-groupdisk-handlers) -- [12. Cluster reset](#12-cluster-reset) +- [10. Hardware mutations](#10-hardware-mutations) +- [11. Cluster teardown and verification](#11-cluster-teardown-and-verification) ## 1. Goals and Non-Goals @@ -60,8 +59,8 @@ design is detailed in the sub-design `design-crowdb-console-ui.md`. ### Non-Goals - Bypassing `crowdb-kv-server` to talk to Paxos / WAL / storage internals. -- Authentication, authorization, multi-tenancy, audit logging. -- Persisting console state beyond local config files. +- Multi-tenancy and a general audit-log service. +- Making a console-local file authoritative for cluster topology. Local deployments use one stable directory per logical server below their runtime namespace. Each server owns its `data/`, `config/`, `log/`, and @@ -102,15 +101,12 @@ path through the shared `ops` module: - **Web**: `user → crowdb-web (Axum) → shared (ops module) → group-0 sysdata + crowdb-kv-server` - **CLI**: `user → crowdb-cli → shared (ops module) → group-0 sysdata + crowdb-kv-server mgmt` -Both frontends build an `OpContext` and call the same `ops::*` -functions. The CLI builds one per invocation from `--system-ip` / -`--system-port`; either endpoint may name any system-group node because the -client discovers the leader. The web backend builds one per request via -`AppState::op_context()`, sharing the cached `CrowdbKvClient` -(topology cache + connection pool) and snapshotting the persisted -`ConsoleConfig`. Mutations inside `ops::*` update the per-request -`OpContext` config snapshot; the handler writes the mutated config -back to `AppState.config` + persists to TOML after `ops::*` returns. +Both frontends build an `OpContext` with Group 0 discovery seeds. The CLI +uses `--system-ip` and `--system-port`; production Web uses a versioned process +configuration. Both read confirmed hardware and logical records through Group 0 +and resolve node management endpoints from live registration. The process +launch registry is local to each bare-metal console. Docker process state is +owned by `crowdb-monitor`. The CLI talks directly to group-0 system metadata via `CrowdbSysmdClient` and to individual `crowdb-kv-server` management @@ -129,8 +125,8 @@ call. ┌───────────────┐ │ shared │ (business logic: │ (lib crate) │ ops module, - └──────┬────────┘ monitor cache, - │ leader resolution, + └──────┬────────┘ leader discovery, + │ registry controls, ┌──────────────┼──────────────┐ SSH session pool) ▼ ▼ ▼ HTTP crowdb-rpc SSH @@ -151,9 +147,10 @@ call. - Both frontends build an `OpContext` and call `shared`'s `ops` module directly — the CLI from `--system-ip` / `--system-port` global flags, the web backend via `AppState::op_context()` (sharing the - cached `CrowdbKvClient` + snapshotting `ConsoleConfig`). -- Both frontends share the same `shared` entry points, so any feature - is reachable from both surfaces by construction. + cached `CrowdbKvClient` and Group 0 management seeds). +- Shared hardware and logical operations provide the same authority and + conditional publication rules to both frontends. Deployment-mode policy + controls which process and hardware mutations Web exposes. ## 3. Data Model @@ -184,23 +181,25 @@ Identity is the parent chain Rooted at **Cluster → Store → Group → Replica…** with a unified replica list (no local/remote split; each replica carries a `node_id`). This is the view that KV traffic, leader resolution, and routine cluster -operations use. The web backend is the only component that needs to -translate logical ids into upstream `(node_id, mgmt_url, rpc_url)` -tuples; the SPA and the CLI never see those. +operations use. Shared operations translate logical IDs into confirmed +membership and live endpoints for both frontends. Identity is `(store_id[, group_id[, replica_id]])`. ### 3.3 Source of truth and freshness -- **Persisted (config file, see §4):** rack/node entries and the - *intended* server deployment record (host, ports, binary path). - These survive restart. -- **Live (rebuilt on every console start):** process state, health, - per-node store/group/replica state, leader hints. The monitor task - (§4) pings each node and fetches per-node state; the logical view is - derived by aggregating those reports. -- **No `ClusterSnapshot` polling endpoint.** The SPA queries - per-resource live endpoints, all served from the monitor cache. +- **Group 0:** rack and node identity, nonsecret SSH connection settings and + credential references, disk hierarchy, bindings, KV stores, groups, + replicas, and service registration. +- **Local process inputs:** a versioned Web process configuration, bare-metal + launch registry, and per-console secret store. Neither topology nor inline + SSH secrets are accepted in the launch registry. +- **Live state:** management health, process identity, and current endpoints. + Docker reads monitor-owned process state; bare-metal process identity is + checked by `LaunchRuntime`. A stopped process is not a live registration. +- **Unavailable authority:** missing or ambiguous registration and Group 0 + outages are reported as unavailable. The monitor and launch registry do not + supply fallback topology. ### 3.4 Design decisions @@ -216,121 +215,49 @@ Identity is `(store_id[, group_id[, replica_id]])`. where possible; the console-side wrapper adds the `node_id` projection that the per-server protocol does not encode. -## 4. Console Backend Persistence and Monitor Task - -### 4.1 Persisted state (config file) - -- Single internal TOML file: - `.crowdb-runtime/persistent/console/crowdb-kv.db.toml`. It is CLI state, not - a user-facing command option. -- Contents: - - `rack` / `node` entries (id, rack_id, host, SSH creds). - - Optional per-node server deployment record: management endpoint, - rpc endpoint, and binary/config path as implementation evolves. - This records the operator's intended deployment target, not - authoritative live state. -- **Plaintext** SSH credentials are acceptable for v1 (internal demo); - a single `ConsoleConfig` struct is the only place that reads / - writes the file, so a future move to OS keychain or libsodium - sealed-box does not touch any caller. -- **Never persisted:** live process state, per-node store/group/ - replica state, leader hints, health flags. These are rebuilt on - every console start. - -### 4.2 Monitor task - -On startup, after loading the rack/node table, `shared` spawns a -long-running monitor task that owns the live cache: - -1. **Ping loop** — every `monitor.ping_interval` (default 2 s), the - task probes each node's `/health` over HTTP (and SSH liveness on - demand for the lifecycle API). It updates `NodeHealth` and - `ProcState` in the cache. -2. **Monitor refresh** — for every node observed `Up`, the task - calls the server's topology-report API to fetch `NodeStore` / - `NodeGroup` data (per-node store, group, local replica, remote - list). The aggregated `StoreView` / `GroupView` / `ReplicaView` - needed by the logical API are derived from these per-node reports. -3. **Event-driven refresh** — every successful mutation through - `shared` (deploy, store create, group create, replica add/remove) - triggers an immediate refresh for the affected nodes so the next - read reflects the change without waiting for the next ping tick. -4. **Cache reads are non-blocking.** API handlers read the most - recent cached value; they do not issue an upstream RPC per - request. A handler that needs a stronger guarantee ("force fresh") - can request an inline refresh, but that is the exception. - -### 4.3 Persistent Cluster Config - -**Problem**: The TOML config file is a single point of failure. Losing -the console host loses the full topology. Per-node server config is also -not persisted independently; a node restart relies on the console to -re-push topology. - -**Solution**: A designated Paxos group, **system group (store 0, -group 0)**, stores the full cluster topology as regular KV entries. -Since it is a Paxos group, the topology is replicated and HA by the -same mechanism that protects user data. No external coordinator -needed. This is the standard industry pattern (closest -analog: CockroachDB system ranges). - -- **Two-phase bootstrap**: - - Phase 1: Console TOML is source of truth (existing behavior). - - Phase 2: `HardwareClient` writes hardware hierarchy (racks, nodes) - and `KVClusterMetaClient` writes KV-cluster topology (stores, - groups, replicas) into group 0 via text-path keys with JSON - values. No readiness flag. diskdb's sync loop treats empty group 0 - as "nothing assigned yet" and retries. - - Console restart: two-way fallback. Group 0 missing → TOML mode; - group 0 exists → group 0 authoritative. - -The TOML file remains available for the whole local-deployment lifecycle. -Group-0 initialization does not make it disposable: subsequent CLI processes -use it to find endpoints and tracked process IDs for status, clean, restart, -and destroy operations. Regression runs keep `console.toml` at the retained -run root after teardown as diagnostic state; it is not stored inside a single -command's invocation directory. - -- **Group-0 sysdata schema** (text-path keys, JSON values): - - `/hw/rack/` — rack metadata (`RackValue`) - - `/hw/node//` — node metadata (`NodeValue`) - - `/hw/dg///` — disk-group metadata - - `/hw/disk////` — disk metadata - - `/hw/owner///` — ownership map - - `/hw/bind///` — bind map - - `/kv/store/` — store metadata (`StoreValue`) - - `/kv/group//` — group metadata (`GroupValue`) - - `/kv/replica///` — replica metadata - - `/srv//` — service registry instances - -- **Per-node config cache** (`conf/node-config.json`): Local cache - derived from the system group. On startup: load cache → create - stores/groups → replay WAL → reconcile with group 0 KV. If cache is - lost, node queries group 0 to rebuild it. - -- **Divergence reconciliation**: On node startup, if group 0 is - reachable and finalized, compare local cache against group 0 KV. - Create missing stores/groups, remove stale ones. If group 0 not - reachable, boot from local cache only (deferred). - -- **Cluster init flow**: `POST /api/cluster/init` on the console - orchestrates: calls `POST /system/init` on selected nodes, wires - remotes for multi-node, persists topology in console config, then - writes hardware + KV-cluster topology into group 0 via - `HardwareClient` + `KVClusterMetaClient`. Data store/group creation - is blocked (`409`) until cluster is initialized. - -- **Management API endpoints** (on `crowdb-kv-server`, internal — only - called by `crowdb-kv-client`'s `KVClusterAdmin`): - - `POST /system/init` — bootstrap store 0 + group 0 on this node - - Lifecycle: `add_store`, `remove_store`, `add_group`, - `remove_group`, `add_remote_replicas`, `remove_remote_replica`, - `step_down`, `join_group_via_snapshot`, `flush_group` - - Query: `GET /topology` (export), `GET /health`, `GET /metrics` - -- **Group 0 membership evolution**: Reuses shipped Model B - reconfiguration (direct HTTP mutation + `membership_epoch` fence). - No new consensus primitive required. +## 4. Console Configuration and Authority + +### 4.1 Separated local configuration + +`WebProcessConfig` contains the listener, Group 0 management seeds, UI and log +paths, deployment mode, and (for Docker) the monitor status path. Production +Web requires this versioned input. Docker rejects a launch registry. + +`LaunchRegistry` contains bare-metal process policy: service, node, host, +binary, service config, workspace and auto-start setting. Runtime PID and +start-time identity are retained separately by `LaunchRuntime`. SSH credential +reference IDs are read from Group 0 and resolved against each console's local +secret store. Group 0 never contains private keys, passwords, PIDs, images or +container IDs. + +The `ConsoleConfig` struct is an ephemeral operation input for bootstrap and +local development. It has no file parser or writer. A sealed `BootstrapIntent` +retains pre-Group-0 identity across interruption and is deleted only after all +committed records are verified. + +### 4.2 Runtime observation + +The production `/api/preview` snapshot reads Group 0 hardware and logical +records and validates live service registrations. Docker overlays monitor +process status, while bare-metal Web uses its local launch runtime. Missing +monitor status makes Docker runtime observation unavailable. The in-process +Web test router keeps a monitor cache for fixture orchestration; that cache is +not production topology authority. + +### 4.3 Group 0 authority + +System group (store 0, group 0) replicates hardware and KV-cluster metadata. +Bootstrap initializes selected KV members, wires peers and conditionally +publishes the rack, node, store, group and replica records. A retry compares +sealed identity and already committed content, writes only missing records, +and rejects conflicting content. Nonmember KV processes receive Group 0 seeds +and must register exactly one live identity before logical operations use them. + +The metadata namespaces are `/hw/rack`, `/hw/node`, `/hw/dg`, `/hw/disk`, +`/hw/owner`, `/hw/bind`, `/kv/store`, `/kv/group`, `/kv/replica`, and `/srv`. +Logical mutations confirm all node-side steps before publishing membership; +conditional writes and confirmed reads reconcile a lost response. A local +launch or monitor record never substitutes for a missing Group 0 result. ### 4.4 Local runtime namespace @@ -370,117 +297,44 @@ namespaces. - Default host: `127.0.0.1` with the current OS user. - Pre-flight: every operation calls `ssh::probe(node)` which performs a real handshake before any side-effecting work. Failure surfaces as `NodeUnreachable { node_id, reason }`. -**SSH credential storage lifecycle** — two phases: - -- **Bootstrap phase** (before group 0 exists) — SSH creds are stored - in the shared TOML config file below - `.crowdb-runtime/persistent/console/` - (via `TomlFileEngine::default_path()` in - `lib/crowdb-console-shared/src/config.rs`). This file stores - rack/node/server/store/group/disk-group/disk entries, with SSH creds - in `NodeEntry` (`ssh_user`, `ssh_key`, `ssh_password`). The CLI and - UI share the same `ConsoleConfig` + `TomlFileEngine` flow — `cluster - rack add` / `cluster node add` write to this file, `kv server deploy` - reads SSH creds from it. No separate CLI-only config file. -- **Steady-state phase** (after group 0 exists) — SSH creds are moved - into group-0 sysdata, encrypted with a default key. Subsequent `kv - server deploy` calls read creds from group-0 sysdata via - `KVClusterMetaClient`. The TOML file is no longer the source of truth - for SSH creds; group 0 is. The TOML file remains as a local cache / - bootstrap fallback. - +**SSH credential boundary:** Group 0 stores only the SSH user, port and +credential reference associated with a node. Each bare-metal console resolves +the reference in its own local secret store. Bootstrap intent rejects inline +private keys and passwords; the launch registry accepts references only. ### 5.3 Process lifecycle (deploy / start / stop) -**SSH path** (`ssh_user` non-empty): -1. SSH into node (`russh` crate, pure Rust async). -2. `nohup crowdb-kv-server --management-addr 127.0.0.1 --management-port

--ports &`; - capture pid via `echo $!`; record in the persisted node server entry. -3. Health-check via the new server's HTTP `/health` until ready or timeout (10 s). +`LaunchRuntime` uses the validated launch registry for local or SSH process +start, restart, stop and readiness checks. It records PID plus process start +time as local runtime identity and refuses to signal an unrelated process. +Auto-start policy is reconciled on Web startup and reload. A successful process +launch is not a substitute for a Group 0 service registration. Docker delegates +child recovery and status to `crowdb-monitor`. -**Local-fork path** (`ssh_user` empty, for tests/dev on `127.0.0.1`): -1. `tokio::process::Command::new(crowdb-kv-server)` with the same args. -2. Stage the binary into the node's stable service directory in the runtime - namespace. -3. Detach the child (do not kill on drop); track the pid. -4. Health-check via `/health`. - -Binary resolution: `$CROWDB_KV_SERVER_BIN` → sibling of current executable → -`$PATH` lookup for `crowdb-kv-server`. +## 6. Web UI Backend (Axum) -(Future: scp the binary to the remote host on first deploy and render -a config template. Not yet implemented. The SSH path assumes the -binary is already present on the remote host.) +### 6.1 Design Rules -`server deploy`, `server restart`, and `server stop` address a node. There is no separate -server id namespace in the console API. +Production Web uses the managed router and a versioned process configuration. +`/api/preview` combines confirmed Group 0 records, live registration, and the +mode-specific process view. `/api/stores/...` provides authenticated logical +mutations through shared operations in both modes. Bare-metal Web additionally +exposes rack, node, disk-group and disk reads and authenticated mutations, plus +registry-backed launch controls and bootstrap. Docker Web does not expose +hardware or process mutation routes. Unknown managed API routes report +unavailable rather than entering an in-memory topology path. -## 6. Web UI Backend (Axum) +A mutation is accepted only after the required node-side steps and Group 0 +publication are confirmed. Authenticated management routes use a bearer token. +The SPA calls the Axum backend; it does not talk directly to KV management +endpoints. -### 6.1 Design Rules +### 6.2 In-process test API -The console-facing API is split along the **two hierarchy views** -defined in §3, and every route lives under exactly one of them. Every -handler builds an `OpContext` via `AppState::op_context()` and -delegates to the matching `ops::*` function — the web backend no -longer hand-rolls orchestration logic (fan-out, rollback, sysdata -sync). The `ops` module owns all multi-step logic; the handler only -parses input, calls `ops::*`, writes back config, and renders output. - -**R1. Two URL trees, one per hierarchy.** -- `/api/racks/...` and `/api/nodes/...` form the **physical** tree. - Every resource is addressed by its parent chain. -- `/api/stores/...` forms the **logical** tree. KV traffic and - cluster-wide operations live here, addressed by - `(store_id[, group_id[, replica_id]])`. Logical-tree responses still - carry `node_id` on every entry so a caller can see placement without - a physical-tree query; only the **path** is node-free. -- A route never crosses trees. - -**R2. Logical reads aggregate; physical reads are per-node.** -The same store, observed through the two trees, returns different -shapes: aggregated `StoreView` vs. that node's local `NodeStore`. -This is how the operator inspects "is the cluster consistent?" vs. -"what does this one node think it has?". - -**R3. Logical writes orchestrate; physical writes act on one node.** -A logical write declares *intent*; the `ops` function fans out -per-node calls and rolls back on partial failure. A physical write -is the low-level primitive. It touches exactly that node, never fans -out. Logical writes are implemented on top of physical primitives. - -**R4. No `server_id` namespace.** -Process lifecycle and reachability probes use -`/api/nodes/:node_id/server/...`. Node identity *is* server identity. - -**R5. `OpContext` per request.** -Each handler builds an `OpContext` from `AppState::op_context()`, -which shares the cached `Arc` (topology cache + -connection pool) and snapshots the persisted `ConsoleConfig`. After -`ops::*` returns, the handler writes the mutated config back to -`AppState.config` (short write-lock, no `await` inside) and persists -via the config engine. On error, the snapshot is discarded — -`AppState.config` is unchanged. - -> **Retired contracts (no compatibility shim):** `?server=` -> query parameter, `/api/servers/:sid/...`, -> `/api/openapi.json?server=`, `/api/cluster/snapshot`, -> `/api/swagger/...`, `/api/nodes/:id/openapi.json`. - -The full endpoint list is defined in the Axum route handlers and the -OpenAPI spec; this section covers design rules only. - -### 6.2 Recursive reads (`?recursive=`) - -Any `GET` in either tree accepts `?recursive=` to inline up to `n` -child levels in one response, avoiding O(N) follow-up requests for -UIs that render a whole sub-tree. `recursive=all` is a capped alias -(default max depth 8) intended for the SPA's initial render. - -Rules: read-only (mutations ignore it), depth counts child hops from -the addressed resource, each tree expands along its own hierarchy, KV -key/value payloads are never inlined, and all responses use the -monitor cache so `recursive` is cheap even at high depth. +The in-process Web router and `--test-mode` retain fixture orchestration for +browser and integration tests. Their recursive physical views and monitor cache +help exercise the UI, but are never selected by a production Web process. +They do not persist a topology file or provide a fallback for managed requests. ### 6.3 Orchestration semantics @@ -506,8 +360,8 @@ these rules: parents; an already absent node-side object permits retry. - **Idempotent retries.** A repeat of the same logical request must converge to the same state. -- **Cache refresh on success.** Every successful mutation triggers an - immediate monitor refresh for the affected nodes. +- **Read after write.** A mutation returns only after the required node-side + and Group 0 confirmation steps complete. ### 6.4 Resolution rules @@ -525,8 +379,8 @@ backend-facing contract here: - Bundle output is `app/crowdb-web/ui/dist/`; `crowdb-web` serves it via SPA fallback. -- The SPA polls per-resource live endpoints on a short interval. No - WebSocket/SSE. All reads are served from the monitor cache. +- The SPA polls the management API on a short interval. An unavailable + authority clears stale logical rows and is shown explicitly. - No `/api/cluster/snapshot` aggregate endpoint. ## 7. CLI Design @@ -557,108 +411,25 @@ this section covers design rules only. ### 7.1 Four-Domain Hierarchy -The CLI is split by service domain into four top-level groups, each -cohesive and focused: - -- **`cluster`** (alias `cls`) — hardware topology (rack, node, - disk-group, disk, including runtime hardware state via - `set-status`) + cluster-level ops (init, reset, clean, status, - topology). `disk-group` and `disk` live here, not under `chunk`, - because they are hardware topology concepts — physical disks grouped - into disk-groups on nodes in racks. The `set-status` / - `set-dg-status` verbs are executed through the diskdb service API, - but the CLI verb belongs under `cluster` because it changes hardware - topology state, not chunk service state. `chunk diskdb` owns only the - diskdb service lifecycle and maintenance (scan/recalc/compact/ - rebuild). -- **`kv`** — KV layer: `kv server` (crowdb-kv-server lifecycle), - `kv store` / `kv group` / `kv replica` (logical concepts), `kv put` - / `get` / `delete` / `scan` / `snapshot` (data-plane). The verb - distinguishes management from data-plane; no `kv` prefix needed on - resource names. -- **`chunk`** — chunk storage service cluster: `chunk diskdb` / - `chunk chunkdb` / `chunk diskio` (server lifecycle + maintenance) + - future chunk data-plane (`allocate` / `free` / `write` / `read` / - `gc`). diskdb (block allocator), chunkdb (chunk metadata), diskio - (disk I/O), and the chunk client lib compose the chunk storage - service cluster; the group name reflects the unified service, not - individual servers. Stubs pending implementation. -- **`bench`** — load injection per layer. - -The four-domain hierarchy is the **standard concept** across the -production system — not CLI-specific. The console UI (`crowdb-web`) -uses the same domain grouping for its navigation and operation -surfaces (see `design-crowdb-console-ui.md`). The operation logic -behind each verb lives in `crowdb-console-shared`'s `ops` module -(§2.2); both frontends call the same shared operations, so CLI and UI -behave identically. +`cluster` owns hardware metadata and bootstrap, clean, destroy and status. +`kv` owns KV server launch controls, logical store/group/replica operations +and KV data commands. `chunk` owns storage-service launch controls and +maintenance. `bench` owns workload runners. The CLI connects to Group 0 +directly and shares the authority operations with production Web. ### 7.2 Command Hierarchy -``` -crowdb-cli -│ -├── cluster (alias: cls) ← hardware topology + cluster-level ops -│ ├── init (--nodes; bootstraps group 0 — §7.3) -│ ├── reset (full teardown — §13) -│ ├── clean (wipe user data, keep metadata + group-0 — §7.4) -│ ├── status -│ ├── topology -│ ├── rack { add, remove, list } -│ ├── node { add, remove, list, ping } -│ ├── disk-group { add, remove, list, set-status } -│ └── disk { add, remove, list, set-status } -│ -├── kv ← KV layer: server + logical concepts + data-plane -│ ├── server { deploy, restart (alias start), stop, delete, list } (delete — §7.5) -│ ├── store { add, remove, list, inspect } -│ ├── group { add, remove, list, inspect } -│ ├── replica { add, remove } -│ ├── put / get / delete / scan -│ └── snapshot { create, list, scan, release } -│ -├── chunk ← chunk storage service cluster (stubs) -│ ├── diskdb { deploy, restart, stop, delete, list, usage, -│ │ scan-status, scan, recalc, compact, rebuild } -│ ├── chunkdb { deploy, restart, stop, delete, list } (future) -│ ├── diskio { deploy, restart, stop, delete, list } (future) -│ └── allocate / free / write / read / gc (future data-plane) -│ -└── bench ← load injection - ├── kv { read, write, scan, mix } - ├── rpc - ├── diskdb { allocate, mix } (future) - ├── chunkdb { allocate, mix } (future) - └── chunk { write, read, mix } (future) -``` - -**Three layers max** — `crowdb-cli ` -(e.g. `kv server deploy`, `kv store add`, `cluster rack list`). -Direct data-plane verbs are two layers (`kv put`, `chunk allocate`). - -**Verb vocabulary:** -- Resource CRUD: `add / remove / list / inspect`. -- Server lifecycle: `deploy / restart / stop / delete` — consistent - across `kv server`, `chunk diskdb`, `chunk chunkdb`, `chunk diskio`. - `start` is an alias of `restart`. Servers are deployed one-per-node - by default; `list` enumerates instances across all nodes. -- Data-plane: `put / get / delete / scan`. The API uses `scan` for - prefix-scan; `list` is management-only (enumerates resources, not - data), never data-plane. -- Hardware state: `set-status` on `cluster disk` / `cluster disk-group`. - -**Logical entity addressing**: store/group/replica/KV commands use -`--store` / `--group`; the backend resolves placement. Server -lifecycle uses `--node`. - -**Leaders are elected, not assigned.** `kv group add` takes no -`--leader` flag; leadership is decided by Paxos election. +The `clap` command enums define the exact verbs and flags. Hardware and +logical commands use Group 0 for identity and membership. Process controls +require `--registry`; `cluster init` additionally requires sealed bootstrap +input for first creation. Development `local-deploy` runs a one-shot loopback +cluster and prints the Group 0 management seed for later CLI invocations. ### 7.3 `cluster init` — bootstrap special case -`cluster init` is the only command that runs before group 0 exists. -It takes `--nodes ` directly (not `--system-ip` / -`--system-port`) and bootstraps group-0/store-0 on those nodes via +`cluster init` requires `--registry` and a versioned `--bootstrap-file` +for first creation, or a sealed retry intent beside the registry. It takes +`--nodes ` and bootstraps group-0/store-0 on those nodes via direct node REST calls (the `POST /system/init` mechanism, §4.3), wires remotes, and writes the hardware + KV-cluster topology into group-0 sysdata. After `cluster init` completes, subsequent commands @@ -673,43 +444,17 @@ registrations rather than treating launch configuration as a live endpoint. ### 7.4 `cluster clean` — data wipe boundary -`cluster clean` wipes user-layer data across all storage services, -keeping services running and group 0 intact: - -- **KV user data** — remove all user stores + groups via the existing - store/group removal flow (cascades to replicas and on-disk WAL/tree - cleanup). group-0/store-0 preserved. -- **chunkdb metadata** — chunkdb stores metadata in CROWDB KV; cleaning - the chunkdb KV store (same as any KV store removal) wipes chunkdb - metadata. -- **diskio data** — diskio writes at positions it points to; later - writes overwrite old data. No explicit clean needed — new writes - supersede old data. -- **diskdb metadata + backing** — remove all diskdb metadata (clean the - diskdb group(s) in KV sysdata). For file-simulated disks, trim or - reset the backing file to reclaim space. For real devices, metadata - removal is sufficient (zones are reclaimed on next allocation). - -Services (`crowdb-kv-server`, `crowdb-diskdb`, `crowdb-chunkdb`, -`crowdb-diskio`) stay running. group-0 leadership continues — leaders -are elected, not assigned; as long as group-0 replicas survive, they -elect a leader. Topology (racks/nodes/disk-groups/disks) is preserved. - -For repeated full-stack benchmarks, `cluster clean --restart-services` -extends the boundary after the KV wipe. The console stops all locally deployed -DiskDB, DiskIO, and ChunkDB processes, then starts DiskDB and DiskIO before -ChunkDB with the same identities, endpoints, working directories, and launch -arguments. It waits for health, service registration, and ChunkDB range -bindings before returning. KV processes remain running so group 0 and hardware -topology survive. Suites with multiple data groups clean every group and request -the service restart on the final clean. A high-volume `mem-block` suite may use -`cluster destroy` followed by a fresh combined deployment for each case. This -process boundary releases the complete in-memory working set and prevents RSS -from accumulating across independent benchmark cases. - -Local auxiliary launch commands are retained in the run-root `console.toml`. -They are diagnostic lifecycle state, are removed with their server entry, and -are cleared by `cluster destroy`. +`cluster clean --store --group ` derives the target replica nodes +from confirmed Group 0 membership and resolves every live KV management +registration. It asks each target to wipe user data, then waits for a new +leader. Group 0 hardware, logical records, and process launch policy remain +intact. A missing group, registration, or acknowledgement fails the operation; +a local launch record cannot justify a wipe. + +`--restart-services` additionally restarts locally configured DiskIO, DiskDB +and ChunkDB processes in dependency order through `LaunchRuntime`. It requires +a validated launch registry before the wipe begins. KV processes stay running +so Group 0 remains available. ### 7.5 `kv server delete` — graceful + require-empty @@ -735,7 +480,9 @@ Verb distinction: - `bench kv ` runs KV workloads against a target store/group. `bench rpc` measures raw RPC transport throughput. -- Both are stubs pending re-wiring to the `ops` module. +- Bench discovery starts from the explicit Group 0 management seed and + resolves metrics hosts from confirmed replica membership and live + registrations. It does not load a console topology file. ### 7.7 Bench lifecycle verbs (deploy / prepare / run / teardown) @@ -787,7 +534,7 @@ path. The location, rather than the caller's global console configuration, is the cluster identity and recovery boundary: - a missing or empty location is initialized as a three-node cluster; -- a location containing `s3-mini-cluster.json` and `console.toml` is restarted +- a location containing `s3-mini-cluster.json` and versioned local launch state is restarted with the same service identities, endpoints, launch commands, KV/WAL/tree directories, and DiskIO files; - a non-empty location without the marker is rejected without modification. @@ -801,11 +548,11 @@ does not remove configuration or storage. `delete` stops the cluster, releases its persistent port claims, and removes the named location. `status` is read-only. -First start is transactional. The complete marker is published only after the -access endpoint is ready; failure stops the processes created by that -invocation. A later start archives an incomplete initialization directory next -to the selected location before retrying, preserving its logs for diagnosis -without treating it as a recoverable cluster. +First start seals bootstrap intent before publishing Group 0 and publishes the +complete marker only after the access endpoint is ready. Interrupted launch +steps replay from retained process and seed inputs; committed Group 0 content +is verified before a missing step is retried. A non-empty foreign directory is +rejected. Local state cannot reconstruct topology during a Group 0 outage. The mini topology is intentionally loopback and places its simulated nodes in one rack, so ChunkDB explicitly permits colocated fragments. This is not the @@ -821,12 +568,10 @@ be used for a non-loopback listener. Bucket and object commands take the same `--root`, discover the persisted endpoint, preserve S3 errors, and do not fall back to another mutation. -The durable record is deliberately small. `console.toml` retains service PIDs -and reproducible launch specifications; `s3-mini-cluster.json` retains only the -format version, storage profile, loopback endpoint, and non-secret tenant name. -The access master key is injected into a child only while it starts and is not -persisted in either record. Runtime liveness is derived from recorded PIDs, -not represented by additional compound cluster states. +The durable local record holds only versioned launch inputs, process +identities, bootstrap seeds, the storage profile, loopback endpoint and +nonsecret tenant name. It does not contain rack, node or logical topology. The +access master key is injected into a child only while it starts. `crowdb-cli bench s3` owns a separate, invocation-scoped memory profile. KV and WAL blocks use memory backing, DiskIO uses memory disks, and chunk-KV keeps its @@ -893,59 +638,23 @@ The following invariants apply: events; metric validation therefore checks metric sections and counters independently of auxiliary log size. -## 10. Open Questions - -- **SSH crate**: `russh` (decided). Defaults to `~/.ssh/*`; `(user, - password)` is an explicit alternative. -- **Frontend bundle**: built on demand; `npm run build` produces `dist/` - which the Axum server serves. The committed repo does not include - `web/dist/`. -- **Credentials storage**: plaintext TOML, accessed only through - `ConsoleConfig` so the source can change later without touching call - sites. -- **Multiple servers per node**: UI and console enforce one; lower - layers remain unrestricted. - -## 11. Sysdata sync — rack/node/disk-group/disk handlers - -Console add/remove handlers for racks, nodes, disk-groups, and disks -delegate to `ops::hardware::*`, which updates the `OpContext` config -snapshot first, then syncs group-0 sysdata via `ctx.sysmd()` -(`HardwareClient`). If group 0 is not yet initialized, the sysdata -sync is skipped — `cluster_init` Phase 5 writes the full hierarchy on -bootstrap. After `ops::*` returns, the handler writes the mutated -config back to `AppState.config` and persists to TOML. - -## 12. Cluster reset - -`cluster destroy` is full teardown. It is implemented in -`crowdb-console-shared`'s `ops::cluster::reset` as a hybrid operation -— group-0 discovery + direct node teardown — so the CLI no longer -depends on a `crowdb-web` endpoint. The flow: - -1. **Discovery** — connect to the system group (via `--system-ip` / - `--system-port`) to enumerate all resources: user stores/groups/ - replicas, diskdb/chunkdb/diskio instances, server entries, topology. -2. **Teardown in dependency order** — erase resources one by one: - remove user groups → user stores → clean group-0 sysdata (rack - cascade + store records + diskdb service unregister) → SIGTERM each - node's processes. -3. **Destroy group 0** — tear down group-0/store-0 itself (last, after - all user resources are gone). -4. **Delete topology** — remove all nodes and racks from - the persistent console configuration. -5. **Fast path** — if group 0 is not created (e.g. `cluster init` - failed or was never run), skip steps 1-3 and use the TOML config - info (rack/node entries) to clean up any stray processes and clear - the config. - -The `POST /internal/reset` endpoint on `crowdb-kv-server` remains for -UI use; the CLI implements its own teardown via the shared `ops` -module. When no KV servers are running, the RPC steps are skipped -(fast path for E2E test fixtures). - -The web backend exposes `POST /api/cluster/reset` (calls -`ops::cluster::reset`) and `POST /api/cluster/clean` (calls -`ops::cluster::clean` — removes orphaned sysdata entries from stopped -servers without full teardown). Both are reachable from the CLI and -the web UI. +## 10. Hardware mutations + +CLI and bare-metal Web use the shared Group 0 hardware operations. Rack, +node, disk-group and disk changes update parent and child records in one +conditional batch where membership changes. Matching retries are confirmed; +conflicting concurrent writes preserve the existing record. Docker Web does +not expose hardware or process mutations. + +## 11. Cluster teardown and verification + +`cluster destroy` requires the local launch registry. It reads confirmed Group +0 membership, removes user stores and groups through shared logical operations, +then removes the system group last. Only after metadata teardown succeeds does +it stop processes named by that console's launch registry. A failed or +unconfirmed step returns an error instead of deleting presumed local topology. + +There is no orphan-guessing reset command. A stopped or unreachable node does +not imply its membership should be deleted. `cluster clean` derives its +replica targets from Group 0, wipes each live target, and waits for a new +leader while preserving topology. diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index 71b17a2f..06920c41 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -90,153 +90,40 @@ configuration, documentation and crash-diagnostics tasks below are pending. Three-service lifecycle regressions and the complete CLI suite, fmt and clippy pass. Logs: `/tmp/crowdb-chunk-launch-*.log`. Removal of legacy startup/restore paths remains coupled to bootstrap cutover below. -- [~] **Remove mixed persistence**: remove the unreleased `ConsoleConfig` - parser/writer, inline SSH secrets, topology restoration and fixtures after - the launch lifecycle and replay-safe bootstrap paths are wired. Preserve - bootstrap intent independently until verified cutover. Update CLI commands, - Web persistence and S3 mini-cluster callers together; no compatibility reader - or migration path because the old configuration was never released. - S3 mini-clusters now persist versioned local process/seed state rather than - `console.toml`; restored KV launch nodes are ephemeral process inputs and the - bundled Web uses `WebProcessConfig`. The full persistent S3 stop/restart and - range-read E2E passes. Web no longer constructs a `TomlFileEngine` or writes - mixed local topology, including its in-process legacy test router; focused - bootstrap, lifecycle and deployer tests pass. The CLI legacy config path and - shared parser/writer still need removal. - `kv server` process commands now require the versioned launch registry; - their former no-registry branch, including persisted PID/topology updates, - is removed. The registry lifecycle regressions and complete CLI suite pass. - `cluster init` now requires the registry and independent versioned bootstrap - input; it cannot seal or publish from the old mixed file. The registry - bootstrap test passes, and a no-registry CLI test verifies the rejection. - `cluster local-deploy` uses a fresh in-memory context for new loopback - clusters and reconstructs incremental DiskDB/ChunkDB inputs from confirmed - Group 0 hardware and live KV registrations. It no longer reads or writes - the old topology file. A real CLI regression reaches Group 0 validation - with a deliberately invalid legacy file. - CLI `cluster clean` now obtains replica membership and live endpoints from - Group 0. Optional storage process restarts use the validated launch registry - in DiskIO, DiskDB, ChunkDB order; without one the command rejects the - restart request before wiping data. - CLI chunk service lists and benchmark discovery no longer load the old - topology file. KV benchmark metrics resolve confirmed replica hosts and - their live management registrations. The direct CLI fixture keeps bootstrap - input only in memory. The complete CLI suite and workspace clippy pass. - The `ConsoleConfigEngine`/`TomlFileEngine` parser and writer are removed; - `ConsoleConfig` remains an ephemeral operation/intent input. CLI `destroy` - requires a launch registry, removes confirmed logical metadata before Group - 0, then stops configured local processes. CLI `reset` verifies live confirmed - hosts without deleting stopped nodes as presumed orphans. Real one-node - CLI regressions cover both commands with an invalid legacy file. -- [ ] **Confirmed hardware operations**: route CLI and bare-metal Web through - shared Group 0 hardware operations; preserve conflicts and uncertain writes - without local-first commits. Docker keeps its hardware restrictions. Hardware - client cascades now stop at a failed child deletion instead of deleting its - parent while a descendant may survive. Extend Group 0 hardware values with - rack names, node management hosts, nonsecret SSH connection settings and - credential reference IDs; resolve secret material locally. Those Group 0 - fields are now part of rack/node records and bootstrap writes the names, - management host, SSH port/user and reference without copying secret material. - Bare-metal snapshots expose the same Group 0 values to separate consoles. - Conditional rack/node creation now confirms matching existing values and - rejects conflicting values without changing either console's local topology. - Group 0 rack/node lists expose shared names, management hosts, SSH settings - and credential references but no private key material. Registry-mode CLI - add/list commands use these operations; a real CLI process regression caught - a management-port-as-RPC seed and now refreshes topology before the hardware - write. Complete Console shared and CLI suites, Rust fmt and workspace clippy - pass. A real RPC proxy drops the committed rack write reply; a confirmed - linearizable read recovers the successful outcome. Registry-mode Web, - deletions, disks and legacy bootstrap still remain. - Node creation now conditionally updates rack membership and creates the node - in one Group 0 batch. Two concurrent consoles retain both child IDs; a - repeated rack add preserves its existing children. Retried node creation - compares immutable connection identity while preserving live status fields. - Registry CLI rack removal now conditionally deletes only a confirmed empty - rack; shared and real CLI regressions cover child conflict and absence. - Bare-metal Web now exposes authenticated Group 0 rack creation/removal and - node creation, with public confirmed rack/node reads. Two Web instances - observe the same records; inline SSH material is rejected. Registry CLI and - bare-metal Web now remove only unused nodes through one conditional Group 0 - rack-membership/node deletion; occupied nodes and unauthenticated Web writes - fail. Disk-group and disk creation/removal now update the child and parent - records in one rack-revision-fenced Group 0 batch. Their names, membership - and disk attributes are read from confirmed authority by CLI and bare-metal - Web; tests cover two consoles, conflicts, occupied deletion, private Web - writes and a lost committed write response. Legacy Web routes and CLI - mixed-config paths still need removal. - Every CLI rack/node/disk-group/disk command now builds a Group 0-only - context, including when `--registry` is omitted. A malformed legacy file - cannot alter or block rack/node reads. The complete CLI suite passes; - remaining mixed CLI operations are outside hardware. - CLI logical store/group/replica mutations and reads, plus cluster status - and node topology, now also ignore the old file. Live node registration - locates the topology endpoint; a management read supplies only the initial - RPC connection hint. The logical round-trip and status/topology tests pass - with a deliberately invalid legacy file, and the complete CLI suite passes. - CLI KV data commands also use this context; the put/get/delete/scan - round-trip passes with an invalid legacy file. -- [ ] **Authority-only reads**: replace local monitor/config topology and - endpoint fallbacks with Group 0 and live registrations. Missing, ambiguous or - expired registrations remain unavailable. - Versioned bare-metal snapshots now read Group 0 without requiring a Docker - monitor; Docker keeps its monitor requirement and overlay. Validate every - replica host as well as the store's original hosts. Real Group 0 regressions - cover missing, duplicate and expired registrations, recovery, and outage - without stale topology. Docker and launch-route regressions, fmt and clippy - pass. Logs: `/tmp/crowdb-bare-authority-*.log`. Legacy physical routes and - monitor refresh still remain for the mixed-config removal. Production Web - startup now requires a versioned process config, so it never loads the old - mixed file; bare-metal rack/node/disk-group/disk detail and collection - routes read Group 0 directly. The old in-process router and CLI no-registry - paths remain to migrate or remove. -- [ ] **Replay-safe bootstrap cutover**: persist bootstrap identity, verify - committed records, write only safely missing content, reject conflicts and - delete topology intent after verified transfer. Clean/destroy use confirmed - authority. Replace S3 mini-cluster persistence against the same contract, - without a migration path for its unreleased mixed configuration. - System initialization now confirms an existing replica's identity after a - conflict or lost response, preserves groups for retry after peer failures, - and requires every peer endpoint and remote-wiring request to succeed before - recording membership. Four focused failure cases, complete shared tests, - Web deploy/restart/migration suites, fmt and clippy pass. Logs: - `/tmp/crowdb-bootstrap-replay-*.log`. Durable intent/cutover remain pending. - A separate versioned bootstrap intent now captures rack/node identity, KV - management endpoints and selected member order without process PIDs, binary - paths or inline SSH secrets. It is atomically sealed with mode 0600; - interrupted retries restore a fresh in-memory context, reject changed - topology before mutating Group 0, then delete the intent only after - confirmed publication. Persistent CLI and legacy Web cluster-init callers - now use this path. Real Web and CLI regressions confirm the intent is removed - after successful Group 0 publication. Versioned bare-metal Web now exposes - an authenticated cluster-init route and accepts the same independent - bootstrap input without writing a mixed console file. Registry CLI accepts - a versioned bootstrap input, seals an immutable - retry copy beside the launch registry, runs the same confirmation path and - deletes that copy after success. It does not write the mixed console file. - The legacy CLI/Web path still writes that file; S3 mini-cluster still needs - bootstrap-interruption replay before the old format can be removed. Its - completed cluster now restarts from launch-only local state and Group 0 - seeds; no local topology is loaded after publication. A full persistent S3 - stop/restart and range-read E2E passes. S3 now saves its local KV launch - state before sealing bootstrap intent and publishing Group 0. An interrupted - launch retains that state for identity-checked retry instead of archiving the - committed cluster. A real failure injected at Chunk KV startup recovers on - the next CLI invocation, with all 15 services ready and no topology file. - The CLI integration test reproduces this interruption and recovery. Partial - storage-service launch sets now clear only the incomplete local process - entries after checking a still-live process's work directory, then replay - provisioning from confirmed Group 0 metadata. A DiskIO startup failure - after DiskDB launch recovers on the next CLI invocation with no local - topology copy. Mixed CLI/Web config remains to remove. - S3 now canonicalizes a relative root before creating child launch paths; - the interrupted CLI test covers a relative root. The S3 CLI mock fixture - supplies the required launch-only state; its three previously failing cases - now pass. The complete Console gate passes after the S3 fixture update. - `cluster clean` now derives its target nodes from confirmed Group 0 replica - membership and resolves each live management registration; local launch - entries cannot justify a wipe. A real Group 0 regression rejects a group - absent from authority even when the console has a local server entry. +- [x] **Remove mixed persistence**: production Web requires versioned + `WebProcessConfig`; bare-metal process controls use `LaunchRegistry` and + `LaunchRuntime`. Docker rejects that registry. CLI and Web do not parse or + write a local topology file. `ConsoleConfigEngine` and `TomlFileEngine` are + removed; `ConsoleConfig` remains an ephemeral bootstrap/development input. + S3 stores only launch inputs and seeds locally. CLI hardware, logical, + data, chunk discovery, benchmarks, clean and destroy do not read the + old file. No migration or compatibility reader is kept. The old + mixed-binary rolling-upgrade fixture was removed because no release exists; + current-version restart cases remain. +- [x] **Confirmed hardware operations**: CLI and bare-metal Web rack, node, + disk-group and disk operations use shared conditional Group 0 writes and + confirmed reads. Parent membership and child records change atomically; + matching retries reconcile lost replies, conflicts preserve the winning + record, and inline SSH material is rejected. Docker Web rejects hardware + writes. Real two-console and dropped-reply regressions pass. +- [x] **Authority-only reads**: production Web snapshots and CLI read Group 0 + membership with live service registration. Docker process state comes from + the monitor; bare-metal process status comes from `LaunchRuntime`. + Missing, duplicated and expired registrations, monitor absence, and Group + 0 outages report unavailable without using local topology as fallback. + The in-process Web test router keeps ephemeral fixture state and is not a + production compatibility path. +- [x] **Replay-safe bootstrap cutover**: versioned sealed `BootstrapIntent` + survives interruption, confirms already committed content, conditionally + publishes safely missing records, rejects conflict and clears only after + verification. Nonmember KV processes receive seed hints and register one + live identity. S3 replays partial KV and storage launches from local + process inputs and confirmed Group 0 metadata, without a topology copy. + `cluster clean` targets confirmed replicas; `cluster destroy` removes + confirmed logical records before Group 0 and stops registry processes. + The old orphan-guessing `cluster reset` command and Web route are removed; + an unreachable service cannot cause metadata deletion. Focused real authority and replay + tests plus the complete Console gate pass. - [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical records, accept matching content without rewriting revisions, reject conflicts, and conditionally create missing records. Reconcile uncertain writes with @@ -302,6 +189,10 @@ requirement. - [ ] **Acceptance and cleanup**: run affected integration cases, full console and UI suites, Rust fmt and lint; update the relevant permanent architecture, then remove the requirement, backlog entry and this plan when complete. + The full Console and UI gates pass (86 component tests, 56 browser tests). + The website deployment route/link tests pass after correcting the chunk + guide's Group 0 wording. Monitor, container, final fmt/lint and cleanup + remain. ## Evidence diff --git a/doc/working/test.md b/doc/working/test.md index 196374ff..b89e9c3a 100644 --- a/doc/working/test.md +++ b/doc/working/test.md @@ -146,7 +146,6 @@ All individual tests or test binaries with wall-clock time >= 7 s. | `test-chunk-client` | 18.33 s | `small_object_writer_e2e` — small-write E2E with real ChunkDB + DiskIO (14) | | `test-console-server` | 18.07 s | `cluster_deployer_test` — deployer lifecycle (3 tests) | | `test-console-shared` | 15.12 s | `lifecycle_e2e_test` — lifecycle E2E (1 test) | -| `test-console-server` | 13.53 s | `rolling_upgrade_test` — rolling upgrade (1 test) | | `test-console-ui` | 10.8 s | `50-chunk-capacity-disk-group:428` — assign disk-group to diskdb via UI | | `test-chunk-client` | 10.39 s | `chunk_reader_e2e` — chunk reader E2E with failure injection (6 tests) | | `test-console-server` | 9.93 s | `cluster_restart_incremental_test` — restart cycles (5 tests) | diff --git a/lib/crowdb-console-shared/src/config.rs b/lib/crowdb-console-shared/src/config.rs index 50c913b6..b2b50037 100644 --- a/lib/crowdb-console-shared/src/config.rs +++ b/lib/crowdb-console-shared/src/config.rs @@ -130,8 +130,7 @@ pub struct DiskEntry { pub device_path: String, } -/// Discriminator for console-deployed server entries. `Kv` is the -/// default for backward compatibility with existing persisted configs. +/// Discriminator for ephemeral console deployment inputs. #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)] pub enum ServiceType { #[default] @@ -142,14 +141,13 @@ pub enum ServiceType { ChunkKv, AccessServer, /// Standalone crowdb-rpc-fb-server (C++ echo server for RPC bench). - /// Not a full KV server — no management port, no sysdata. Tracked - /// in config only for PID/port lifecycle via `cluster destroy`. + /// Not a full KV server — no management port or sysdata. Rpc, } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct ServerEntry { - /// Console-side identifier; must be unique within the file. + /// Console-side identifier; must be unique within an operation context. pub id: String, /// Service URL. For KV this is the `crowdb-kv-server` management base /// URL; for `DiskDB` this is its public crowdb-rpc endpoint. @@ -174,13 +172,11 @@ pub struct ServerEntry { pub election_profile: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub pid: Option, - /// Service type discriminator (R77). Defaults to `Kv` for - /// backward compatibility with pre-R77 persisted configs. + /// Service type discriminator. KV is the default for local fixtures. #[serde(default, skip_serializing_if = "is_default_service_type")] pub service_type: ServiceType, /// `--rpc-workers` value passed to the spawned `crowdb-kv-server`. - /// `None` means the server's default (2) is used. Persisted so - /// restart reuses the same value. + /// `None` means the server's default (2) is used. #[serde(default, skip_serializing_if = "Option::is_none")] pub rpc_workers: Option, /// `--no-fsync` flag passed to the spawned `crowdb-kv-server`. diff --git a/lib/crowdb-console-shared/src/ops/cluster.rs b/lib/crowdb-console-shared/src/ops/cluster.rs index 70454b6c..9709fbed 100644 --- a/lib/crowdb-console-shared/src/ops/cluster.rs +++ b/lib/crowdb-console-shared/src/ops/cluster.rs @@ -1,13 +1,12 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Cluster-level operations: status, topology, init, reset, clean. +//! Cluster-level operations: status, topology, init, destroy, clean. //! //! `init` bootstraps group 0 (store 0, group 0) on the selected nodes, //! wires remotes, and writes the hardware + KV-cluster topology into -//! group-0 sysdata. `reset` tears down the cluster in dependency order. -//! `clean` removes orphaned sysdata entries without touching running -//! servers. +//! group-0 sysdata. `destroy` tears down confirmed membership in +//! dependency order. `clean` wipes data on confirmed replicas. use std::collections::{HashMap, HashSet}; use std::sync::Arc; @@ -97,21 +96,6 @@ pub async fn destroy(ctx: &OpContext) -> Result<()> { Ok(()) } -/// Verify that every confirmed store host has one live registration. A -/// stopped or unreachable node is not evidence that its metadata is orphaned. -/// -/// # Errors -/// Returns an error if any confirmed host cannot be verified. -pub async fn reset(ctx: &OpContext) -> Result<()> { - for store in ctx.sysmd().list_stores().await? { - for node_id in store.node_ids { - let url = ctx.live_node_mgmt_url(node_id).await?; - ServerClient::new(&url)?.health().await?; - } - } - Ok(()) -} - /// Result of [`clean`] — wipe user data + wait for re-election. #[derive(Debug, Clone, serde::Serialize)] pub struct CleanResult { From 0c776166a7a0813d20ec4eb3314412ac90f14a57 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 16:01:17 +0800 Subject: [PATCH 71/74] Complete console authority requirement --- doc/backlog/R188-console-group0-authority.md | 183 ----------- doc/backlog/backlog.md | 8 - doc/working/plan-console-authority.md | 311 ------------------- 3 files changed, 502 deletions(-) delete mode 100644 doc/backlog/R188-console-group0-authority.md delete mode 100644 doc/working/plan-console-authority.md diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md deleted file mode 100644 index e6cc33e4..00000000 --- a/doc/backlog/R188-console-group0-authority.md +++ /dev/null @@ -1,183 +0,0 @@ - - - -### R188: console — Group 0 authority and deployment configuration cleanup - -## Problem - -The unreleased `ConsoleConfig` in -[`design-crowdb-console.md`](../design/console/design-crowdb-console.md) -currently mixes cluster topology with host, SSH, binary, port, PID, and local -launch information. CLI and bare-metal Web can write the local file before a -Group 0 mutation succeeds, and some reads and restart paths still accept that -file or a monitor cache as topology authority. A second console can therefore -observe a different cluster, while an outage can resurrect stale topology. - -R187's Docker monitor already owns container process supervision; Group 0 owns -CROWDB system metadata, not container IDs, images, mounts, PIDs, restart -generations, or machine-local launch policy. Completing a cross-mode console -rewrite is not a prerequisite for packaging that monitor and the single-node -profile. The remaining boundary cleanup belongs in this separate requirement. - -At the user's request, this requirement also owns the deferred container crash -diagnostics work: core collection, bounded retention and source-line -symbolization. Existing crash recovery is implemented, but usable diagnostic -dumps depend on the host collector and exact-build symbols. This follow-up does -not block R187 completion. - -## Solution - -1. Keep Group 0 as the durable authority for CROWDB hardware hierarchy, - including rack names, node management hosts, nonsecret SSH connection - settings and credential reference IDs, ownership and binding maps, KV - store/group/replica metadata, and service registration. Do not store SSH - private keys or passwords in Group 0. Do not add process deployment records - to Group 0. Docker process state comes from `crowdb-monitor`; bare-metal launch - policy remains local. Deployment mode changes which lifecycle and hardware - controls are allowed, not the meaning of Group 0 records. -2. Replace the mixed `ConsoleConfig` persistence in - `crowdb-console-shared::config`, `crowdb-web`, and `crowdb-cli` with a - versioned Web process configuration and a separate bare-metal launch-only - registry. Finish wiring the existing `LaunchRegistry` parser to actual - bare-metal deploy/restart operations; remove the unreleased mixed - parser/writer, topology fields, restore path, fixtures, and fallback rather - than adding a compatibility reader. Resolve Group 0 credential reference IDs - through each console's local secret store. Retain binary/config paths, - workspace, and auto-start policy locally; never persist inline secrets or - runtime PID as topology. Docker Web rejects a launch - registry and keeps its monitor-owned process path. -3. Unify CLI and bare-metal Web hardware mutations through Group 0-backed - operations in `crowdb-console-shared::ops::hardware`. Confirm writes before - updating a read model; preserve conflicts and uncertain results. Docker Web - continues to reject hardware and process mutations. -4. Complete the common Group 0-backed logical store/group/replica flow in - `crowdb-console-shared::ops::kv_logical` for CLI and both Web modes. Reconcile - lost responses by reading confirmed authority, test multi-node fan-out and - rollback, and remove local logical-topology commits. A failed node-side - deletion must not erase surviving Group 0 membership. -5. Replace config-backed monitor refresh, KV endpoint fallback, and - physical/deployment topology reads with Group 0 membership and live service - registration. A missing or ambiguous live endpoint fails unavailable; a - stopped service may still have local launch policy but is not reported as - live. Neither a local launch registry nor a monitor cache is an authority - fallback during a Group 0 outage. -6. Keep pre-Group-0 bootstrap intent separate. After creation, verify every - committed hardware and logical record and delete the local topology copy. - Persist enough bootstrap identity to resume an interrupted transfer, prove - already committed content, and reject conflict. Destroy/clean must use - confirmed Group 0 state. If nonmember KV processes were launched before - Group 0 exists, propagate usable Group 0 seed hints after initialization - before treating their registration as live; seed hints are not topology. -7. Replace the S3 mini-cluster's local `console.toml` and restart path under the - same authority boundary. Retain only launch inputs and bootstrap seeds - locally after Group 0 cutover; do not replay a local topology copy. This - configuration has not been released, so no migration or compatibility path - is needed. -8. Publish the verified bare-metal deployment and operations material under - `/nv/cpp/crowdb-web/site/docs/`, organized by KV cluster, chunk layer, and - data access servers. State that bare-metal is not yet production-ready. - Keep Docker deployment documentation independent and do not add deployment - guides to this repository. -9. Complete container crash diagnostics without changing host-wide collector - policy. Respect file-based core patterns, Ubuntu Apport, systemd-coredump and - Docker Desktop's Linux VM; document where dumps actually go or why collection - is unavailable. Where relative `core` file dumps are supported, use a - private directory inside the container data volume and retain only the - newest core across child recovery and monitor restart. Bound each dump with - Docker's core ulimit. Provide an exact-build source-line - symbolization workflow for child and monitor crashes. Dumps can contain - secrets and user data; diagnostics must not expose them in ordinary logs. - Optionally ship exact-build debug symbols as a separate GitHub Release asset - generated from the same staged runtime as the image, indexed by version and - source revision. Omit this large asset by default so its upload cannot block - image publication. The release preparation script in `tools/` runs manually, - shows a dry-run plan, updates versions, creates the tag and GitHub Release, - then dispatches the existing verified DockerHub publication workflow. Its - actual use is deferred to the operator's later release; no release execution - is required for this requirement. Document host crash collection, GDB and - exact-build symbols in `doc/dev/crash_debugging.md`. The user will validate - a real core when a future crash occurs; this requirement does not change - the host collector or require a new crash test. - -## Dependencies - -- R187 provides the working single-node Docker profile, monitor-owned process - state, managed Web baseline, and Group 0-backed system metadata. R187 image - verification does not depend on this cross-mode cleanup. -- The existing Group 0 schema and `crowdb-kv-client` service APIs remain the - authority. If a live registration is absent, operations fail unavailable or - wait for registration; local launch policy never substitutes for it. -- The old mixed console file is unreleased. No on-disk compatibility promise or - migration tool is required, but bootstrap replay must not overwrite a - confirmed initialized cluster. - -## Acceptance - -- Given a Docker process restart and a bare-metal process restart, when runtime - state is queried, assert Docker PID/restart state comes from the monitor and - bare-metal launch policy stays local, while neither appears as Group 0 - topology. Invariant: deployment state is not sysdata. Integration test. -- Given a mixed legacy config and valid/invalid launch registries, when Web and - CLI start, assert only versioned process and launch inputs are accepted, no - local topology is restored, Docker rejects the registry, and inline secrets - or topology fields fail validation. Invariant: separated configuration. - Integration test. -- Given two bare-metal consoles and one ready Group 0, when each mutates racks, - nodes, disk groups, or disks and a write conflicts or loses its response, - assert both read one confirmed result and neither commits a local-first - topology change. Invariant: hardware authority. Integration test. -- Given two consoles with different local launch registries, when both read the - same rack and node, assert Group 0 supplies identical names, management hosts, - SSH connection settings and credential reference IDs while each console - resolves secret material only from its local secret store. Invariant: shared - hardware display and connection identity never depend on local topology. - Integration test. -- Given CLI, Docker Web, and bare-metal Web with the same Group 0, when each - performs authenticated logical store/group/replica operations, assert one - shared result, correct fan-out/rollback, and no local logical copy. - Invariant: common logical authority. Integration test. -- Given missing, duplicated, or expired registrations and then a Group 0 - outage, when topology, endpoint, or deployment status is read, assert no - stale local endpoint or monitor snapshot is presented as authoritative. - Invariant: fail-closed discovery. Integration test. -- Given a crash before and after each bootstrap commit and before local - deletion, when startup resumes, assert it proves identity and committed - content, writes only safely missing records, and rejects conflict without - overwriting Group 0. Invariant: replay-safe cutover. Integration test. -- Given nonmember KV processes launched before Group 0 initialization, when - Group 0 is created and seed hints are propagated, assert each process - registers exactly one live node identity before logical operations use it. - Invariant: registration readiness. E2E test. -- Given an S3 mini-cluster started with the new launch-only configuration and - a Group 0 outage, when it restarts or tears down, assert local launch data - cannot recreate or mask cluster topology. Invariant: no secondary authority. - Integration test. -- Given the two deployment guides and a reader following bare-metal steps, - when the reader deploys KV, chunk services, and Iceberg or S3 access servers, - assert each layer has a verified setup and health check, the non-production - boundary is explicit, and no link targets the removed combined guide. - Invariant: deployment guidance follows its implementation. E2E test. -- Crash debugging documentation is complete when `doc/dev/crash_debugging.md` - explains host collector selection and rollback, private core location and - limits, exact-image symbols and GDB, and direct GDB use for unstripped - bare-metal binaries. The user will verify a real core during a future - incident; no crash test or host configuration change is required now. - -Required gates: - -- `pixi run clean-env && pixi run test-console` -- `pixi run clean-env && pixi run test-console-ui` -- `pixi run test-monitor` -- `pixi run test-single-node-container` -- `pixi run rs-fmt-check` -- `pixi run rs-lint` - -## Open Issues - -- This host routes `core_pattern` to Apport. A disposable container KV child - aborted and the monitor recovered it, but Apport did not create a CROWDB - report: its log says `/opt/crowdb/bin/crowdb-kv-server` does not exist on the - host. Exact-build symbolization passed with a debugger-generated monitor - core. The user accepted the crash debugging guide as completion and will - validate file collection and source lines when a future real crash occurs. - The host collector was not changed. diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index 3d67512e..4f57be26 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -77,14 +77,6 @@ Caches, selected ORC and container engine workflows remain separate. engine and optional ingest scenarios against the single-node image; publish only tested compatibility recipes. -### Planned — Console authority and deployment - -- **[R188](R188-console-group0-authority.md)** — Group 0 authority and - deployment configuration cleanup — Area: console / CLI / KV — Separate bare-metal launch policy from - cluster sysdata, remove the mixed local topology fallback, and finish - cross-mode console consistency without moving Docker process state into - Group 0. - ### High Priority - **[R103](R103-chunkdb-range-migration.md)** — chunkdb range ownership diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md deleted file mode 100644 index 06920c41..00000000 --- a/doc/working/plan-console-authority.md +++ /dev/null @@ -1,311 +0,0 @@ - - - -# Console Authority Plan - -Upstream: [R188](../backlog/R188-console-group0-authority.md). - -Goal: make Group 0 the shared CLI/Web authority while retaining only process -and launch inputs locally. - -Status: active after the single-node CI repair. The remaining authority, -configuration, documentation and crash-diagnostics tasks below are pending. - -## Registration and acceptance failures - -- [x] **Stable registration across restart**: persist generated instance IDs - under the KV node's config root, reject changed explicit identities and - corruption, and prove restart replaces an unexpired old registration rather - than creating ambiguity. Diagnose the concurrent restart suite without - increasing election timeouts. Files: KV server startup/background identity, - discovery integration and Web incremental restart tests. -- [x] **Pre-bootstrap nonmember registration**: reproduce the missing live - registration for servers started before Group 0 but excluded from its member - set. Propagate discovery seeds after confirmed initialization, retain them - across restart as launch inputs, and wait for exactly one live identity before - declaring the bootstrap complete. Do not create local topology fallback or - restart processes into a different workspace. Files: KV server keepalive and - management modules, console shared cluster initialization and HTTP client, - focused integration tests, UI node-inspection and replica flows. -- [x] **Unavailable logical view**: preserve the explicit unavailable state - before Group 0 exists and during outages, clear stale logical rows, and avoid - treating an expected unavailable response as an unhandled browser exception. - Update the canvas navigation assertions to distinguish unavailable authority - from a confirmed empty cluster. Files: UI logical-tree data hook, KV panel, - shell/canvas/full-chain specs. - -## Common logical operations - -- [x] **Group and replica creation fan-out**: reject missing peer registrations, - missing peer endpoints and failed - remote wiring, roll back created local groups, and publish no Group 0 group - or replica records on these failures. Prove with real Group 0 and controlled - management endpoints. Files: shared `ops/kv_logical.rs` and - `tests/ops_logical_fanout_test.rs`. -- [x] **Logical deletion cleanup**: derive store hosts from both store and - replica membership, confirm every node deletion before removing authority, - and remove descendant records before parents. Test success, sibling - preservation, later replica hosts and node-side failure. Files: shared - `ops/kv_logical.rs`, `tests/ops_logical_delete_test.rs`. -- [x] **Replica creation cleanup**: clean a newly created target store when - local group creation fails, preserve pre-existing target groups, and report - incomplete rollback. Test injected creation and cleanup failures. -- [x] **Confirmed logical mutations**: reconcile lost responses through - confirmed authority, complete replica fan-out/rollback and delete cleanup; - preserve Group 0 membership when node-side deletion fails. Reuse the common - flow in CLI and both Web modes, with no local topology commit. - Conditional publication replaces overwrite writes for stores, groups and - replicas; test concurrent matching/conflicting records and a real RPC reply - dropped after commit. New groups and their initial replicas now use one - conditional batch; confirm the complete member set after a lost reply. - Reconcile failed delete responses with a confirmed absence read. - -## Configuration and hardware operations - -- [x] **Launch registry lifecycle**: wire `WebProcessConfig` and `LaunchRegistry` - into CLI and bare-metal Web deploy/restart paths. Consume binary, service - config, workspace, host and auto-start policy; retain PIDs only in runtime - state and resolve SSH credentials through references. Files: console shared - config/lifecycle, CLI startup, Web startup/state/lifecycle and tests. - First complete launch arguments/readiness inputs, shared local/SSH lifecycle - and runtime-only process identity. Then connect Web auto-start and CLI - deployment/restart callers before removing mixed persistence. - Shared primitives are implemented in `launch.rs` and its local/remote/runtime - modules: private PID/start-time records, idempotent start, referenced service - config and SSH keys, readiness checks, and failure cleanup. Local and real - SSH transport regressions pass, including a native KV launch and refusal to - adopt an unrelated healthy endpoint. Complete Console shared tests, fmt and - clippy pass. Logs: `/tmp/crowdb-launch-{shared,lint,fmt}.log`. - Web now loads and reconciles auto-start policy, exposes authenticated - start/restart/stop and runtime views, and reloads policy on each request. - CLI `--registry` deploy/start/restart/stop/delete uses the same runtime; - deletion checks confirmed replica membership before removing launch policy. - Native Web and CLI integration tests pass, including ignored legacy state, - idempotent start, changed restart identity, and policy edits without a Web - restart. Complete shared/CLI/Web regressions, fmt and clippy pass. - Logs: `/tmp/crowdb-launch-consumers-{full,lint-3,fmt}.log`. - Generic CLI `launch list/start/restart/stop` and chunk diskdb/chunkdb/diskio - deployment controls now share the same launch runtime. Process controls - work before Group 0; chunk service lists still read live registration. - Three-service lifecycle regressions and the complete CLI suite, fmt and - clippy pass. Logs: `/tmp/crowdb-chunk-launch-*.log`. Removal of legacy - startup/restore paths remains coupled to bootstrap cutover below. -- [x] **Remove mixed persistence**: production Web requires versioned - `WebProcessConfig`; bare-metal process controls use `LaunchRegistry` and - `LaunchRuntime`. Docker rejects that registry. CLI and Web do not parse or - write a local topology file. `ConsoleConfigEngine` and `TomlFileEngine` are - removed; `ConsoleConfig` remains an ephemeral bootstrap/development input. - S3 stores only launch inputs and seeds locally. CLI hardware, logical, - data, chunk discovery, benchmarks, clean and destroy do not read the - old file. No migration or compatibility reader is kept. The old - mixed-binary rolling-upgrade fixture was removed because no release exists; - current-version restart cases remain. -- [x] **Confirmed hardware operations**: CLI and bare-metal Web rack, node, - disk-group and disk operations use shared conditional Group 0 writes and - confirmed reads. Parent membership and child records change atomically; - matching retries reconcile lost replies, conflicts preserve the winning - record, and inline SSH material is rejected. Docker Web rejects hardware - writes. Real two-console and dropped-reply regressions pass. -- [x] **Authority-only reads**: production Web snapshots and CLI read Group 0 - membership with live service registration. Docker process state comes from - the monitor; bare-metal process status comes from `LaunchRuntime`. - Missing, duplicated and expired registrations, monitor absence, and Group - 0 outages report unavailable without using local topology as fallback. - The in-process Web test router keeps ephemeral fixture state and is not a - production compatibility path. -- [x] **Replay-safe bootstrap cutover**: versioned sealed `BootstrapIntent` - survives interruption, confirms already committed content, conditionally - publishes safely missing records, rejects conflict and clears only after - verification. Nonmember KV processes receive seed hints and register one - live identity. S3 replays partial KV and storage launches from local - process inputs and confirmed Group 0 metadata, without a topology copy. - `cluster clean` targets confirmed replicas; `cluster destroy` removes - confirmed logical records before Group 0 and stops registry processes. - The old orphan-guessing `cluster reset` command and Web route are removed; - an unreachable service cannot cause metadata deletion. Focused real authority and replay - tests plus the complete Console gate pass. -- [x] **Confirmed bootstrap metadata**: preflight existing hardware and logical - records, accept matching content without rewriting revisions, reject conflicts, - and conditionally create missing records. Reconcile uncertain writes with - confirmed reads; record local membership only after publication is confirmed. - Three real-authority regressions failed before the fix and now pass. Strict - publication exposed missing leader discovery in conditional KV writes: - explicit no-hint not-leader rejections now use the existing bounded retry - policy, while ambiguous dispatch still returns `OutcomeUnknown`. - -## Crash diagnostics follow-up - -Transferred from R187 by user request. It does not block the single-node image -requirement. - -- [x] **Crash dump location and retention**: document how Linux - host `core_pattern`, Docker's core ulimit, and the non-root container affect - CROWDB child and PID 1 crashes. Cover a plain relative core-file pattern, - Ubuntu Apport, systemd-coredump, and Docker Desktop's Linux VM. Choose a - bounded, private location under the mounted `/opt/crowdb/data` volume where - the host permits file dumps; otherwise report the host collector location - and provide explicit setup guidance instead of claiming the volume contains - a core. Use a private data-volume directory and retain the newest `core` - file after child recovery and monitor restart. Require Docker's core ulimit - for a per-dump size bound. Document how to inspect a real child crash, - retention, secret exposure, and exact-build symbolization. Do not - change the host-wide `core_pattern` from inside the container. Files: - `container/single-node-container/{Dockerfile,entrypoint.sh,tests/**}`, - `container/crowdb-monitor/src/**`, - `container/single-node-container/README.md`. - The single-node README now states the host collector boundary and identifies - Apport, systemd-coredump and Docker Desktop lookup paths without promising a - volume dump. This host reports an Apport pipe pattern and core ulimit 0. - The container now creates a private crash directory after the bootstrap - manifest is opened, runs the monitor and children there, and retains the - newest regular `core` file after restart or child recovery. A 1 GiB Docker - core ulimit example bounds each dump. Focused retention and monitor suites - pass. The complete container release, image and E2E gate passes, including - startup, crash and hang recovery, persisted-volume restart, exhausted restart - budget and monitor death. Rust fmt and clippy pass. This host's Apport pipe - can export a packaged program's real dump, but a CROWDB KV child abort left - no report because Apport cannot resolve its container-only executable path. - The monitor recovered the child. A symbols-enabled image and archive for the - same revision passed hashes, debuglink CRCs and `.debug_line` checks; the - symbolizer resolved a debugger-generated CROWDB monitor core to - `container/crowdb-monitor/src/main.rs:55`. The developer guide at - `doc/dev/crash_debugging.md` now covers host configuration and rollback, - private core handling, GDB with exact-image symbols, and bare-metal GDB with - unstripped binaries. The user accepted documentation as completion and - deferred live core verification until a future incident; no new crash test - or host configuration change is required in this task. - -## Documentation and completion - -- [x] **Bare-metal documentation**: publish verified KV, chunk and access - setup under `/nv/cpp/crowdb-web/site/docs/`, state the non-production - boundary, then fix website links and remove obsolete combined material. - Keep Docker deployment notes independent; do not put these guides in crowdb. - The KV, chunk and access guides, deployment index, navigation and sitemap - are published in crowdb-web commit `1c6fc3c`. CLI command shapes were - checked against the executable help and the website route/link test passes. - Final R188 acceptance will verify the complete deployment path after the - remaining hardware operations are wired. -- [ ] **Acceptance and cleanup**: run affected integration cases, full console - and UI suites, Rust fmt and lint; update the relevant permanent architecture, - then remove the requirement, backlog entry and this plan when complete. - The full Console and UI gates pass (86 component tests, 56 browser tests). - The website deployment route/link tests pass after correcting the chunk - guide's Group 0 wording. Monitor, container, final fmt/lint and cleanup - remain. - -## Evidence - -- Bootstrap checkpoint passes complete KV client and Console shared/CLI/Web - suites, five affected browser lifecycle/full-chain cases (53.8s), Rust fmt - and workspace clippy. Logs: `/tmp/crowdb-cas-retry-{baseline,suite,lint}.log`, - `/tmp/crowdb-bootstrap-confirmed-{console,ui,lint}.log`. - Lost-response fixtures now advertise their RPC proxy through management - topology, so discovery refresh cannot bypass the injected reply loss. - -- Deletion reconciliation passes all six cases, including a real dropped - metadata reply for both store and group deletion. Complete Console shared, - affected Web migration/replica tests, fmt and clippy pass. - Logs: `/tmp/crowdb-delete-reconcile-*.log`. - -- Group publication baseline gives the group and initial replica different - committed revisions (3 and 4), exposing partial publication on interruption. - Conditional batch publication passes the shared-revision and lost-batch-reply - tests, full Console shared tests, and Web restart/migration/replica tests. - Orphan membership is rejected before local mutation; all three focused - regressions, fmt and clippy pass. Logs: `/tmp/crowdb-group-publication-*.log`. - -- Conditional publication baseline overwrites a competing store record and - reports success. Both matching and conflicting race tests pass after the - CAS change. A real RPC proxy discards the committed write reply; linearizable - confirmation succeeds and exactly one reply is dropped. Full Console shared - and CLI pass. Full Web passes after the restart-fixture correction below, - including all five concurrent restart cases. Rust fmt and clippy pass. - Logs: `/tmp/crowdb-publication-*.log`. -- The complete Console gate reaches a three-node restart failure: no complete - store view within 3s. Its persisted identities are stable; node logs show - repeated Group 0 elections and late registration, including election churn - before restart. This real-process case uses the paused-clock `test` profile - (5ms heartbeat, 30–60ms election) while all larger clusters use `e2e`. - Exact isolation passes in 4.43s; default-concurrency rerun passes in 13.33s, - and serial execution passes. Use the existing `e2e` fixture for the three-node - process case and retain its 3s acceptance assertion. Add the last HTTP - observation to store-wait failures, as already done for group waits. - Original logs remain in `restart-3n-1g-20260927-234030.467` under the ephemeral - Web E2E root; do not clean them during diagnosis. - -- Replica cleanup passes four regressions: failed group creation cleans its - newly created store, cleanup failure is explicit, existing replica hosts - are rejected before mutation, and automatic identity exhaustion returns - validation rather than panicking. Complete Console shared tests, Web replica - tests, fmt and clippy pass; helper extraction also passes all nine focused - creation/fan-out regressions. Logs: `/tmp/crowdb-replica-cleanup-*.log`. - -- Deletion baseline fails three of four cases: later replica hosts are skipped, - node-side failure reports success, and group deletion leaves replica records. - All four now pass, including idempotent node-side 404 and preservation of - sibling groups. Complete Console shared tests, affected Web migration/replica - tests, fmt and workspace clippy pass. Logs: `/tmp/crowdb-delete-*.log`. - -- Group fan-out baseline: both rejected remote wiring and missing peer endpoint - returned success. Both failure-injection cases now pass, and the complete - Console shared/Web gates passed for the group fix. Replica baseline adds - three failures: missing peer registration/address reports success, and a - missing new endpoint leaves its local group behind. Resolve all existing - peers before mutation and roll back the target when its endpoint is missing. - All five regressions pass, along with complete Console shared tests, affected - Web replica/migration tests, six browser store/reconfiguration flows, Rust - fmt and workspace clippy. Logs: `/tmp/crowdb-all-fanout-{shared,web,ui}.log`. - -- Complete Web integration gate passes after stable identity persistence. - Full Console UI passes all 86 component tests and 56 browser tests (4.6m), - including all five original failures. Logs: - `/tmp/crowdb-final-console-server.log`, `/tmp/crowdb-final-console-ui.log`. - -- Discovery integration passes duplicate seed submission, invalid origins, - exactly one live nonmember identity, no accidental membership, and restart - with persisted hints. Rust fmt and workspace clippy pass. -- Affected browser cases now pass: shell dialogs (11.9s), shell health (3.3s), - node inspection (8.3s), node cross-jump (2.7s), full-chain flow (4.8s), and - all three canvas cases (3.2s / 0.8s / 4.2s). The canvas assertion is scoped - to the main panel because the same unavailable text appears in a notification. -- Logical-tree hook regression passes confirmed reads, outage clearing, and - recovery. Full KV Server and Console shared-library gates pass. Complete - CLI/Web and browser gates remain pending for requirement completion. -- Full CLI passes. Web gate reached a failure in - `cluster_restart_incremental_test::restart_6node_2group_overlap`: after all - nodes restarted, group 11/1 did not converge to one leader within 3s. The - unchanged exact test passes alone in 13.27s. Preserve the original timeout; - compare the complete restart suite at default concurrency and serially, - without concurrent release compilation, before attributing the failure. - Logs: `/tmp/crowdb-discovery-console.log`, - `/tmp/crowdb-overlap-restart-isolated.log`. -- Generated registration identity changed across a normal restart in the - focused baseline (deterministic assertion failure). Persisting identity fixes - that case and the missed-unregister case; changed explicit IDs, changed node - IDs and malformed files fail closed. Full KV Server passes. The five restart - cases pass at default concurrency after the fix (13.55s); before the fix, - serial execution passed (44.73s) while default execution failed on different - groups. No election timeout or retry count changed. Full Web still remains. - -- Initial full Console UI baseline: 85 component tests pass; 51 browser tests - pass and five fail. Missing live registration affects node 203 in shell - replica creation and node 262 in node inspection. Two canvas assertions - expect an empty-store view before Group 0 exists; the full-chain flow records - logical-tree fetch errors during that same uninitialized phase. -- Isolated `12-cluster-node-inspect.spec.ts` reproduces HTTP 404 for node 262; - one test passes and one fails in 25.2 seconds. This is not solely an ordering - issue in the full browser suite. Keepalive uses its local bootstrap endpoint - when launched without seeds; cluster initialization currently waits only for - selected Group 0 members and does not propagate seeds to nonmembers. - -## Tests - -- Focused: KV server discovery/keepalive integration, console shared bootstrap - and operation tests, affected shell/node-inspection/canvas/full-chain specs. -- Full: `pixi run clean-env && pixi run test-console` and - `pixi run clean-env && pixi run test-console-ui`, sequentially. -- Crash diagnostics: monitor retention tests and disposable-container crash, - collector/export and exact-build source-line symbolization acceptance through - `pixi run test-monitor` and `pixi run test-single-node-container`. -- Style: `pixi run rs-fmt-check` and `pixi run rs-lint`. From edf0363c13440278dc74a6b71076e76a72975697 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 17:13:23 +0800 Subject: [PATCH 72/74] Align CI test tasks and manual acceptance workflows --- .github/workflows/ci.yml | 42 ++----------------- .github/workflows/docker-preview.yml | 41 ++++++++++++++++++ .github/workflows/iceberg-rust-sdk.yml | 42 +++++++++++++++++++ doc/working/test.md | 42 +++++++++++-------- pixi.toml | 5 ++- tools/README.md | 4 +- ...ask-coverage.py => check-ci-test-tasks.py} | 39 ++++++++++++----- tools/pixi-tasks/test-iceberg-sdk.sh | 1 - tools/pixi-tasks/test-pyiceberg-e2e.sh | 12 ++---- tools/pixi-tasks/test-suite.sh | 1 + tools/pixi-tasks/test-unit.sh | 1 + 11 files changed, 151 insertions(+), 79 deletions(-) create mode 100644 .github/workflows/docker-preview.yml create mode 100644 .github/workflows/iceberg-rust-sdk.yml rename tools/ci-checks/{check-test-task-coverage.py => check-ci-test-tasks.py} (74%) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9e7d3c7f..cf80c7d7 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -14,7 +14,7 @@ env: RUST_BACKTRACE: 1 RUST_LIB_BACKTRACE: 1 -# 10 parallel jobs. See doc/working/test.md § "Current CI Test Design" +# 9 parallel jobs. See doc/working/test.md § "Current CI Test Design" # for the assignment rule and how to add new test tasks. jobs: Lint: @@ -37,8 +37,8 @@ jobs: uses: Swatinem/rust-cache@v2 - name: Check formatting run: pixi run cargo fmt --all -- --check - - name: Check test-task coverage - run: pixi run test-task-coverage + - name: Check CI test tasks + run: pixi run check-ci-test-tasks - name: Run clippy run: pixi run cargo clippy --all-targets -- -D warnings @@ -395,39 +395,3 @@ jobs: - name: Clean subprocesses if: always() run: pixi run clean-env - - DockerPreview: - runs-on: ubuntu-24.04 - permissions: - contents: read - steps: - - uses: actions/checkout@v4 - with: - submodules: true - - name: Free disk space - run: | - sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache - sudo apt-get clean - df -h / - - uses: prefix-dev/setup-pixi@v0.8.1 - with: - pixi-version: latest - - name: Build and test single-node preview image - env: - CROWDB_PREVIEW_TEST_ARTIFACTS: ${{ runner.temp }}/crowdb-preview-artifacts - run: pixi run test-single-node-container - - name: Capture Docker diagnostics on failure - if: failure() - run: | - docker ps -a - docker images - docker info - df -h - - name: Upload preview failure logs - if: failure() - uses: actions/upload-artifact@v4 - with: - name: docker-preview-${{ github.run_attempt }} - path: ${{ runner.temp }}/crowdb-preview-artifacts - if-no-files-found: ignore - retention-days: 7 diff --git a/.github/workflows/docker-preview.yml b/.github/workflows/docker-preview.yml new file mode 100644 index 00000000..fcb0bed0 --- /dev/null +++ b/.github/workflows/docker-preview.yml @@ -0,0 +1,41 @@ +name: DockerPreview + +on: + workflow_dispatch: + +jobs: + DockerPreview: + runs-on: ubuntu-24.04 + permissions: + contents: read + steps: + - uses: actions/checkout@v4 + with: + submodules: true + - name: Free disk space + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache + sudo apt-get clean + df -h / + - uses: prefix-dev/setup-pixi@v0.8.1 + with: + pixi-version: latest + - name: Build and test single-node preview image + env: + CROWDB_PREVIEW_TEST_ARTIFACTS: ${{ runner.temp }}/crowdb-preview-artifacts + run: pixi run test-single-node-container + - name: Capture Docker diagnostics on failure + if: failure() + run: | + docker ps -a + docker images + docker info + df -h + - name: Upload preview failure logs + if: failure() + uses: actions/upload-artifact@v4 + with: + name: docker-preview-${{ github.run_attempt }} + path: ${{ runner.temp }}/crowdb-preview-artifacts + if-no-files-found: ignore + retention-days: 7 diff --git a/.github/workflows/iceberg-rust-sdk.yml b/.github/workflows/iceberg-rust-sdk.yml new file mode 100644 index 00000000..9e0c265e --- /dev/null +++ b/.github/workflows/iceberg-rust-sdk.yml @@ -0,0 +1,42 @@ +name: IcebergRustSDK + +on: + workflow_dispatch: + +env: + CARGO_TERM_COLOR: always + CARGO_INCREMENTAL: "0" + CARGO_PROFILE_DEV_DEBUG: line-tables-only + CARGO_PROFILE_TEST_DEBUG: line-tables-only + RUST_BACKTRACE: 1 + RUST_LIB_BACKTRACE: 1 + +jobs: + IcebergRustSDK: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + with: + submodules: true + - name: Free disk space + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /opt/hostedtoolcache + sudo apt-get clean + df -h / + - uses: prefix-dev/setup-pixi@v0.8.1 + with: + pixi-version: latest + - uses: Swatinem/rust-cache@v2 + - name: Run Rust Iceberg SDK acceptance + run: pixi run -e iceberg-e2e test-rust-iceberg-e2e + - name: Upload failure logs + if: failure() + uses: actions/upload-artifact@v4 + with: + name: runtime-iceberg-rust-sdk-${{ github.run_attempt }} + path: .crowdb-runtime/ + if-no-files-found: ignore + retention-days: 7 + - name: Clean subprocesses + if: always() + run: pixi run clean-env diff --git a/doc/working/test.md b/doc/working/test.md index b89e9c3a..e2647e54 100644 --- a/doc/working/test.md +++ b/doc/working/test.md @@ -16,7 +16,7 @@ For test strategy, layer scope, and coverage details, see [`design/kv/design-cro ## Current CI Test Design -CI uses ten parallel jobs, grouped by runtime requirements. Component tasks in +Regular CI uses nine parallel jobs, grouped by runtime requirements. Component tasks in `pixi.toml` select library, binary, and integration test targets with `--tests`; benchmark targets are excluded. Group scripts under `tools/pixi-tasks/` define execution order. GitHub Actions calls those group @@ -24,16 +24,15 @@ tasks. See [tools/README.md](../../tools/README.md) for the tooling map. | Job | Group task | Coverage | | ------------- | --------------------------------------- | ------------------------------------------------ | -| Lint | `test-task-coverage`, fmt, clippy | Package assignments and reachable CI tasks | +| Lint | `check-ci-test-tasks`, fmt, clippy | Package assignments and reachable CI tasks | | CppTests | `test-cpp` | C++ and Rust FFI | | UnitTests | `test-unit` | Rust libraries, including `test-access-iceberg` | | ServerTests | `test-server` | Native services, access server and monitor | | S3E2E | `-e s3-e2e test-boto3-e2e` | Access S3, access server and 17 boto3 cases | | IcebergE2E | `-e iceberg-e2e test-iceberg-e2e` | PyIceberg, native storage, GC and crash recovery | -| IcebergSDK | `-e iceberg-e2e test-iceberg-sdk` | Official Java/Rust SDKs and pinned Apache RCK | +| IcebergSDK | `-e iceberg-e2e test-iceberg-sdk` | Official Java SDK and pinned Apache RCK | | ConsoleTests | `test-console` | Shared operations, CLI and Web | | UITests | `test-console-ui` | Vitest and real-backend Playwright | -| DockerPreview | `test-single-node-container` | Linux amd64 image smoke and container E2E | Subprocess suites run sequentially inside each job and clean disposable runtime state. Iceberg jobs use the pinned `iceberg-e2e` Pixi environment for Python, @@ -41,18 +40,27 @@ Maven and Java; Rust/native builds use the default environment. The RCK task fetches and verifies its exact Apache Iceberg source revision. Test-only child listener functions remain ignored and are invoked by their parent crash tests. -`test-suite` runs the host groups, including both Iceberg groups. Docker is a -separate explicit `test-single-node-container` task requiring a Linux amd64 -Docker host. It is always included in the DockerPreview CI job. +`test-suite` runs the host groups, including both Iceberg groups and the Rust +SDK task. The Rust SDK task is available through +`pixi run -e iceberg-e2e test-rust-iceberg-e2e` and the manual-only +`IcebergRustSDK` workflow. It does not run on regular pushes or pull requests. +DockerPreview is a manual-only workflow that runs +`pixi run test-single-node-container` on a Linux amd64 Docker host. The release +workflow also runs this image test before publication. -### Coverage guard +IcebergE2E uses the release profile for PyIceberg and native acceptance. The +access-server component suite runs in ServerTests, so IcebergE2E does not run it +again. -`pixi run test-task-coverage` validates every workspace package against -`TASK_PACKAGES` in `tools/ci-checks/check-test-task-coverage.py`, including the +### CI test-task check + +`pixi run check-ci-test-tasks` validates every workspace package against +`TASK_PACKAGES` in `tools/ci-checks/check-ci-test-tasks.py`, including the test harness's own runtime-namespace tests. -The guard follows Pixi group calls and checked-in shell scripts from CI, so an -existing component task disconnected from its job fails validation. It also -requires explicit CI reachability for the container and client acceptance tasks. +The guard follows Pixi group calls and checked-in shell scripts from CI, so a +required component task disconnected from its job fails validation. It also +checks that DockerPreview and IcebergRustSDK are reachable from their manual +workflows and absent from regular CI. ### Adding tests @@ -62,7 +70,7 @@ requires explicit CI reachability for the container and client acceptance tasks. 3. Add the component to the group script matching its runtime requirements. 4. Feature-gated or ignored tests require explicit task selectors. Do not count compiling an ignored test as executing it; exclude subprocess helper entries. -5. Run `pixi run test-task-coverage`, the affected suites, and workflow validation. +5. Run `pixi run check-ci-test-tasks`, the affected suites, and workflow validation. 6. Measure changed suites with `pixi run bash tools/test-metrics/measure.sh TASK...` and update the timing table. Environment selection is automatic. @@ -75,9 +83,9 @@ reported test counts and exit codes are saved under `.crowdb-runtime/artifacts/measure-tests/`. Timing includes incremental builds and subprocess startup/shutdown, so feature changes and cold builds affect it. Counts are runner-reported cases, not assertions; ignored cases are excluded. -Native Iceberg and Java/Rust/RCK SDK acceptance use release binaries, matching -the published container profile. Component suites retain their default test -profile. The focused debug native 100 MiB multipart upload, completion, +IcebergE2E and Java/Rust/RCK SDK acceptance use release binaries, matching +the published container profile. Other component suites retain their default +test profile. The focused debug native 100 MiB multipart upload, completion, restart, replay, and full read passed on 2026-09-29 in 54.40 s. Its previous 10 s completion deadline failure did not recur after the streaming I/O changes. diff --git a/pixi.toml b/pixi.toml index c6b6194d..826d9fcf 100644 --- a/pixi.toml +++ b/pixi.toml @@ -158,8 +158,8 @@ clean-env = "bash tools/runtime/clean-runtime.sh env" # Optional full-workspace compile check. CI test groups use their component # tasks to avoid building unrelated test targets on each runner. build-tests = "cargo test --workspace --all-targets --no-run" -# Verify every workspace package with tests is assigned to a CI test task. -test-task-coverage = "python tools/ci-checks/check-test-task-coverage.py" +# Verify every workspace package is assigned to a test task reachable from CI. +check-ci-test-tasks = "python tools/ci-checks/check-ci-test-tasks.py" # ── Lint job ── # (uses rs-fmt / rs-lint / tree-lint defined above, no separate test-* tasks) @@ -184,6 +184,7 @@ test-chunk-kv = { cmd = "cargo test -p crowdb-chunk-kv --tests" } test-chunk-stream = { cmd = "cargo test -p crowdb-chunk-stream --tests" } test-chunk-kv-client = { cmd = "cargo test -p crowdb-chunk-kv-client --tests" } test-chunk-kv-server = { cmd = "cargo test -p crowdb-chunk-kv-server --tests" } +test-access-multipart = { cmd = "cargo test -p crowdb-access-multipart --tests" } test-unit = { cmd = "bash tools/pixi-tasks/test-unit.sh" } # ── ServerTests job: spawns crowdb-kv-server / crowdb-diskdb / crowdb-diskio ── diff --git a/tools/README.md b/tools/README.md index 600abd3b..a7c3e473 100644 --- a/tools/README.md +++ b/tools/README.md @@ -27,7 +27,7 @@ Run commands through `pixi run`. Read only the directory relevant to the task. Common entry points: ```sh -pixi run test-task-coverage +pixi run check-ci-test-tasks pixi run check-version pixi run clean-env pixi run bash tools/test-metrics/measure.sh test-access-iceberg test-monitor @@ -35,7 +35,7 @@ pixi run bash tools/test-metrics/measure.sh test-access-iceberg test-monitor For test ownership and timings, read [`doc/working/test.md`](../doc/working/test.md). CI calls Pixi group tasks; -update the group script when adding a component, then run `test-task-coverage`. +update the group script when adding a component, then run `check-ci-test-tasks`. Feature-gated or ignored client tests need explicit task selection. Container packaging and its acceptance scripts live under diff --git a/tools/ci-checks/check-test-task-coverage.py b/tools/ci-checks/check-ci-test-tasks.py similarity index 74% rename from tools/ci-checks/check-test-task-coverage.py rename to tools/ci-checks/check-ci-test-tasks.py index 190fd6ac..14b93150 100644 --- a/tools/ci-checks/check-test-task-coverage.py +++ b/tools/ci-checks/check-ci-test-tasks.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Check that every tested Rust workspace package has a CI task assignment.""" +"""Check workspace package assignments and CI test-task reachability.""" import json import subprocess @@ -26,6 +26,7 @@ "test-chunkdb": {"crowdb-chunkdb"}, "test-chunk-client": {"crowdb-chunk-client"}, "test-diskio-client": {"crowdb-diskio-client"}, + "test-access-multipart": {"crowdb-access-multipart"}, "test-access-s3": {"crowdb-access-s3"}, "test-access-iceberg": {"crowdb-access-iceberg"}, "test-access-server": {"crowdb-access-server"}, @@ -62,10 +63,10 @@ def main() -> int: missing = sorted(packages - set(assignments) - set(SUPPORT_PACKAGES)) unknown_support = sorted(set(SUPPORT_PACKAGES) - packages) if missing: - print("Rust workspace packages missing from CI test-task coverage:") + print("Rust workspace packages missing from CI test tasks:") for package in missing: print(f" {package}") - print("Add the package to TASK_PACKAGES in tools/ci-checks/check-test-task-coverage.py") + print("Add the package to TASK_PACKAGES in tools/ci-checks/check-ci-test-tasks.py") return 1 if unknown_support: print("Support-package allowlist contains packages not in the workspace:") @@ -84,7 +85,7 @@ def main() -> int: } missing_tasks = [task for task in TASK_PACKAGES if task not in pixi_tasks] if missing_tasks: - print("Coverage map references missing Pixi tasks:") + print("CI test-task map references missing Pixi tasks:") for task in missing_tasks: print(f" {task}") return 1 @@ -95,32 +96,50 @@ def main() -> int: if f"-p {package}" not in pixi_tasks[task] ] if missing_commands: - print("Coverage map packages are not targeted by their Pixi tasks:") + print("CI test-task map packages are not targeted by their Pixi tasks:") for task, package in missing_commands: print(f" {task}: {package}") return 1 - reachable = graph.reachable((root / ".github/workflows/ci.yml").read_text()) + workflows = root / ".github/workflows" + reachable = graph.reachable((workflows / "ci.yml").read_text()) required = {("default", task) for task in TASK_PACKAGES} required.update({ - ("default", "test-single-node-container"), ("default", "test-console-ui"), ("s3-e2e", "test-boto3-e2e"), ("iceberg-e2e", "test-pyiceberg-e2e"), ("iceberg-e2e", "test-iceberg-native"), ("iceberg-e2e", "test-java-iceberg-e2e"), ("iceberg-e2e", "test-java-iceberg-fileio-e2e"), - ("iceberg-e2e", "test-rust-iceberg-e2e"), ("iceberg-e2e", "test-iceberg-rck"), }) unreachable = sorted(required - reachable) if unreachable: - print("Test tasks not reachable from CI:") + print("Test tasks not reachable from regular CI:") for environment, task in unreachable: print(f" {environment}: {task}") return 1 - print(f"Test-task coverage verified for {len(packages)} workspace packages") + manual_tasks = { + "docker-preview.yml": {("default", "test-single-node-container")}, + "iceberg-rust-sdk.yml": {("iceberg-e2e", "test-rust-iceberg-e2e")}, + } + for workflow, tasks in manual_tasks.items(): + manual_reachable = graph.reachable((workflows / workflow).read_text()) + missing = sorted(tasks - manual_reachable) + if missing: + print(f"Manual workflow {workflow} does not reach its test tasks:") + for environment, task in missing: + print(f" {environment}: {task}") + return 1 + unexpected = sorted(tasks & reachable) + if unexpected: + print("Manual-only test tasks also reachable from regular CI:") + for environment, task in unexpected: + print(f" {environment}: {task}") + return 1 + + print(f"CI test tasks verified for {len(packages)} workspace packages") for package in sorted(assignments): print(f" {package}: {assignments[package]}") for package, reason in sorted(SUPPORT_PACKAGES.items()): diff --git a/tools/pixi-tasks/test-iceberg-sdk.sh b/tools/pixi-tasks/test-iceberg-sdk.sh index c245215f..ae9e8efc 100644 --- a/tools/pixi-tasks/test-iceberg-sdk.sh +++ b/tools/pixi-tasks/test-iceberg-sdk.sh @@ -5,5 +5,4 @@ set -euo pipefail cd "${PIXI_PROJECT_ROOT:?}" pixi run -e iceberg-e2e test-java-iceberg-e2e -pixi run -e iceberg-e2e test-rust-iceberg-e2e pixi run -e iceberg-e2e test-iceberg-rck diff --git a/tools/pixi-tasks/test-pyiceberg-e2e.sh b/tools/pixi-tasks/test-pyiceberg-e2e.sh index 20ad924a..9657e85b 100644 --- a/tools/pixi-tasks/test-pyiceberg-e2e.sh +++ b/tools/pixi-tasks/test-pyiceberg-e2e.sh @@ -4,14 +4,10 @@ set -euo pipefail cd "${PIXI_PROJECT_ROOT:?}" -pixi run -e default -- cmake -S app/crowdb-diskio -B app/crowdb-diskio/build -DCMAKE_BUILD_TYPE=Release -pixi run -e default -- cmake --build app/crowdb-diskio/build -j 4 --target crowdb-diskio -pixi run -e default -- cargo build -p crowdb-kv-server -p crowdb-diskdb -p crowdb-chunkdb -p crowdb-chunk-kv-server -p crowdb-access-server -pixi run -e default clean-env -pixi run -e default test-access-server -CROWDB_RUNTIME_ROOT="$PIXI_PROJECT_ROOT/.crowdb-runtime/ephemeral/iceberg-e2e" CROWDB_ICEBERG_E2E_PYTHON="$PIXI_PROJECT_ROOT/.pixi/envs/iceberg-e2e/bin/python" pixi run -e default -- cargo test -p crowdb-access-server --features iceberg-e2e --test iceberg_full_stack_test -- --nocapture +source tools/pixi-tasks/prepare-iceberg.sh +CROWDB_RUNTIME_ROOT="$PIXI_PROJECT_ROOT/.crowdb-runtime/ephemeral/iceberg-e2e" CROWDB_ICEBERG_E2E_PYTHON="$PIXI_PROJECT_ROOT/.pixi/envs/iceberg-e2e/bin/python" pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e --test iceberg_full_stack_test -- --nocapture -CROWDB_ICEBERG_E2E_PYTHON="$CONDA_PREFIX/bin/python" pixi run -e default -- cargo test -p crowdb-access-server --features iceberg-e2e \ +CROWDB_ICEBERG_E2E_PYTHON="$CONDA_PREFIX/bin/python" pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e \ --test iceberg_namespace_sdk_test official_complete_listing_rejects_each_spool_limit_and_releases_resources -- --ignored --exact -CROWDB_ICEBERG_E2E_PYTHON="$CONDA_PREFIX/bin/python" pixi run -e default -- cargo test -p crowdb-access-server --features iceberg-e2e \ +CROWDB_ICEBERG_E2E_PYTHON="$CONDA_PREFIX/bin/python" pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e \ --test iceberg_gc_control_test official_sdk_foreground_progresses_under_gc_backlog -- --ignored --exact diff --git a/tools/pixi-tasks/test-suite.sh b/tools/pixi-tasks/test-suite.sh index 30d67f64..c00831e1 100644 --- a/tools/pixi-tasks/test-suite.sh +++ b/tools/pixi-tasks/test-suite.sh @@ -10,6 +10,7 @@ pixi run test-server pixi run -e s3-e2e test-boto3-e2e pixi run -e iceberg-e2e test-iceberg-e2e pixi run -e iceberg-e2e test-iceberg-sdk +pixi run -e iceberg-e2e test-rust-iceberg-e2e pixi run test-console pixi run clean-env pixi run test-console-ui diff --git a/tools/pixi-tasks/test-unit.sh b/tools/pixi-tasks/test-unit.sh index 5e1d7205..6b8493d6 100644 --- a/tools/pixi-tasks/test-unit.sh +++ b/tools/pixi-tasks/test-unit.sh @@ -14,4 +14,5 @@ pixi run test-chunk-kv pixi run test-chunk-stream pixi run test-chunk-kv-client pixi run test-chunk-kv-server +pixi run test-access-multipart pixi run test-access-iceberg From 08ac54d934c887cdea608b094aa2595bb08dc737 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 17:17:26 +0800 Subject: [PATCH 73/74] Update preview release policy for manual workflow --- .../single-node-container/tests/release-policy.sh | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/container/single-node-container/tests/release-policy.sh b/container/single-node-container/tests/release-policy.sh index 51d5917e..cac10ab8 100644 --- a/container/single-node-container/tests/release-policy.sh +++ b/container/single-node-container/tests/release-policy.sh @@ -2,7 +2,7 @@ set -euo pipefail release=.github/workflows/release-container.yml -ci=.github/workflows/ci.yml +preview=.github/workflows/docker-preview.yml events=$(sed -n '/^on:/,/^concurrency:/p' "$release") [[ "$events" == *'workflow_dispatch:'* ]] @@ -55,10 +55,13 @@ done [[ "$publish_job" != *'RELEASE_ENABLED'* ]] [[ "$publish_job" == *'[[ "$(git rev-parse HEAD)" == "$REVISION" ]]'* ]] -ci_job=$(sed -n '/^ DockerPreview:/,$p' "$ci") -[[ "$ci_job" == *'contents: read'* && "$ci_job" == *'pixi run test-single-node-container'* ]] -[[ "$ci_job" == *'Upload preview failure logs'* && "$ci_job" == *'CROWDB_PREVIEW_TEST_ARTIFACTS'* ]] -! grep -Eq 'secrets\.|docker/login-action|docker/build-push-action' <<<"$ci_job" +preview_events=$(sed -n '/^on:/,/^jobs:/p' "$preview") +[[ "$preview_events" == *'workflow_dispatch:'* ]] +! grep -Eq '^ (push|pull_request|release|create):' <<<"$preview_events" +preview_job=$(sed -n '/^ DockerPreview:/,$p' "$preview") +[[ "$preview_job" == *'contents: read'* && "$preview_job" == *'pixi run test-single-node-container'* ]] +[[ "$preview_job" == *'Upload preview failure logs'* && "$preview_job" == *'CROWDB_PREVIEW_TEST_ARTIFACTS'* ]] +! grep -Eq 'secrets\.|docker/login-action|docker/build-push-action' <<<"$preview_job" release_tool=tools/release.py for required in '--dry-run' '--execute' '--symbols' 'git", "push", "--atomic"' \ From 5ae55920680fdf9329a9924e24d79d145d39b491 Mon Sep 17 00:00:00 2001 From: Gian Date: Tue, 29 Sep 2026 17:54:33 +0800 Subject: [PATCH 74/74] Make release and Iceberg SDK verification reproducible --- .github/workflows/ci.yml | 4 +- .github/workflows/release-container.yml | 154 ++++++++++++------ container/single-node-container/Dockerfile | 4 +- container/single-node-container/README.md | 29 +++- .../tests/release-policy.sh | 33 ++-- tools/pixi-tasks/test-iceberg-rck.sh | 11 +- tools/release.py | 16 +- 7 files changed, 173 insertions(+), 78 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index cf80c7d7..20ef248c 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -280,8 +280,8 @@ jobs: name: runtime-icebergsdk-${{ github.run_attempt }} path: | .crowdb-runtime/ - target/iceberg-rck/open-api/build/reports/tests/ - target/iceberg-rck/open-api/build/test-results/ + ${{ runner.temp }}/iceberg-rck/open-api/build/reports/tests/ + ${{ runner.temp }}/iceberg-rck/open-api/build/test-results/ if-no-files-found: ignore retention-days: 7 - name: Clean subprocesses diff --git a/.github/workflows/release-container.yml b/.github/workflows/release-container.yml index dc3e9e05..01d96adc 100644 --- a/.github/workflows/release-container.yml +++ b/.github/workflows/release-container.yml @@ -3,9 +3,9 @@ name: Publish crowdb-iceberge docker container on: workflow_dispatch: inputs: - tag: - description: Existing Git release tag to publish - required: true + candidate_sha: + description: Expected main commit (optional when started manually) + required: false type: string include_symbols: description: Build and attach the large exact-build symbol archive @@ -27,13 +27,16 @@ jobs: CROWDB_PACKAGE_SYMBOLS: ${{ inputs.include_symbols && '1' || '0' }} permissions: contents: read + actions: read outputs: version: ${{ steps.source.outputs.version }} revision: ${{ steps.source.outputs.revision }} + tag: ${{ steps.source.outputs.tag }} + runtime_sha256: ${{ steps.runtime_digest.outputs.sha256 }} steps: - uses: actions/checkout@v4 with: - ref: ${{ inputs.tag }} + ref: ${{ github.sha }} fetch-depth: 0 submodules: true - uses: prefix-dev/setup-pixi@v0.8.1 @@ -44,17 +47,17 @@ jobs: - name: Verify release source id: source env: - RELEASE_TAG: ${{ inputs.tag }} - GH_TOKEN: ${{ github.token }} + CANDIDATE_SHA: ${{ inputs.candidate_sha }} run: | pixi run bash -euc ' [[ "$GITHUB_REPOSITORY" == buzzcrow/crowdb ]] - [[ "$RELEASE_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+(-[0-9A-Za-z.-]+)?$ ]] - [[ "$RELEASE_TAG" == "v$(cat VERSION)" ]] - revision=$(git rev-parse --verify "refs/tags/$RELEASE_TAG^{commit}") - [[ "$revision" == "$(git rev-parse HEAD)" ]] - [[ "$(gh release view "$RELEASE_TAG" --json isDraft --jq .isDraft)" == true ]] - printf "version=%s\nrevision=%s\n" "${RELEASE_TAG#v}" "$revision" >> "$GITHUB_OUTPUT" + [[ "$GITHUB_REF" == refs/heads/main ]] + version=$(cat VERSION) + [[ "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]] + revision=$(git rev-parse HEAD) + [[ "$revision" == "$GITHUB_SHA" ]] + [[ -z "$CANDIDATE_SHA" || "$revision" == "$CANDIDATE_SHA" ]] + printf "version=%s\nrevision=%s\ntag=v%s\n" "$version" "$revision" "$version" >> "$GITHUB_OUTPUT" ' - name: Free disk space run: | @@ -65,39 +68,37 @@ jobs: env: CROWDB_PREVIEW_TEST_ARTIFACTS: ${{ runner.temp }}/crowdb-preview-artifacts run: pixi run test-single-node-container - - name: Free preview build artifacts before acceptance tests - run: | - pixi run docker image rm crowdb-iceberg-single-node:dev - pixi run docker builder prune --all --force - pixi run cargo clean --release - pixi run df -h / - - name: Run S3 client acceptance - run: pixi run clean-env && pixi run -e s3-e2e test-boto3-e2e - - name: Run Iceberg client acceptance - run: pixi run clean-env && pixi run -e iceberg-e2e test-pyiceberg-e2e - - name: Run console acceptance - run: pixi run clean-env && pixi run test-console - - name: Require installed system browser + - name: Require CI success for the release commit + timeout-minutes: 90 + env: + GH_TOKEN: ${{ github.token }} + REVISION: ${{ steps.source.outputs.revision }} run: | pixi run bash -euc ' - for browser in /snap/bin/chromium /usr/bin/chromium /usr/bin/chromium-browser /usr/bin/google-chrome /usr/bin/google-chrome-stable /usr/bin/microsoft-edge; do - [[ ! -x "$browser" ]] || exit 0 + for attempt in {1..180}; do + state=$(gh api "repos/$GITHUB_REPOSITORY/actions/workflows/ci.yml/runs?head_sha=$REVISION&branch=main&event=push&per_page=20" \ + --jq ".workflow_runs | map(select(.head_sha == \"$REVISION\" and .head_branch == \"main\" and .event == \"push\")) | sort_by(.run_number, .run_attempt) | last | if . == null then \"pending\" else .status + \"/\" + (.conclusion // \"pending\") end") + case "$state" in + completed/success) exit 0 ;; + completed/*) echo "CI failed for $REVISION: $state" >&2; exit 1 ;; + esac + echo "Waiting for CI on $REVISION: $state" + sleep 30 done - echo "Release runner requires an installed system browser" >&2 + echo "Timed out waiting for CI on $REVISION" >&2 exit 1 ' - - name: Run console UI acceptance - run: pixi run clean-env && pixi run test-console-ui - - name: Check Rust formatting and lint - run: pixi run rs-fmt-check && pixi run rs-lint - name: Verify exact-build source-line symbols if: inputs.include_symbols run: pixi run -- python tools/ci-checks/check-container-symbols.py - name: Archive verified runtime files run: pixi run tar -C target/container-runtime -czf target/container-runtime.tar.gz . + - name: Record verified runtime digest + run: pixi run bash -euc 'printf "sha256=%s\n" "$(sha256sum target/container-runtime.tar.gz | cut -d " " -f1)" >> "$GITHUB_OUTPUT"' + id: runtime_digest - name: Archive exact-build symbols if: inputs.include_symbols - run: pixi run tar -C target/container-symbols -I zstd -cf "target/crowdb-symbols-${{ inputs.tag }}-git-${{ steps.source.outputs.revision }}-linux-amd64.tar.zst" . + run: pixi run tar -C target/container-symbols -I zstd -cf "target/crowdb-symbols-${{ steps.source.outputs.tag }}-git-${{ steps.source.outputs.revision }}-linux-amd64.tar.zst" . - uses: actions/upload-artifact@v4 with: name: verified-container-runtime @@ -132,15 +133,15 @@ jobs: steps: - uses: actions/checkout@v4 with: - ref: ${{ inputs.tag }} + ref: ${{ needs.verify.outputs.revision }} fetch-depth: 0 submodules: true - uses: prefix-dev/setup-pixi@v0.8.1 with: pixi-version: latest - - name: Require publication credentials and unused immutable tags + - name: Require publication credentials and current release source env: - RELEASE_TAG: ${{ inputs.tag }} + RELEASE_TAG: ${{ needs.verify.outputs.tag }} REVISION: ${{ needs.verify.outputs.revision }} DOCKERHUB_USERNAME: ${{ vars.DOCKERHUB_USERNAME }} DOCKERHUB_TOKEN: ${{ secrets.DOCKERHUB_TOKEN }} @@ -148,11 +149,8 @@ jobs: pixi run bash -euc ' [[ -n "$DOCKERHUB_USERNAME" && -n "$DOCKERHUB_TOKEN" ]] [[ "$(git rev-parse HEAD)" == "$REVISION" ]] - for tag in "$RELEASE_TAG" "git-$REVISION"; do - status=$(curl --silent --show-error --output /dev/null --write-out "%{http_code}" \ - "https://hub.docker.com/v2/namespaces/crowdb/repositories/crowdb-iceberg/tags/$tag") - [[ "$status" == 404 ]] || { echo "Immutable tag $tag is present or registry unavailable (HTTP $status)" >&2; exit 1; } - done + [[ "$(git ls-remote origin refs/heads/main | cut -f1)" == "$REVISION" ]] + [[ "$RELEASE_TAG" == "v$(cat VERSION)" ]] ' - uses: docker/setup-buildx-action@v4 - uses: actions/download-artifact@v4 @@ -167,15 +165,70 @@ jobs: name: verified-container-symbols path: target - name: Extract verified runtime files + env: + RUNTIME_SHA256: ${{ needs.verify.outputs.runtime_sha256 }} run: | + pixi run bash -euc '[[ "$(sha256sum target/container-runtime.tar.gz | cut -d " " -f1)" == "$RUNTIME_SHA256" ]]' pixi run bash -euc 'mkdir -p target/container-runtime && tar -C target/container-runtime -xzf target/container-runtime.tar.gz' pixi run rm target/container-runtime.tar.gz - uses: docker/login-action@v4 with: username: ${{ vars.DOCKERHUB_USERNAME }} password: ${{ secrets.DOCKERHUB_TOKEN }} + - name: Check immutable registry tags for safe retry + id: registry + env: + RELEASE_TAG: ${{ needs.verify.outputs.tag }} + REVISION: ${{ needs.verify.outputs.revision }} + VERSION: ${{ needs.verify.outputs.version }} + RUNTIME_SHA256: ${{ needs.verify.outputs.runtime_sha256 }} + run: | + pixi run bash -euc ' + for tag in "$RELEASE_TAG" "git-$REVISION"; do + status=$(curl --silent --show-error --output /dev/null --write-out "%{http_code}" \ + "https://hub.docker.com/v2/namespaces/crowdb/repositories/crowdb-iceberg/tags/$tag") + case "$status" in + 200|404) printf "%s=%s\n" "$tag" "$status" ;; + *) echo "Registry check failed for $tag (HTTP $status)" >&2; exit 1 ;; + esac + if [[ "$tag" == "$RELEASE_TAG" ]]; then version_status=$status; else revision_status=$status; fi + done + if [[ "$version_status" == 404 && "$revision_status" == 404 ]]; then + echo "reuse=false" >> "$GITHUB_OUTPUT" + exit 0 + fi + [[ "$version_status" == 200 && "$revision_status" == 200 ]] || { + echo "Only one immutable registry tag exists" >&2; exit 1; + } + image=docker.io/crowdb/crowdb-iceberg + version_digest=$(docker buildx imagetools inspect "$image:$RELEASE_TAG" --format "{{json .Manifest}}" | jq -er .digest) + revision_digest=$(docker buildx imagetools inspect "$image:git-$REVISION" --format "{{json .Manifest}}" | jq -er .digest) + [[ "$version_digest" == "$revision_digest" ]] + docker pull --platform linux/amd64 "$image:$RELEASE_TAG" + [[ "$(docker image inspect --format "{{index .Config.Labels \"org.opencontainers.image.revision\"}}" "$image:$RELEASE_TAG")" == "$REVISION" ]] + [[ "$(docker image inspect --format "{{index .Config.Labels \"org.opencontainers.image.version\"}}" "$image:$RELEASE_TAG")" == "$VERSION" ]] + [[ "$(docker image inspect --format "{{index .Config.Labels \"org.crowdb.runtime.sha256\"}}" "$image:$RELEASE_TAG")" == "$RUNTIME_SHA256" ]] + printf "reuse=true\ndigest=%s\n" "$version_digest" >> "$GITHUB_OUTPUT" + ' + - name: Create release tag after verification + env: + RELEASE_TAG: ${{ needs.verify.outputs.tag }} + REVISION: ${{ needs.verify.outputs.revision }} + run: | + pixi run bash -euc ' + git config user.name github-actions[bot] + git config user.email 41898282+github-actions[bot]@users.noreply.github.com + if git ls-remote --exit-code origin "refs/tags/$RELEASE_TAG" >/dev/null; then + git fetch origin "refs/tags/$RELEASE_TAG:refs/tags/$RELEASE_TAG" + [[ "$(git rev-parse "refs/tags/$RELEASE_TAG^{commit}")" == "$REVISION" ]] + else + git tag -a "$RELEASE_TAG" -m "$RELEASE_TAG" "$REVISION" + git push origin "refs/tags/$RELEASE_TAG" + fi + ' - name: Build and publish signed-source image with attestations id: build + if: steps.registry.outputs.reuse != 'true' uses: docker/build-push-action@v7 with: context: target/container-runtime @@ -187,26 +240,35 @@ jobs: build-args: | SOURCE_REVISION=${{ needs.verify.outputs.revision }} PREVIEW_VERSION=${{ needs.verify.outputs.version }} + RUNTIME_SHA256=${{ needs.verify.outputs.runtime_sha256 }} tags: | - docker.io/crowdb/crowdb-iceberg:${{ inputs.tag }} + docker.io/crowdb/crowdb-iceberg:${{ needs.verify.outputs.tag }} docker.io/crowdb/crowdb-iceberg:git-${{ needs.verify.outputs.revision }} - uses: sigstore/cosign-installer@v4.1.2 - name: Sign published digest env: - DIGEST: ${{ steps.build.outputs.digest }} + DIGEST: ${{ steps.registry.outputs.reuse == 'true' && steps.registry.outputs.digest || steps.build.outputs.digest }} run: pixi run cosign sign --yes "docker.io/crowdb/crowdb-iceberg@$DIGEST" - name: Publish verified GitHub Release env: GH_TOKEN: ${{ github.token }} - RELEASE_TAG: ${{ inputs.tag }} - run: pixi run gh release edit "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --draft=false + RELEASE_TAG: ${{ needs.verify.outputs.tag }} + run: | + pixi run bash -euc ' + if gh release view "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --json isDraft --jq .isDraft >/dev/null 2>&1; then + [[ "$(gh release view "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --json isDraft --jq .isDraft)" == true ]] + else + gh release create "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --verify-tag --generate-notes --draft + fi + gh release edit "$RELEASE_TAG" --repo "$GITHUB_REPOSITORY" --draft=false + ' - name: Attach exact-build symbols to GitHub Release id: symbol_upload if: inputs.include_symbols && steps.symbols_download.outcome == 'success' continue-on-error: true env: GH_TOKEN: ${{ github.token }} - RELEASE_TAG: ${{ inputs.tag }} + RELEASE_TAG: ${{ needs.verify.outputs.tag }} REVISION: ${{ needs.verify.outputs.revision }} run: pixi run gh release upload "$RELEASE_TAG" "target/crowdb-symbols-$RELEASE_TAG-git-$REVISION-linux-amd64.tar.zst" --repo "$GITHUB_REPOSITORY" - name: Report optional symbol upload failure diff --git a/container/single-node-container/Dockerfile b/container/single-node-container/Dockerfile index db18caef..c32d125b 100644 --- a/container/single-node-container/Dockerfile +++ b/container/single-node-container/Dockerfile @@ -21,6 +21,7 @@ RUN chmod 0755 /opt/crowdb/bin/entrypoint \ ARG SOURCE_REVISION ARG PREVIEW_VERSION +ARG RUNTIME_SHA256 RUN --mount=type=bind,source=.,target=/staged,ro \ test -n "$SOURCE_REVISION" && test -n "$PREVIEW_VERSION" \ && test "$(cat /staged/SOURCE_REVISION)" = "$SOURCE_REVISION" \ @@ -32,7 +33,8 @@ LABEL org.opencontainers.image.title="CROWDB Iceberg" \ org.opencontainers.image.authors="Gian " \ org.opencontainers.image.licenses="Apache-2.0" \ org.opencontainers.image.revision="$SOURCE_REVISION" \ - org.opencontainers.image.version="$PREVIEW_VERSION" + org.opencontainers.image.version="$PREVIEW_VERSION" \ + org.crowdb.runtime.sha256="$RUNTIME_SHA256" ENV PATH="/opt/crowdb/bin:${PATH}" \ LD_LIBRARY_PATH="/opt/crowdb/lib" \ CROWDB_RUNTIME_ROOT="/opt/crowdb/run" diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index 57dfe7bf..8fb2891c 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -34,14 +34,27 @@ pixi run -- python tools/release.py --execute ``` `--bump minor` and `--bump major` select larger version changes. The script -updates every version manifest, commits and tags the release, atomically pushes -`main` and the tag, creates a draft GitHub Release, then dispatches the existing -verified DockerHub workflow. Add `--symbols` to either command to include the -large exact-build symbol archive; the default release skips it. Execution -requires authenticated `gh` and GitHub -permission to push `main`; the dry run changes no files or remote state. The -workflow archives the verified runtime, then packages those same files in its -publish job without recompiling them. With `--symbols`, it also archives +updates every version manifest, commits and pushes the candidate to `main`, +then dispatches the release workflow. It does not create a tag or GitHub +Release. You can also run the workflow manually on `main`; it derives the tag +from `VERSION`, so no tag input is needed. Add `--symbols` to either command +to include the large exact-build symbol archive; the default release skips it. +Execution requires authenticated `gh` and GitHub permission to push `main`; +the dry run changes no files or remote state. +The script passes its candidate commit SHA to the workflow so a later push to +`main` cannot silently change which commit gets published. + +The workflow checks that CI passed for the exact candidate commit, builds and +tests the container, then waits for DockerHub environment approval. Only after +verification does it create the Git tag and publish the signed Docker image +and GitHub Release. A failed verification leaves no tag or draft release. Fix +the candidate and run the workflow again; if code changes after a tag was +created, use the next patch version. A publication retry for the same commit +reuses an existing image only when both registry tags have the same digest and +the image labels match the release version, commit, and verified runtime archive. + +The workflow archives the verified runtime, then packages those same files in +its publish job without recompiling them. With `--symbols`, it also archives exact-build symbols from that build and attaches `crowdb-symbols--git--linux-amd64.tar.zst` to the GitHub Release. The workflow publishes the GitHub Release after the Docker image and diff --git a/container/single-node-container/tests/release-policy.sh b/container/single-node-container/tests/release-policy.sh index cac10ab8..8fc721df 100644 --- a/container/single-node-container/tests/release-policy.sh +++ b/container/single-node-container/tests/release-policy.sh @@ -11,14 +11,23 @@ events=$(sed -n '/^on:/,/^concurrency:/p' "$release") for required in \ 'environment: DockerHub' \ 'DOCKERHUB_TOKEN' \ - 'ref: ${{ inputs.tag }}' \ - 'git rev-parse --verify "refs/tags/$RELEASE_TAG^{commit}"' \ - '[[ "$revision" == "$(git rev-parse HEAD)" ]]' \ + 'ref: ${{ github.sha }}' \ + 'ref: ${{ needs.verify.outputs.revision }}' \ + 'git rev-parse "refs/tags/$RELEASE_TAG^{commit}"' \ + '[[ "$revision" == "$GITHUB_SHA" ]]' \ 'gh release view "$RELEASE_TAG"' \ - '== true ]]' \ - '[[ "$status" == 404 ]]' \ + 'gh release create "$RELEASE_TAG"' \ + 'git push origin "refs/tags/$RELEASE_TAG"' \ + 'actions: read' \ + 'head_sha=$REVISION&branch=main&event=push' \ + 'completed/success) exit 0' \ + 'reuse=false' \ + 'reuse=true' \ + 'runtime_sha256: ${{ steps.runtime_digest.outputs.sha256 }}' \ + 'RUNTIME_SHA256=${{ needs.verify.outputs.runtime_sha256 }}' \ + 'org.crowdb.runtime.sha256' \ 'needs: verify' \ - 'docker.io/crowdb/crowdb-iceberg:${{ inputs.tag }}' \ + 'docker.io/crowdb/crowdb-iceberg:${{ needs.verify.outputs.tag }}' \ 'docker.io/crowdb/crowdb-iceberg:git-${{ needs.verify.outputs.revision }}' \ 'provenance: mode=max' \ 'sbom: true' \ @@ -29,6 +38,7 @@ done [[ $(grep -c 'push: true' "$release") == 1 ]] [[ $(grep -c 'id-token: write' "$release") == 1 ]] [[ "$events" != *'schedule:'* ]] +[[ "$events" != *' tag:'* ]] verify_job=$(sed -n '/^ verify:/,/^ publish:/p' "$release") publish_job=$(sed -n '/^ publish:/,$p' "$release") @@ -46,14 +56,16 @@ publish_job=$(sed -n '/^ publish:/,$p' "$release") [[ "$publish_job" == *'gh release upload "$RELEASE_TAG"'* ]] [[ "$publish_job" == *'gh release edit "$RELEASE_TAG"'* ]] [[ "$publish_job" == *'context: target/container-runtime'* ]] -for gate in 'pixi run test-single-node-container' 'test-boto3-e2e' 'test-pyiceberg-e2e' \ - 'pixi run test-console' 'pixi run test-console-ui' 'pixi run rs-fmt-check && pixi run rs-lint'; do +for gate in 'pixi run test-single-node-container' 'Require CI success for the release commit'; do [[ "$verify_job" == *"$gate"* ]] done +[[ "$verify_job" != *'git push origin'* && "$verify_job" != *'gh release create'* ]] ! grep -Eq 'DOCKERHUB_|push: true|id-token: write' <<<"$verify_job" [[ "$publish_job" == *'needs: verify'* && "$publish_job" == *'environment: DockerHub'* ]] [[ "$publish_job" != *'RELEASE_ENABLED'* ]] [[ "$publish_job" == *'[[ "$(git rev-parse HEAD)" == "$REVISION" ]]'* ]] +[[ "$publish_job" == *'Create release tag after verification'* ]] +grep -Fq 'org.crowdb.runtime.sha256="$RUNTIME_SHA256"' container/single-node-container/Dockerfile preview_events=$(sed -n '/^on:/,/^jobs:/p' "$preview") [[ "$preview_events" == *'workflow_dispatch:'* ]] @@ -64,7 +76,8 @@ preview_job=$(sed -n '/^ DockerPreview:/,$p' "$preview") ! grep -Eq 'secrets\.|docker/login-action|docker/build-push-action' <<<"$preview_job" release_tool=tools/release.py -for required in '--dry-run' '--execute' '--symbols' 'git", "push", "--atomic"' \ - '"release", "create"' '"workflow", "run"'; do +for required in '--dry-run' '--execute' '--symbols' '"push", "origin", "HEAD:refs/heads/main"' \ + '"workflow", "run"' '"--ref", "main"'; do grep -Fq -- "$required" "$release_tool" done +! grep -Eq '"release", "create"|"tag", "-a"' "$release_tool" diff --git a/tools/pixi-tasks/test-iceberg-rck.sh b/tools/pixi-tasks/test-iceberg-rck.sh index eb94ed90..7ad9a7a7 100644 --- a/tools/pixi-tasks/test-iceberg-rck.sh +++ b/tools/pixi-tasks/test-iceberg-rck.sh @@ -6,15 +6,20 @@ cd "${PIXI_PROJECT_ROOT:?}" source tools/pixi-tasks/prepare-iceberg.sh revision=6976e020b894f6a6777704df2b8c4458cb291ae9 -export CROWDB_ICEBERG_RCK_ROOT="${CROWDB_ICEBERG_RCK_ROOT:-$PIXI_PROJECT_ROOT/target/iceberg-rck}" +export CROWDB_ICEBERG_RCK_ROOT="${CROWDB_ICEBERG_RCK_ROOT:-${RUNNER_TEMP:-$PIXI_PROJECT_ROOT/target}/iceberg-rck}" if [[ ! -e "$CROWDB_ICEBERG_RCK_ROOT" ]]; then mkdir -p "$CROWDB_ICEBERG_RCK_ROOT" git -C "$CROWDB_ICEBERG_RCK_ROOT" init git -C "$CROWDB_ICEBERG_RCK_ROOT" fetch --depth 1 https://github.com/apache/iceberg.git "$revision" git -C "$CROWDB_ICEBERG_RCK_ROOT" checkout --detach FETCH_HEAD fi -[[ "$(git -C "$CROWDB_ICEBERG_RCK_ROOT" rev-parse HEAD)" == "$revision" ]] || { - echo 'Iceberg RCK requires the pinned source revision; existing checkout was preserved' >&2 +[[ -d "$CROWDB_ICEBERG_RCK_ROOT/.git" ]] || { + echo "Iceberg RCK source directory is not a checkout: $CROWDB_ICEBERG_RCK_ROOT" >&2 + exit 1 +} +actual=$(git -C "$CROWDB_ICEBERG_RCK_ROOT" rev-parse HEAD) +[[ "$actual" == "$revision" ]] || { + echo "Iceberg RCK requires $revision; found $actual at $CROWDB_ICEBERG_RCK_ROOT (checkout preserved)" >&2 exit 1 } pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e \ diff --git a/tools/release.py b/tools/release.py index e5ca7f52..0ab079bf 100644 --- a/tools/release.py +++ b/tools/release.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 # Copyright 2026-present Gian # Licensed under the Apache License, Version 2.0. -"""Prepare a versioned release and dispatch the verified container workflow. +"""Prepare a versioned candidate and dispatch the verified container workflow. Run through Pixi: pixi run -- python tools/release.py --dry-run pixi run -- python tools/release.py --execute @@ -110,8 +110,8 @@ def main() -> None: print(f"Release {current} -> {target} ({tag})", flush=True) for path in updates: print(f" update {path.relative_to(ROOT)}", flush=True) - print(" check versions and diff; commit; tag; atomically push main + tag", flush=True) - print(f" create draft GitHub Release; dispatch release-container.yml (symbols: {args.symbols})", flush=True) + print(" check versions and diff; commit and push the candidate to main", flush=True) + print(f" dispatch release-container.yml (symbols: {args.symbols}); tag after verification", flush=True) if args.dry_run: for path, updated in updates.items(): original = path.read_text(encoding="utf-8") @@ -129,12 +129,12 @@ def main() -> None: command("git", "diff", "--check") command("git", "add", *(str(path.relative_to(ROOT)) for path in updates)) command("git", "commit", "-m", f"Release {target}") - command("git", "tag", "-a", tag, "-m", tag) - command("git", "push", "--atomic", "origin", "HEAD:refs/heads/main", f"refs/tags/{tag}") - command("gh", "release", "create", tag, "--repo", REPO, "--verify-tag", "--generate-notes", "--draft") + command("git", "push", "origin", "HEAD:refs/heads/main") + revision = command("git", "rev-parse", "HEAD", capture=True) command("gh", "workflow", "run", "release-container.yml", "--repo", REPO, - "--ref", tag, "-f", f"tag={tag}", "-f", f"include_symbols={str(args.symbols).lower()}") - print(f"Started verified publication for {tag}") + "--ref", "main", "-f", f"candidate_sha={revision}", + "-f", f"include_symbols={str(args.symbols).lower()}") + print(f"Started candidate verification for {tag}") if __name__ == "__main__":